@inproceedings{bb246200,
        AUTHOR = "Wu, P.H. and Xie, S.",
        TITLE = "V*: Guided Visual Search as a Core Mechanism in Multimodal LLMs",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13084-13094",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241106"}

@inproceedings{bb246201,
        AUTHOR = "He, R. and Cascante Bonilla, P. and Yang, Z.Y. and Berg, A.C. and Ordonez, V.",
        TITLE = "Improved Visual Grounding through Self-Consistent Explanations",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13095-13105",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241107"}

@inproceedings{bb246202,
        AUTHOR = "Feng, C. and Hsu, J. and Liu, W.Y. and Wu, J.J.",
        TITLE = "Naturally Supervised 3D Visual Grounding with Language-Regularized
Concept Learners",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13269-13278",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241108"}

@inproceedings{bb246203,
        AUTHOR = "He, J.W. and Wang, Y.F. and Wang, L.J. and Lu, H.C. and He, J.Y. and Lan, J.P. and Luo, B. and Xie, X.",
        TITLE = "Multi-Modal Instruction Tuned LLMs with Fine-Grained Visual
Perception",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13980-13990",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241109"}

@inproceedings{bb246204,
        AUTHOR = "Yuan, Z.H. and Ren, J. and Feng, C.M. and Zhao, H.S. and Cui, S.G. and Li, Z.",
        TITLE = "Visual Programming for Zero-Shot Open-Vocabulary 3D Visual Grounding",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "20623-20633",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241110"}

@inproceedings{bb246205,
        AUTHOR = "Chen, G. and Shen, L. and Shao, R. and Deng, X. and Nie, L.Q.",
        TITLE = "LION: Empowering Multimodal Large Language Model with Dual-Level
Visual Knowledge",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "26530-26540",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241111"}

@inproceedings{bb246206,
        AUTHOR = "Qu, M.X. and Chen, X.D. and Liu, W. and Li, A. and Zhao, Y.",
        TITLE = "ChatVTG: Video Temporal Grounding via Chat with Video Dialogue Large
Language Models",
        BOOKTITLE = PVUW24,
        YEAR = "2024",
        PAGES = "1847-1856",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241112"}

@inproceedings{bb246207,
        AUTHOR = "Zhang, Y. and Ma, Z.Q. and Gao, X.F. and Shakiah, S. and Gao, Q. and Chai, J.",
        TITLE = "Groundhog Grounding Large Language Models to Holistic Segmentation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14227-14238",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241113"}

@inproceedings{bb246208,
        AUTHOR = "Kim, K. and Yoon, K. and Jeon, J. and In, Y. and Moon, J. and Kim, D.H. and Park, C.",
        TITLE = "LLM4SGG: Large Language Models for Weakly Supervised Scene Graph
Generation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "28306-28316",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803llmgr4.html#TT241114"}

@article{bb246209,
        AUTHOR = "Liang, J.W. and Jiang, L. and Cao, L.L. and Kalantidis, Y. and Li, L.J. and Hauptmann, A.G.",
        TITLE = "Focal Visual-Text Attention for Memex Question Answering",
        JOURNAL = PAMI,
        VOLUME = "41",
        YEAR = "2019",
        NUMBER = "8",
        MONTH = "August",
        PAGES = "1893-1908",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241116"}

@inproceedings{bb246210,
        AUTHOR = "Liang, J.W. and Jiang, L. and Cao, L.L. and Li, L.J. and Hauptmann, A.G.",
        TITLE = "Focal Visual-Text Attention for Visual Question Answering",
        BOOKTITLE = CVPR18,
        YEAR = "2018",
        PAGES = "6135-6143",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241117"}

@article{bb246211,
        AUTHOR = "Riquelme, F. and de Goyeneche, A. and Zhang, Y.D. and Niebles, J.C. and Soto, A.",
        TITLE = "Explaining VQA predictions using visual grounding and a knowledge
base",
        JOURNAL = IVC,
        VOLUME = "101",
        YEAR = "2020",
        PAGES = "103968",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241118"}

@article{bb246212,
        AUTHOR = "Plummer, B.A. and Shih, K.J. and Li, Y.C. and Xu, K. and Lazebnik, S. and Sclaroff, S. and Saenko, K.",
        TITLE = "Revisiting Image-Language Networks for Open-Ended Phrase Detection",
        JOURNAL = PAMI,
        VOLUME = "44",
        YEAR = "2022",
        NUMBER = "4",
        MONTH = "April",
        PAGES = "2155-2167",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241119"}

@inproceedings{bb246213,
        AUTHOR = "Burns, A. and Tan, R. and Saenko, K. and Sclaroff, S. and Plummer, B.A.",
        TITLE = "Language Features Matter: Effective Language Representations for
Vision-Language Tasks",
        BOOKTITLE = ICCV19,
        YEAR = "2019",
        PAGES = "7473-7482",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241120"}

@inproceedings{bb246214,
        AUTHOR = "Arbelle, A. and Doveh, S. and Alfassy, A. and Shtok, J. and Lev, G. and Schwartz, E. and Kuehne, H. and Levi, H.B. and Sattigeri, P. and Panda, R. and Chen, C.F. and Bronstein, A.M. and Saenko, K. and Ullman, S. and Giryes, R. and Feris, R.S. and Karlinsky, L.",
        TITLE = "Detector-Free Weakly Supervised Grounding by Separation",
        BOOKTITLE = ICCV21,
        YEAR = "2021",
        PAGES = "1781-1792",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241121"}

@inproceedings{bb246215,
        AUTHOR = "Whitehead, S. and Wu, H. and Ji, H. and Feris, R.S. and Saenko, K.",
        TITLE = "Separating Skills and Concepts for Novel Visual Question Answering",
        BOOKTITLE = CVPR21,
        YEAR = "2021",
        PAGES = "5628-5637",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241122"}

@article{bb246216,
        AUTHOR = "Zhao, L.C. and Cai, D.G. and Zhang, J. and Sheng, L. and Xu, D. and Zheng, R. and Zhao, Y.J. and Wang, L.P. and Fan, X.",
        TITLE = "Toward Explainable 3D Grounded Visual Question Answering: A New
Benchmark and Strong Baseline",
        JOURNAL = CirSysVideo,
        VOLUME = "33",
        YEAR = "2023",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "2935-2949",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241123"}

@article{bb246217,
        AUTHOR = "Zhu, L.J. and Peng, L. and Zhou, W.N. and Yang, J.L.",
        TITLE = "Dual-decoder transformer network for answer grounding in visual
question answering",
        JOURNAL = PRL,
        VOLUME = "171",
        YEAR = "2023",
        PAGES = "53-60",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241124"}

@article{bb246218,
        AUTHOR = "Li, Y.C. and Wang, X. and Xiao, J.B. and Ji, W. and Chua, T.S.",
        TITLE = "Transformer-Empowered Invariant Grounding for Video Question
Answering",
        JOURNAL = PAMI,
        VOLUME = "47",
        YEAR = "2025",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "9510-9522",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241125"}

@inproceedings{bb246219,
        AUTHOR = "Li, Y.C. and Wang, X. and Xiao, J.B. and Ji, W. and Chua, T.S.",
        TITLE = "Invariant Grounding for Video Question Answering",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "2918-2927",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241126"}

@inproceedings{bb246220,
        AUTHOR = "Huang, J.Y. and Jia, B.X. and Wang, Y. and Zhu, Z.Y. and Linghu, X.K. and Li, Q. and Zhu, S.C. and Huang, S.Y.",
        TITLE = "Unveiling the Mist over 3D Vision-Language Understanding:
Object-centric Evaluation with Chain-of-Analysis",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "24570-24581",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241127"}

@inproceedings{bb246221,
        AUTHOR = "Chen, K. and Wu, X.Q.",
        TITLE = "VTQA: Visual Text Question Answering via Entity Alignment and
Cross-Media Reasoning",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "27208-27217",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241128"}

@inproceedings{bb246222,
        AUTHOR = "Di, S.Z. and Xie, W.",
        TITLE = "Grounded Question-Answering in Long Egocentric Videos",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "12934-12943",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241129"}

@inproceedings{bb246223,
        AUTHOR = "Chen, C.Y. and Anjum, S. and Gurari, D.",
        TITLE = "VQA Therapy: Exploring Answer Differences by Visually Grounding
Answers",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15269-15279",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241130"}

@inproceedings{bb246224,
        AUTHOR = "Le, T.M. and Le, V. and Gupta, S.I. and Venkatesh, S. and Tran, T.",
        TITLE = "Guiding Visual Question Answering with Attention Priors",
        BOOKTITLE = WACV23,
        YEAR = "2023",
        PAGES = "4370-4379",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241131"}

@inproceedings{bb246225,
        AUTHOR = "Khan, A.U. and Kuehne, H. and Gan, C. and da Vitoria Lobo, N. and Shah, M.",
        TITLE = "Weakly Supervised Grounding for VQA in Vision-Language Transformers",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXXV:652-670",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241132"}

@inproceedings{bb246226,
        AUTHOR = "Gupta, K. and Gautam, D. and Mamidi, R.",
        TITLE = "cViL: Cross-Lingual Training of Vision-Language Models using
Knowledge Distillation",
        BOOKTITLE = "ICPR22",
        YEAR = "2022",
        PAGES = "1734-1741",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241133"}

@inproceedings{bb246227,
        AUTHOR = "Lu, X.P. and Fan, Z. and Wang, Y. and Oh, J. and Rose, C.P.",
        TITLE = "Localize, Group, and Select: Boosting Text-VQA by Scene Text Modeling",
        BOOKTITLE = XSAnim21,
        YEAR = "2021",
        PAGES = "2631-2639",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241134"}

@inproceedings{bb246228,
        AUTHOR = "Khan, A.U. and Kuehne, H. and Duarte, K. and Gan, C. and Lobo, N. and Shah, M.",
        TITLE = "Found a Reason for me? Weakly-supervised Grounded Visual Question
Answering using Capsules",
        BOOKTITLE = CVPR21,
        YEAR = "2021",
        PAGES = "8461-8470",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241135"}

@inproceedings{bb246229,
        AUTHOR = "Selvaraju, R.R. and Tendulkar, P. and Parikh, D. and Horvitz, E. and Tulio Ribeiro, M. and Nushi, B. and Kamar, E.",
        TITLE = "SQuINTing at VQA Models: Introspecting VQA Models With Sub-Questions",
        BOOKTITLE = CVPR20,
        YEAR = "2020",
        PAGES = "10000-10008",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241136"}

@inproceedings{bb246230,
        AUTHOR = "Gouthaman, K.V. and Mittal, A.",
        TITLE = "Reducing Language Biases in Visual Question Answering with
Visually-grounded Question Encoder",
        BOOKTITLE = ECCV20,
        YEAR = "2020",
        PAGES = "XIII:18-34",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241137"}

@inproceedings{bb246231,
        AUTHOR = "Tan, H.L. and Leong, M.C. and Xu, Q. and Li, L. and Fang, F. and Cheng, Y. and Gauthier, N. and Sun, Y. and Lim, J.H.",
        TITLE = "Task-Oriented Multi-Modal Question Answering For Collaborative
Applications",
        BOOKTITLE = ICIP20,
        YEAR = "2020",
        PAGES = "1426-1430",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241138"}

@inproceedings{bb246232,
        AUTHOR = "Selvaraju, R.R. and Lee, S. and Shen, Y. and Jin, H. and Ghosh, S. and Heck, L. and Batra, D. and Parikh, D.",
        TITLE = "Taking a HINT: Leveraging Explanations to Make Vision and Language
Models More Grounded",
        BOOKTITLE = ICCV19,
        YEAR = "2019",
        PAGES = "2591-2600",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241139"}

@inproceedings{bb246233,
        AUTHOR = "Zhang, Y. and Niebles, J.C. and Soto, A.",
        TITLE = "Interpretable Visual Question Answering by Visual Grounding From
Attention Supervision Mining",
        BOOKTITLE = WACV19,
        YEAR = "2019",
        PAGES = "349-357",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT241140"}

@article{bb246234,
        AUTHOR = "Li, X. and Jiang, S.",
        TITLE = "Bundled Object Context for Referring Expressions",
        JOURNAL = MultMed,
        VOLUME = "20",
        YEAR = "2018",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "2749-2760",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241141"}

@article{bb246235,
        AUTHOR = "Wang, J.M. and Cui, E. and Liu, K.L. and Sun, Y.K. and Liang, J.Y. and Yuan, C.M. and Duan, X.J. and Jin, G.H. and Chung, T.S.",
        TITLE = "Referring expression comprehension model with matching detection and
linguistic feedback",
        JOURNAL = IET-CV,
        VOLUME = "14",
        YEAR = "2020",
        NUMBER = "8",
        MONTH = "December",
        PAGES = "625-633",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241142"}

@article{bb246236,
        AUTHOR = "Qiao, Y.Y. and Deng, C.R. and Wu, Q.",
        TITLE = "Referring Expression Comprehension: A Survey of Methods and Datasets",
        JOURNAL = MultMed,
        VOLUME = "23",
        YEAR = "2021",
        PAGES = "4426-4440",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241143"}

@article{bb246237,
        AUTHOR = "Niu, Y.L. and Zhang, H.W. and Lu, Z.W. and Chang, S.F.",
        TITLE = "Variational Context: Exploiting Visual and Textual Context for
Grounding Referring Expressions",
        JOURNAL = PAMI,
        VOLUME = "43",
        YEAR = "2021",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "347-359",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241144"}

@article{bb246238,
        AUTHOR = "Yang, S. and Li, G.B. and Yu, Y.Z.",
        TITLE = "Relationship-Embedded Representation Learning for Grounding Referring
Expressions",
        JOURNAL = PAMI,
        VOLUME = "43",
        YEAR = "2021",
        NUMBER = "8",
        MONTH = "August",
        PAGES = "2765-2779",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241145"}

@inproceedings{bb246239,
        AUTHOR = "Yang, S. and Li, G.B. and Yu, Y.Z.",
        TITLE = "Cross-Modal Relationship Inference for Grounding Referring Expressions",
        BOOKTITLE = CVPR19,
        YEAR = "2019",
        PAGES = "4140-4149",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241146"}

@article{bb246240,
        AUTHOR = "Sun, M.J. and Xiao, J. and Lim, E.G. and Liu, S. and Goulermas, J.Y.",
        TITLE = "Discriminative Triad Matching and Reconstruction for Weakly Referring
Expression Grounding",
        JOURNAL = PAMI,
        VOLUME = "43",
        YEAR = "2021",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "4189-4195",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241147"}

@article{bb246241,
        AUTHOR = "Lin, L. and Yan, P.X. and Xu, X.Q. and Yang, S. and Zeng, K. and Li, G.B.",
        TITLE = "Structured Attention Network for Referring Image Segmentation",
        JOURNAL = MultMed,
        VOLUME = "24",
        YEAR = "2022",
        PAGES = "1922-1932",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241148"}

@article{bb246242,
        AUTHOR = "Yang, X. and Wang, H. and Xie, D. and Deng, C. and Tao, D.C.",
        TITLE = "Object-Agnostic Transformers for Video Referring Segmentation",
        JOURNAL = IP,
        VOLUME = "31",
        YEAR = "2022",
        PAGES = "2839-2849",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241149"}

@article{bb246243,
        AUTHOR = "Wang, X. and Xie, D. and Zheng, Y.S.",
        TITLE = "Referring expression grounding by multi-context reasoning",
        JOURNAL = PRL,
        VOLUME = "160",
        YEAR = "2022",
        PAGES = "66-72",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241150"}

@article{bb246244,
        AUTHOR = "Shen, H.T. and Chen, C. and Wang, P. and Gao, L.L. and Wang, M. and Song, J.K.",
        TITLE = "Continual Referring Expression Comprehension via Dual Modular
Memorization",
        JOURNAL = IP,
        VOLUME = "31",
        YEAR = "2022",
        PAGES = "6694-6706",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241151"}

@article{bb246245,
        AUTHOR = "Chen, Y.W. and Tsai, Y.H. and Yang, M.H.",
        TITLE = "Understanding Synonymous Referring Expressions via Contrastive Features",
        JOURNAL = IJCV,
        VOLUME = "130",
        YEAR = "2022",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "2501-2516",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241152"}

@article{bb246246,
        AUTHOR = "Suo, W. and Sun, M.Y. and Wang, P. and Zhang, Y.N. and Wu, Q.",
        TITLE = "Rethinking and Improving Feature Pyramids for One-Stage Referring
Expression Comprehension",
        JOURNAL = IP,
        VOLUME = "32",
        YEAR = "2023",
        PAGES = "854-864",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241153"}

@article{bb246247,
        AUTHOR = "Liu, X.J. and Li, L. and Wang, S.H. and Zha, Z.J. and Li, Z.C. and Tian, Q. and Huang, Q.M.",
        TITLE = "Entity-Enhanced Adaptive Reconstruction Network for Weakly Supervised
Referring Expression Grounding",
        JOURNAL = PAMI,
        VOLUME = "45",
        YEAR = "2023",
        NUMBER = "3",
        MONTH = "March",
        PAGES = "3003-3018",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241154"}

@inproceedings{bb246248,
        AUTHOR = "Liu, X.J. and Li, L. and Wang, S.H. and Zha, Z.J. and Meng, D.C. and Huang, Q.M.",
        TITLE = "Adaptive Reconstruction Network for Weakly Supervised Referring
Expression Grounding",
        BOOKTITLE = ICCV19,
        YEAR = "2019",
        PAGES = "2611-2620",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241155"}

@article{bb246249,
        AUTHOR = "Feng, G. and Zhang, L. and Sun, J.Y. and Hu, Z.W. and Lu, H.C.",
        TITLE = "Referring Segmentation via Encoder-Fused Cross-Modal Attention
Network",
        JOURNAL = PAMI,
        VOLUME = "45",
        YEAR = "2023",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "7654-7667",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241156"}

@inproceedings{bb246250,
        AUTHOR = "Feng, G. and Hu, Z.W. and Zhang, L. and Lu, H.C.",
        TITLE = "Encoder Fusion Network with Co-Attention Embedding for Referring
Image Segmentation",
        BOOKTITLE = CVPR21,
        YEAR = "2021",
        PAGES = "15501-15510",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241157"}

@article{bb246251,
        AUTHOR = "Liu, D.Z. and Zhou, P. and Xu, Z. and Wang, H.Z. and Li, R.X.",
        TITLE = "Few-Shot Temporal Sentence Grounding via Memory-Guided Semantic
Learning",
        JOURNAL = CirSysVideo,
        VOLUME = "33",
        YEAR = "2023",
        NUMBER = "5",
        MONTH = "May",
        PAGES = "2491-2505",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241158"}

@article{bb246252,
        AUTHOR = "Sun, M.J. and Xiao, J. and Lim, E.G. and Zhao, Y.",
        TITLE = "Cycle-Free Weakly Referring Expression Grounding With Self-Paced
Learning",
        JOURNAL = MultMed,
        VOLUME = "25",
        YEAR = "2023",
        PAGES = "1611-1621",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241159"}

@article{bb246253,
        AUTHOR = "Sun, M.Y. and Suo, W. and Wang, P. and Zhang, Y.N. and Wu, Q.",
        TITLE = "A Proposal-Free One-Stage Framework for Referring Expression
Comprehension and Generation via Dense Cross-Attention",
        JOURNAL = MultMed,
        VOLUME = "25",
        YEAR = "2023",
        PAGES = "2446-2458",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241160"}

@article{bb246254,
        AUTHOR = "Sun, Y.F. and Zhang, Y. and Jiang, H. and Hu, Y.L. and Yin, B.C.",
        TITLE = "Multi-level attention for referring expression comprehension",
        JOURNAL = PRL,
        VOLUME = "172",
        YEAR = "2023",
        PAGES = "252-258",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241161"}

@article{bb246255,
        AUTHOR = "Wang, R. and Tang, Z. and Zhou, Q.L. and Liu, X.Q. and Hui, T.R. and Tan, Q. and Liu, S.",
        TITLE = "Unified Transformer with Isomorphic Branches for Natural Language
Tracking",
        JOURNAL = CirSysVideo,
        VOLUME = "33",
        YEAR = "2023",
        NUMBER = "9",
        MONTH = "September",
        PAGES = "4529-4541",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241162"}

@article{bb246256,
        AUTHOR = "Li, H. and Sun, M.J. and Xiao, J. and Lim, E.G. and Zhao, Y.",
        TITLE = "Fully and Weakly Supervised Referring Expression Segmentation With
End-to-End Learning",
        JOURNAL = CirSysVideo,
        VOLUME = "33",
        YEAR = "2023",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "5999-6012",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241163"}

@article{bb246257,
        AUTHOR = "Liu, C. and Jiang, X.D. and Ding, H.H.",
        TITLE = "Instance-Specific Feature Propagation for Referring Segmentation",
        JOURNAL = MultMed,
        VOLUME = "25",
        YEAR = "2023",
        PAGES = "3657-3667",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241164"}

@article{bb246258,
        AUTHOR = "Song, Y.Z. and Chen, Y.S. and Shuai, H.H.",
        TITLE = "Decoupling-Cooperative Framework for Referring Expression
Comprehension",
        JOURNAL = SPLetters,
        VOLUME = "30",
        YEAR = "2023",
        PAGES = "1542-1546",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241165"}

@article{bb246259,
        AUTHOR = "Hua, G.G. and Liao, M. and Tian, S. and Zhang, Y.H. and Zou, W.B.",
        TITLE = "Multiple Relational Learning Network for Joint Referring Expression
Comprehension and Segmentation",
        JOURNAL = MultMed,
        VOLUME = "25",
        YEAR = "2023",
        PAGES = "8805-8816",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241166"}

@article{bb246260,
        AUTHOR = "Wang, W.B. and Pagnucco, M. and Xu, C.P. and Song, Y.",
        TITLE = "InterREC: An Interpretable Method for Referring Expression
Comprehension",
        JOURNAL = MultMed,
        VOLUME = "25",
        YEAR = "2023",
        PAGES = "9330-9342",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241167"}

@article{bb246261,
        AUTHOR = "Ke, J.C. and Wang, J. and Chen, J.C. and Jhuo, I.H. and Lin, C.W. and Lin, Y.Y.",
        TITLE = "CLIPREC: Graph-Based Domain Adaptive Network for Zero-Shot Referring
Expression Comprehension",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "2480-2492",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241168"}

@article{bb246262,
        AUTHOR = "Ke, J.C. and Wang, J. and Wong, W.K. and Toomey, A. and Wen, J.",
        TITLE = "Graph-Based Group Division Network for Referring Expression
Comprehension",
        JOURNAL = CirSysVideo,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "6170-6183",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241169"}

@article{bb246263,
        AUTHOR = "Li, X.C. and Fan, B.Y. and Zhang, R.Z. and Zhao, K. and Guo, Z.H. and Zhao, Y.Q. and Li, R.",
        TITLE = "Inexactly Matched Referring Expression Comprehension With Rationale",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "3937-3950",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241170"}

@article{bb246264,
        AUTHOR = "Luo, G. and Zhou, Y.Y. and Sun, J. and Sun, X.S. and Ji, R.R.",
        TITLE = "A Survivor in the Era of Large-Scale Pretraining: An Empirical Study
of One-Stage Referring Expression Comprehension",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "3689-3700",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241171"}

@article{bb246265,
        AUTHOR = "Miao, P.H. and Su, W. and Wang, G.A. and Li, X.W. and Xi, L.",
        TITLE = "Self-Paced Multi-Grained Cross-Modal Interaction Modeling for
Referring Expression Comprehension",
        JOURNAL = IP,
        VOLUME = "33",
        YEAR = "2024",
        PAGES = "1497-1507",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241172"}

@article{bb246266,
        AUTHOR = "Liu, Z.T. and Xu, T.Y. and Song, X.N. and Wu, X.J.",
        TITLE = "Unified Referring Expression Generation for Bounding Boxes and
Segmentations",
        JOURNAL = SPLetters,
        VOLUME = "31",
        YEAR = "2024",
        PAGES = "636-640",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241173"}

@article{bb246267,
        AUTHOR = "Zhang, Y.J. and Li, Q.Z. and Pan, Y. and Zhao, X.G. and Tan, M.",
        TITLE = "Multi-Stage Image-Language Cross-Generative Fusion Network for
Video-Based Referring Expression Comprehension",
        JOURNAL = IP,
        VOLUME = "33",
        YEAR = "2024",
        PAGES = "3256-3270",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241174"}

@article{bb246268,
        AUTHOR = "Lu, M.C. and Li, R.F. and Feng, F.X. and Ma, Z.Y. and Wang, X.J.",
        TITLE = "LGR-NET: Language Guided Reasoning Network for Referring Expression
Comprehension",
        JOURNAL = CirSysVideo,
        VOLUME = "34",
        YEAR = "2024",
        NUMBER = "8",
        MONTH = "August",
        PAGES = "7771-7784",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241175"}

@article{bb246269,
        AUTHOR = "Yao, H.B. and Wang, L.P. and Cai, C.T. and Wang, W. and Zhang, Z. and Shang, X.B.",
        TITLE = "Language conditioned multi-scale visual attention networks for visual
grounding",
        JOURNAL = IVC,
        VOLUME = "150",
        YEAR = "2024",
        PAGES = "105242",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241176"}

@article{bb246270,
        AUTHOR = "Ji, Z. and Wu, J. and Wang, Y.D. and Yang, A.P. and Han, J.G.",
        TITLE = "Progressive Semantic Reconstruction Network for Weakly Supervised
Referring Expression Grounding",
        JOURNAL = CirSysVideo,
        VOLUME = "34",
        YEAR = "2024",
        NUMBER = "12",
        MONTH = "December",
        PAGES = "13058-13070",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241177"}

@article{bb246271,
        AUTHOR = "Wu, J. and Ji, Z. and Wang, Y.D. and Pang, Y.W. and Han, J.G.",
        TITLE = "Cyclic Pseudo-Label Generation and Refinement for Weakly Supervised
Referring Expression Grounding",
        JOURNAL = CirSysVideo,
        VOLUME = "36",
        YEAR = "2026",
        NUMBER = "5",
        MONTH = "May",
        PAGES = "5839-5851",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241178"}

@article{bb246272,
        AUTHOR = "Qiu, H.Q. and Wang, L.X. and Zhao, T. and Meng, F.M. and Wu, Q.B. and Li, H.L.",
        TITLE = "MCCE-REC: MLLM-Driven Cross-Modal Contrastive Entropy Model for
Zero-Shot Referring Expression Comprehension",
        JOURNAL = CirSysVideo,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "754-768",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241179"}

@article{bb246273,
        AUTHOR = "Ke, J.C. and Zhang, Q. and Wang, J. and Ding, H.Q. and Zhang, P.F. and Wen, J.",
        TITLE = "Graph-based referring expression comprehension with expression-guided
selective filtering and noun-oriented reasoning",
        JOURNAL = PR,
        VOLUME = "161",
        YEAR = "2025",
        PAGES = "111222",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241180"}

@article{bb246274,
        AUTHOR = "Ke, J.C. and Wang, D. and Chen, J.C. and Jhuo, I.H. and Lin, C.W. and Lin, Y.Y.",
        TITLE = "Make Graph-Based Referring Expression Comprehension Great Again
Through Expression-Guided Dynamic Gating and Regression",
        JOURNAL = MultMed,
        VOLUME = "27",
        YEAR = "2025",
        PAGES = "1950-1961",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241181"}

@article{bb246275,
        AUTHOR = "Huang, S.J. and Li, F. and Zhang, H. and Liu, S.L. and Zhang, L. and Wang, L.W.",
        TITLE = "A Mutual Supervision Framework for Referring Expression Segmentation
and Generation",
        JOURNAL = IJCV,
        VOLUME = "133",
        YEAR = "2025",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "3597-3612",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241182"}

@article{bb246276,
        AUTHOR = "Ke, X. and Xu, P.R. and Guo, W.Z.",
        TITLE = "Language-Image Consistency Augmentation and Distillation Network for
visual grounding",
        JOURNAL = PR,
        VOLUME = "166",
        YEAR = "2025",
        PAGES = "111663",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241183"}

@article{bb246277,
        AUTHOR = "Yang, X.Z. and Liu, J.Z. and Wang, P. and Wang, G.Q. and Yang, Y. and Shen, H.T.",
        TITLE = "New Dataset and Methods for Fine-Grained Compositional Referring
Expression Comprehension via Specialist-MLLM Collaboration",
        JOURNAL = PAMI,
        VOLUME = "47",
        YEAR = "2025",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "8598-8612",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241184"}

@article{bb246278,
        AUTHOR = "Guo, H. and Fan, W. and Wei, B. and Zhu, J.F. and Tian, J. and Yi, C.Z. and Jiang, F.",
        TITLE = "AD-DINO: Attention-Dynamic DINO for Distance-Aware Embodied Reference
Understanding",
        JOURNAL = CirSysVideo,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "10238-10249",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241185"}

@article{bb246279,
        AUTHOR = "Ke, J.C. and Wen, J. and Wang, H.T. and Cheng, W.H. and Wang, J.",
        TITLE = "Multi-Perspective Cross-Modal Object Encoding for Referring
Expression Comprehension",
        JOURNAL = IP,
        VOLUME = "34",
        YEAR = "2025",
        PAGES = "6911-6924",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241186"}

@article{bb246280,
        AUTHOR = "Li, J. and Wen, Z. and Zhang, Y. and Wang, W.X. and Cai, Y.X. and Zhang, T.X. and He, X.J. and Liu, J.",
        TITLE = "Generalized referring expression segmentation driven by
instance-oriented queries",
        JOURNAL = PR,
        VOLUME = "172",
        YEAR = "2026",
        PAGES = "112524",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241187"}

@article{bb246281,
        AUTHOR = "Liu, X.Y. and Liu, T. and Huang, S. and Xin, Y. and Hu, Y. and Qin, L. and Wang, D.L. and Wu, Y.Y. and Chen, H.G.",
        TITLE = "M2IST: Multi-Modal Interactive Side-Tuning for Efficient Referring
Expression Comprehension",
        JOURNAL = CirSysVideo,
        VOLUME = "36",
        YEAR = "2026",
        NUMBER = "2",
        MONTH = "February",
        PAGES = "1341-1354",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241188"}

@article{bb246282,
        AUTHOR = "Li, R.F. and Lu, M.C. and Lin, P.Y. and Yu, Z.H. and Ma, Z.Y.",
        TITLE = "Improving Scene Knowledge Referring Expression Comprehension With
Large Language Models",
        JOURNAL = MultMedMag,
        VOLUME = "33",
        YEAR = "2026",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "72-80",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241189"}

@article{bb246283,
        AUTHOR = "Zhang, Z. and Guan, Z. and Zhao, T.C. and Shen, H.Z. and Cai, Y.X. and Su, Z.G. and Shang, Y.H. and Liu, Z.J. and Yin, J.W. and Li, X.",
        TITLE = "Geo-R1: Improving few-shot geospatial referring expression
understanding with reinforcement fine-tuning",
        JOURNAL = PandRS,
        VOLUME = "237",
        YEAR = "2026",
        PAGES = "113-129",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241190"}

@article{bb246284,
        AUTHOR = "Cheng, W.X. and Dai, M. and Yang, W.K.",
        TITLE = "PLRVG: Progressive layer-wise refinement for visual grounding via
deep-to-shallow decoding",
        JOURNAL = PR,
        VOLUME = "179",
        YEAR = "2026",
        PAGES = "113555",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241191"}

@article{bb246285,
        AUTHOR = "Yang, F. and Zhu, Y. and Zhan, Y.F. and Zhao, H.Y. and Li, X. and Wang, Y.W. and Tang, M. and Ning, X. and Wang, J.Q.",
        TITLE = "Seg-LLaVA: Empowering pixel-level understanding with large vision
language model",
        JOURNAL = PR,
        VOLUME = "179",
        YEAR = "2026",
        PAGES = "113560",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241192"}

@article{bb246286,
        AUTHOR = "Wu, C.L. and Chen, Q. and Ji, J.Y. and Liu, Y.H. and Ma, Y.W. and Sun, X.S. and Cao, L.J.",
        TITLE = "3D-STMN++: Leveraging semantic proxies to enhance superpoint-text
matching for 3D Referring Expression Segmentation",
        JOURNAL = PR,
        VOLUME = "179",
        YEAR = "2026",
        PAGES = "113854",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241193"}

@article{bb246287,
        AUTHOR = "Wang, K.Y. and Wu, G. and Fu, X. and Wang, X. and Liu, K. and Lu, X. and Ge, C.J. and Zhai, W. and Zha, Z.J.",
        TITLE = "SkyFind: A Large-Scale Benchmark Unveiling Referring Expression
Comprehension for UAV",
        JOURNAL = PAMI,
        VOLUME = "48",
        YEAR = "2026",
        NUMBER = "8",
        MONTH = "August",
        PAGES = "9859-9875",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241194"}

@inproceedings{bb246288,
        AUTHOR = "Chen, J. and Wei, F.Y. and Zhao, J.J. and Song, S. and Wu, B.H. and Peng, Z.X. and Chan, S.H.G. and Zhang, H.Y.",
        TITLE = "Revisiting Referring Expression Comprehension Evaluation in the Era
of Large Multimodal Models",
        BOOKTITLE = "AIBench25",
        YEAR = "2025",
        PAGES = "513-524",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241195"}

@inproceedings{bb246289,
        AUTHOR = "Wang, Z.C. and Pan, Z.Y. and Peng, Z. and Cheng, J. and Xiao, L.W. and Jiang, W. and Cao, Z.G.",
        TITLE = "Exploring Contextual Attribute Density in Referring Expression
Counting",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "19587-19596",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241196"}

@inproceedings{bb246290,
        AUTHOR = "Chen, X. and Luo, Y.X. and Luo, G. and Ji, J.Y. and Ding, H.H. and Zhou, Y.",
        TITLE = "DViN: Dynamic Visual Routing Network for Weakly Supervised Referring
Expression Comprehension",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "14347-14357",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241197"}

@inproceedings{bb246291,
        AUTHOR = "Wang, S.J. and Kim, D. and Taalimi, A. and Sun, C. and Kuo, W.C.",
        TITLE = "Learning Visual Grounding from Generative Vision and Language Model",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "8057-8067",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241198"}

@inproceedings{bb246292,
        AUTHOR = "Wu, T.Y. and Huang, S.Y. and Wang, Y.C.A.F.",
        TITLE = "Data-Efficient 3D Visual Grounding via Order-Aware Referring",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "3107-3117",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241199"}

@inproceedings{bb246293,
        AUTHOR = "Chu, T.Y. and Lin, Y.X. and Huang, C.C. and Hua, K.L.",
        TITLE = "Enhancing Anchor-based Weakly Supervised Referring Expression
Comprehension with Cross-modality Attention",
        BOOKTITLE = ACCV24,
        YEAR = "2024",
        PAGES = "III: 131-147",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241200"}

@inproceedings{bb246294,
        AUTHOR = "Nag, S. and Goswami, K. and Karanam, S.",
        TITLE = "Safari: Adaptive Sequence Transformer for Weakly Supervised Referring
Expression Segmentation",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XLIV: 485-503",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241201"}

@inproceedings{bb246295,
        AUTHOR = "Dai, S.Y. and Liu, J. and Cheung, N.M.",
        TITLE = "Referring Expression Counting",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "16985-16995",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241202"}

@inproceedings{bb246296,
        AUTHOR = "Han, Z. and Zhu, F.R. and Lao, Q. and Jiang, H.",
        TITLE = "Zero-Shot Referring Expression Comprehension via Structural
Similarity Between Images and Captions",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "14364-14375",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241203"}

@inproceedings{bb246297,
        AUTHOR = "Su, W. and Miao, P.H. and Dou, H.Z. and Li, X.",
        TITLE = "ScanFormer: Referring Expression Comprehension by Iteratively
Scanning",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13449-13458",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241204"}

@inproceedings{bb246298,
        AUTHOR = "Yu, Z.H. and Li, R.",
        TITLE = "Revisiting Counterfactual Problems in Referring Expression
Comprehension",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "13438-13448",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241205"}

@inproceedings{bb246299,
        AUTHOR = "Li, X. and Qiu, K. and Wang, J.L. and Xu, X.H. and Singh, R. and Yamazaki, K. and Chen, H. and Huang, X.N. and Raj, B.",
        TITLE = "R^2-Bench: Benchmarking the Robustness of Referring Perception Models
Under Perturbations",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "IX: 211-230",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT241206"}

Last update:Aug 19, 2026 at 13:26:35