@article{bb248400,
AUTHOR = "Plummer, B.A. and Shih, K.J. and Li, Y.C. and Xu, K. and Lazebnik, S. and Sclaroff, S. and Saenko, K.",
TITLE = "Revisiting Image-Language Networks for Open-Ended Phrase Detection",
JOURNAL = PAMI,
VOLUME = "44",
YEAR = "2022",
NUMBER = "4",
MONTH = "April",
PAGES = "2155-2167",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243293"}
@inproceedings{bb248401,
AUTHOR = "Burns, A. and Tan, R. and Saenko, K. and Sclaroff, S. and Plummer, B.A.",
TITLE = "Language Features Matter: Effective Language Representations for
Vision-Language Tasks",
BOOKTITLE = ICCV19,
YEAR = "2019",
PAGES = "7473-7482",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243294"}
@inproceedings{bb248402,
AUTHOR = "Arbelle, A. and Doveh, S. and Alfassy, A. and Shtok, J. and Lev, G. and Schwartz, E. and Kuehne, H. and Levi, H.B. and Sattigeri, P. and Panda, R. and Chen, C.F. and Bronstein, A.M. and Saenko, K. and Ullman, S. and Giryes, R. and Feris, R.S. and Karlinsky, L.",
TITLE = "Detector-Free Weakly Supervised Grounding by Separation",
BOOKTITLE = ICCV21,
YEAR = "2021",
PAGES = "1781-1792",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243295"}
@inproceedings{bb248403,
AUTHOR = "Whitehead, S. and Wu, H. and Ji, H. and Feris, R.S. and Saenko, K.",
TITLE = "Separating Skills and Concepts for Novel Visual Question Answering",
BOOKTITLE = CVPR21,
YEAR = "2021",
PAGES = "5628-5637",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243296"}
@article{bb248404,
AUTHOR = "Zhao, L.C. and Cai, D.G. and Zhang, J. and Sheng, L. and Xu, D. and Zheng, R. and Zhao, Y.J. and Wang, L.P. and Fan, X.",
TITLE = "Toward Explainable 3D Grounded Visual Question Answering: A New
Benchmark and Strong Baseline",
JOURNAL = CirSysVideo,
VOLUME = "33",
YEAR = "2023",
NUMBER = "6",
MONTH = "June",
PAGES = "2935-2949",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243297"}
@article{bb248405,
AUTHOR = "Zhu, L.J. and Peng, L. and Zhou, W.N. and Yang, J.L.",
TITLE = "Dual-decoder transformer network for answer grounding in visual
question answering",
JOURNAL = PRL,
VOLUME = "171",
YEAR = "2023",
PAGES = "53-60",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243298"}
@article{bb248406,
AUTHOR = "Li, Y.C. and Wang, X. and Xiao, J.B. and Ji, W. and Chua, T.S.",
TITLE = "Transformer-Empowered Invariant Grounding for Video Question
Answering",
JOURNAL = PAMI,
VOLUME = "47",
YEAR = "2025",
NUMBER = "11",
MONTH = "November",
PAGES = "9510-9522",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243299"}
@inproceedings{bb248407,
AUTHOR = "Li, Y.C. and Wang, X. and Xiao, J.B. and Ji, W. and Chua, T.S.",
TITLE = "Invariant Grounding for Video Question Answering",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "2918-2927",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243300"}
@article{bb248408,
AUTHOR = "Zhu, L. and Pan, L.M. and Tan, S. and Zhang, C.Y. and Liu, D. and Wu, L.Y.B. and Boussaid, F. and Bennamoun, M.",
TITLE = "HAViG: Hierarchical adaptive visual grounding framework for video
question answering",
JOURNAL = PR,
VOLUME = "180",
YEAR = "2026",
PAGES = "114178",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243301"}
@inproceedings{bb248409,
AUTHOR = "Huang, J.Y. and Jia, B.X. and Wang, Y. and Zhu, Z.Y. and Linghu, X.K. and Li, Q. and Zhu, S.C. and Huang, S.Y.",
TITLE = "Unveiling the Mist over 3D Vision-Language Understanding:
Object-centric Evaluation with Chain-of-Analysis",
BOOKTITLE = CVPR25,
YEAR = "2025",
PAGES = "24570-24581",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243302"}
@inproceedings{bb248410,
AUTHOR = "Chen, K. and Wu, X.Q.",
TITLE = "VTQA: Visual Text Question Answering via Entity Alignment and
Cross-Media Reasoning",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "27208-27217",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243303"}
@inproceedings{bb248411,
AUTHOR = "Di, S.Z. and Xie, W.",
TITLE = "Grounded Question-Answering in Long Egocentric Videos",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "12934-12943",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243304"}
@inproceedings{bb248412,
AUTHOR = "Chen, C.Y. and Anjum, S. and Gurari, D.",
TITLE = "VQA Therapy: Exploring Answer Differences by Visually Grounding
Answers",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "15269-15279",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243305"}
@inproceedings{bb248413,
AUTHOR = "Le, T.M. and Le, V. and Gupta, S.I. and Venkatesh, S. and Tran, T.",
TITLE = "Guiding Visual Question Answering with Attention Priors",
BOOKTITLE = WACV23,
YEAR = "2023",
PAGES = "4370-4379",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243306"}
@inproceedings{bb248414,
AUTHOR = "Khan, A.U. and Kuehne, H. and Gan, C. and da Vitoria Lobo, N. and Shah, M.",
TITLE = "Weakly Supervised Grounding for VQA in Vision-Language Transformers",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "XXXV:652-670",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243307"}
@inproceedings{bb248415,
AUTHOR = "Gupta, K. and Gautam, D. and Mamidi, R.",
TITLE = "cViL: Cross-Lingual Training of Vision-Language Models using
Knowledge Distillation",
BOOKTITLE = "ICPR22",
YEAR = "2022",
PAGES = "1734-1741",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243308"}
@inproceedings{bb248416,
AUTHOR = "Lu, X.P. and Fan, Z. and Wang, Y. and Oh, J. and Rose, C.P.",
TITLE = "Localize, Group, and Select: Boosting Text-VQA by Scene Text Modeling",
BOOKTITLE = XSAnim21,
YEAR = "2021",
PAGES = "2631-2639",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243309"}
@inproceedings{bb248417,
AUTHOR = "Khan, A.U. and Kuehne, H. and Duarte, K. and Gan, C. and Lobo, N. and Shah, M.",
TITLE = "Found a Reason for me? Weakly-supervised Grounded Visual Question
Answering using Capsules",
BOOKTITLE = CVPR21,
YEAR = "2021",
PAGES = "8461-8470",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243310"}
@inproceedings{bb248418,
AUTHOR = "Selvaraju, R.R. and Tendulkar, P. and Parikh, D. and Horvitz, E. and Tulio Ribeiro, M. and Nushi, B. and Kamar, E.",
TITLE = "SQuINTing at VQA Models: Introspecting VQA Models With Sub-Questions",
BOOKTITLE = CVPR20,
YEAR = "2020",
PAGES = "10000-10008",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243311"}
@inproceedings{bb248419,
AUTHOR = "Gouthaman, K.V. and Mittal, A.",
TITLE = "Reducing Language Biases in Visual Question Answering with
Visually-grounded Question Encoder",
BOOKTITLE = ECCV20,
YEAR = "2020",
PAGES = "XIII:18-34",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243312"}
@inproceedings{bb248420,
AUTHOR = "Tan, H.L. and Leong, M.C. and Xu, Q. and Li, L. and Fang, F. and Cheng, Y. and Gauthier, N. and Sun, Y. and Lim, J.H.",
TITLE = "Task-Oriented Multi-Modal Question Answering For Collaborative
Applications",
BOOKTITLE = ICIP20,
YEAR = "2020",
PAGES = "1426-1430",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243313"}
@inproceedings{bb248421,
AUTHOR = "Selvaraju, R.R. and Lee, S. and Shen, Y. and Jin, H. and Ghosh, S. and Heck, L. and Batra, D. and Parikh, D.",
TITLE = "Taking a HINT: Leveraging Explanations to Make Vision and Language
Models More Grounded",
BOOKTITLE = ICCV19,
YEAR = "2019",
PAGES = "2591-2600",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243314"}
@inproceedings{bb248422,
AUTHOR = "Zhang, Y. and Niebles, J.C. and Soto, A.",
TITLE = "Interpretable Visual Question Answering by Visual Grounding From
Attention Supervision Mining",
BOOKTITLE = WACV19,
YEAR = "2019",
PAGES = "349-357",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vgrqa3.html#TT243315"}
@article{bb248423,
AUTHOR = "Li, X. and Jiang, S.",
TITLE = "Bundled Object Context for Referring Expressions",
JOURNAL = MultMed,
VOLUME = "20",
YEAR = "2018",
NUMBER = "10",
MONTH = "October",
PAGES = "2749-2760",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243316"}
@article{bb248424,
AUTHOR = "Wang, J.M. and Cui, E. and Liu, K.L. and Sun, Y.K. and Liang, J.Y. and Yuan, C.M. and Duan, X.J. and Jin, G.H. and Chung, T.S.",
TITLE = "Referring expression comprehension model with matching detection and
linguistic feedback",
JOURNAL = IET-CV,
VOLUME = "14",
YEAR = "2020",
NUMBER = "8",
MONTH = "December",
PAGES = "625-633",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243317"}
@article{bb248425,
AUTHOR = "Qiao, Y.Y. and Deng, C.R. and Wu, Q.",
TITLE = "Referring Expression Comprehension: A Survey of Methods and Datasets",
JOURNAL = MultMed,
VOLUME = "23",
YEAR = "2021",
PAGES = "4426-4440",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243318"}
@article{bb248426,
AUTHOR = "Niu, Y.L. and Zhang, H.W. and Lu, Z.W. and Chang, S.F.",
TITLE = "Variational Context: Exploiting Visual and Textual Context for
Grounding Referring Expressions",
JOURNAL = PAMI,
VOLUME = "43",
YEAR = "2021",
NUMBER = "1",
MONTH = "January",
PAGES = "347-359",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243319"}
@article{bb248427,
AUTHOR = "Yang, S. and Li, G.B. and Yu, Y.Z.",
TITLE = "Relationship-Embedded Representation Learning for Grounding Referring
Expressions",
JOURNAL = PAMI,
VOLUME = "43",
YEAR = "2021",
NUMBER = "8",
MONTH = "August",
PAGES = "2765-2779",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243320"}
@inproceedings{bb248428,
AUTHOR = "Yang, S. and Li, G.B. and Yu, Y.Z.",
TITLE = "Cross-Modal Relationship Inference for Grounding Referring Expressions",
BOOKTITLE = CVPR19,
YEAR = "2019",
PAGES = "4140-4149",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243321"}
@article{bb248429,
AUTHOR = "Sun, M.J. and Xiao, J. and Lim, E.G. and Liu, S. and Goulermas, J.Y.",
TITLE = "Discriminative Triad Matching and Reconstruction for Weakly Referring
Expression Grounding",
JOURNAL = PAMI,
VOLUME = "43",
YEAR = "2021",
NUMBER = "11",
MONTH = "November",
PAGES = "4189-4195",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243322"}
@article{bb248430,
AUTHOR = "Lin, L. and Yan, P.X. and Xu, X.Q. and Yang, S. and Zeng, K. and Li, G.B.",
TITLE = "Structured Attention Network for Referring Image Segmentation",
JOURNAL = MultMed,
VOLUME = "24",
YEAR = "2022",
PAGES = "1922-1932",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243323"}
@article{bb248431,
AUTHOR = "Yang, X. and Wang, H. and Xie, D. and Deng, C. and Tao, D.C.",
TITLE = "Object-Agnostic Transformers for Video Referring Segmentation",
JOURNAL = IP,
VOLUME = "31",
YEAR = "2022",
PAGES = "2839-2849",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243324"}
@article{bb248432,
AUTHOR = "Wang, X. and Xie, D. and Zheng, Y.S.",
TITLE = "Referring expression grounding by multi-context reasoning",
JOURNAL = PRL,
VOLUME = "160",
YEAR = "2022",
PAGES = "66-72",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243325"}
@article{bb248433,
AUTHOR = "Shen, H.T. and Chen, C. and Wang, P. and Gao, L.L. and Wang, M. and Song, J.K.",
TITLE = "Continual Referring Expression Comprehension via Dual Modular
Memorization",
JOURNAL = IP,
VOLUME = "31",
YEAR = "2022",
PAGES = "6694-6706",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243326"}
@article{bb248434,
AUTHOR = "Chen, Y.W. and Tsai, Y.H. and Yang, M.H.",
TITLE = "Understanding Synonymous Referring Expressions via Contrastive Features",
JOURNAL = IJCV,
VOLUME = "130",
YEAR = "2022",
NUMBER = "10",
MONTH = "October",
PAGES = "2501-2516",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243327"}
@article{bb248435,
AUTHOR = "Suo, W. and Sun, M.Y. and Wang, P. and Zhang, Y.N. and Wu, Q.",
TITLE = "Rethinking and Improving Feature Pyramids for One-Stage Referring
Expression Comprehension",
JOURNAL = IP,
VOLUME = "32",
YEAR = "2023",
PAGES = "854-864",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243328"}
@article{bb248436,
AUTHOR = "Liu, X.J. and Li, L. and Wang, S.H. and Zha, Z.J. and Li, Z.C. and Tian, Q. and Huang, Q.M.",
TITLE = "Entity-Enhanced Adaptive Reconstruction Network for Weakly Supervised
Referring Expression Grounding",
JOURNAL = PAMI,
VOLUME = "45",
YEAR = "2023",
NUMBER = "3",
MONTH = "March",
PAGES = "3003-3018",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243329"}
@inproceedings{bb248437,
AUTHOR = "Liu, X.J. and Li, L. and Wang, S.H. and Zha, Z.J. and Meng, D.C. and Huang, Q.M.",
TITLE = "Adaptive Reconstruction Network for Weakly Supervised Referring
Expression Grounding",
BOOKTITLE = ICCV19,
YEAR = "2019",
PAGES = "2611-2620",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243330"}
@article{bb248438,
AUTHOR = "Feng, G. and Zhang, L. and Sun, J.Y. and Hu, Z.W. and Lu, H.C.",
TITLE = "Referring Segmentation via Encoder-Fused Cross-Modal Attention
Network",
JOURNAL = PAMI,
VOLUME = "45",
YEAR = "2023",
NUMBER = "6",
MONTH = "June",
PAGES = "7654-7667",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243331"}
@inproceedings{bb248439,
AUTHOR = "Feng, G. and Hu, Z.W. and Zhang, L. and Lu, H.C.",
TITLE = "Encoder Fusion Network with Co-Attention Embedding for Referring
Image Segmentation",
BOOKTITLE = CVPR21,
YEAR = "2021",
PAGES = "15501-15510",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243332"}
@article{bb248440,
AUTHOR = "Liu, D.Z. and Zhou, P. and Xu, Z. and Wang, H.Z. and Li, R.X.",
TITLE = "Few-Shot Temporal Sentence Grounding via Memory-Guided Semantic
Learning",
JOURNAL = CirSysVideo,
VOLUME = "33",
YEAR = "2023",
NUMBER = "5",
MONTH = "May",
PAGES = "2491-2505",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243333"}
@article{bb248441,
AUTHOR = "Sun, M.J. and Xiao, J. and Lim, E.G. and Zhao, Y.",
TITLE = "Cycle-Free Weakly Referring Expression Grounding With Self-Paced
Learning",
JOURNAL = MultMed,
VOLUME = "25",
YEAR = "2023",
PAGES = "1611-1621",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243334"}
@article{bb248442,
AUTHOR = "Sun, M.Y. and Suo, W. and Wang, P. and Zhang, Y.N. and Wu, Q.",
TITLE = "A Proposal-Free One-Stage Framework for Referring Expression
Comprehension and Generation via Dense Cross-Attention",
JOURNAL = MultMed,
VOLUME = "25",
YEAR = "2023",
PAGES = "2446-2458",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243335"}
@article{bb248443,
AUTHOR = "Sun, Y.F. and Zhang, Y. and Jiang, H. and Hu, Y.L. and Yin, B.C.",
TITLE = "Multi-level attention for referring expression comprehension",
JOURNAL = PRL,
VOLUME = "172",
YEAR = "2023",
PAGES = "252-258",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243336"}
@article{bb248444,
AUTHOR = "Wang, R. and Tang, Z. and Zhou, Q.L. and Liu, X.Q. and Hui, T.R. and Tan, Q. and Liu, S.",
TITLE = "Unified Transformer with Isomorphic Branches for Natural Language
Tracking",
JOURNAL = CirSysVideo,
VOLUME = "33",
YEAR = "2023",
NUMBER = "9",
MONTH = "September",
PAGES = "4529-4541",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243337"}
@article{bb248445,
AUTHOR = "Li, H. and Sun, M.J. and Xiao, J. and Lim, E.G. and Zhao, Y.",
TITLE = "Fully and Weakly Supervised Referring Expression Segmentation With
End-to-End Learning",
JOURNAL = CirSysVideo,
VOLUME = "33",
YEAR = "2023",
NUMBER = "10",
MONTH = "October",
PAGES = "5999-6012",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243338"}
@article{bb248446,
AUTHOR = "Liu, C. and Jiang, X.D. and Ding, H.H.",
TITLE = "Instance-Specific Feature Propagation for Referring Segmentation",
JOURNAL = MultMed,
VOLUME = "25",
YEAR = "2023",
PAGES = "3657-3667",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243339"}
@article{bb248447,
AUTHOR = "Song, Y.Z. and Chen, Y.S. and Shuai, H.H.",
TITLE = "Decoupling-Cooperative Framework for Referring Expression
Comprehension",
JOURNAL = SPLetters,
VOLUME = "30",
YEAR = "2023",
PAGES = "1542-1546",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243340"}
@article{bb248448,
AUTHOR = "Hua, G.G. and Liao, M. and Tian, S. and Zhang, Y.H. and Zou, W.B.",
TITLE = "Multiple Relational Learning Network for Joint Referring Expression
Comprehension and Segmentation",
JOURNAL = MultMed,
VOLUME = "25",
YEAR = "2023",
PAGES = "8805-8816",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243341"}
@article{bb248449,
AUTHOR = "Wang, W.B. and Pagnucco, M. and Xu, C.P. and Song, Y.",
TITLE = "InterREC: An Interpretable Method for Referring Expression
Comprehension",
JOURNAL = MultMed,
VOLUME = "25",
YEAR = "2023",
PAGES = "9330-9342",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243342"}
@article{bb248450,
AUTHOR = "Ke, J.C. and Wang, J. and Chen, J.C. and Jhuo, I.H. and Lin, C.W. and Lin, Y.Y.",
TITLE = "CLIPREC: Graph-Based Domain Adaptive Network for Zero-Shot Referring
Expression Comprehension",
JOURNAL = MultMed,
VOLUME = "26",
YEAR = "2024",
PAGES = "2480-2492",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243343"}
@article{bb248451,
AUTHOR = "Ke, J.C. and Wang, J. and Wong, W.K. and Toomey, A. and Wen, J.",
TITLE = "Graph-Based Group Division Network for Referring Expression
Comprehension",
JOURNAL = CirSysVideo,
VOLUME = "35",
YEAR = "2025",
NUMBER = "6",
MONTH = "June",
PAGES = "6170-6183",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243344"}
@article{bb248452,
AUTHOR = "Li, X.C. and Fan, B.Y. and Zhang, R.Z. and Zhao, K. and Guo, Z.H. and Zhao, Y.Q. and Li, R.",
TITLE = "Inexactly Matched Referring Expression Comprehension With Rationale",
JOURNAL = MultMed,
VOLUME = "26",
YEAR = "2024",
PAGES = "3937-3950",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243345"}
@article{bb248453,
AUTHOR = "Luo, G. and Zhou, Y.Y. and Sun, J. and Sun, X.S. and Ji, R.R.",
TITLE = "A Survivor in the Era of Large-Scale Pretraining: An Empirical Study
of One-Stage Referring Expression Comprehension",
JOURNAL = MultMed,
VOLUME = "26",
YEAR = "2024",
PAGES = "3689-3700",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243346"}
@article{bb248454,
AUTHOR = "Miao, P.H. and Su, W. and Wang, G.A. and Li, X.W. and Xi, L.",
TITLE = "Self-Paced Multi-Grained Cross-Modal Interaction Modeling for
Referring Expression Comprehension",
JOURNAL = IP,
VOLUME = "33",
YEAR = "2024",
PAGES = "1497-1507",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243347"}
@article{bb248455,
AUTHOR = "Liu, Z.T. and Xu, T.Y. and Song, X.N. and Wu, X.J.",
TITLE = "Unified Referring Expression Generation for Bounding Boxes and
Segmentations",
JOURNAL = SPLetters,
VOLUME = "31",
YEAR = "2024",
PAGES = "636-640",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243348"}
@article{bb248456,
AUTHOR = "Zhang, Y.J. and Li, Q.Z. and Pan, Y. and Zhao, X.G. and Tan, M.",
TITLE = "Multi-Stage Image-Language Cross-Generative Fusion Network for
Video-Based Referring Expression Comprehension",
JOURNAL = IP,
VOLUME = "33",
YEAR = "2024",
PAGES = "3256-3270",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243349"}
@article{bb248457,
AUTHOR = "Lu, M.C. and Li, R.F. and Feng, F.X. and Ma, Z.Y. and Wang, X.J.",
TITLE = "LGR-NET: Language Guided Reasoning Network for Referring Expression
Comprehension",
JOURNAL = CirSysVideo,
VOLUME = "34",
YEAR = "2024",
NUMBER = "8",
MONTH = "August",
PAGES = "7771-7784",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243350"}
@article{bb248458,
AUTHOR = "Yao, H.B. and Wang, L.P. and Cai, C.T. and Wang, W. and Zhang, Z. and Shang, X.B.",
TITLE = "Language conditioned multi-scale visual attention networks for visual
grounding",
JOURNAL = IVC,
VOLUME = "150",
YEAR = "2024",
PAGES = "105242",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243351"}
@article{bb248459,
AUTHOR = "Ji, Z. and Wu, J. and Wang, Y.D. and Yang, A.P. and Han, J.G.",
TITLE = "Progressive Semantic Reconstruction Network for Weakly Supervised
Referring Expression Grounding",
JOURNAL = CirSysVideo,
VOLUME = "34",
YEAR = "2024",
NUMBER = "12",
MONTH = "December",
PAGES = "13058-13070",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243352"}
@article{bb248460,
AUTHOR = "Wu, J. and Ji, Z. and Wang, Y.D. and Pang, Y.W. and Han, J.G.",
TITLE = "Cyclic Pseudo-Label Generation and Refinement for Weakly Supervised
Referring Expression Grounding",
JOURNAL = CirSysVideo,
VOLUME = "36",
YEAR = "2026",
NUMBER = "5",
MONTH = "May",
PAGES = "5839-5851",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243353"}
@article{bb248461,
AUTHOR = "Qiu, H.Q. and Wang, L.X. and Zhao, T. and Meng, F.M. and Wu, Q.B. and Li, H.L.",
TITLE = "MCCE-REC: MLLM-Driven Cross-Modal Contrastive Entropy Model for
Zero-Shot Referring Expression Comprehension",
JOURNAL = CirSysVideo,
VOLUME = "35",
YEAR = "2025",
NUMBER = "1",
MONTH = "January",
PAGES = "754-768",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243354"}
@article{bb248462,
AUTHOR = "Ke, J.C. and Zhang, Q. and Wang, J. and Ding, H.Q. and Zhang, P.F. and Wen, J.",
TITLE = "Graph-based referring expression comprehension with expression-guided
selective filtering and noun-oriented reasoning",
JOURNAL = PR,
VOLUME = "161",
YEAR = "2025",
PAGES = "111222",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243355"}
@article{bb248463,
AUTHOR = "Ke, J.C. and Wang, D. and Chen, J.C. and Jhuo, I.H. and Lin, C.W. and Lin, Y.Y.",
TITLE = "Make Graph-Based Referring Expression Comprehension Great Again
Through Expression-Guided Dynamic Gating and Regression",
JOURNAL = MultMed,
VOLUME = "27",
YEAR = "2025",
PAGES = "1950-1961",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243356"}
@article{bb248464,
AUTHOR = "Huang, S.J. and Li, F. and Zhang, H. and Liu, S.L. and Zhang, L. and Wang, L.W.",
TITLE = "A Mutual Supervision Framework for Referring Expression Segmentation
and Generation",
JOURNAL = IJCV,
VOLUME = "133",
YEAR = "2025",
NUMBER = "6",
MONTH = "June",
PAGES = "3597-3612",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243357"}
@article{bb248465,
AUTHOR = "Ke, X. and Xu, P.R. and Guo, W.Z.",
TITLE = "Language-Image Consistency Augmentation and Distillation Network for
visual grounding",
JOURNAL = PR,
VOLUME = "166",
YEAR = "2025",
PAGES = "111663",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243358"}
@article{bb248466,
AUTHOR = "Yang, X.Z. and Liu, J.Z. and Wang, P. and Wang, G.Q. and Yang, Y. and Shen, H.T.",
TITLE = "New Dataset and Methods for Fine-Grained Compositional Referring
Expression Comprehension via Specialist-MLLM Collaboration",
JOURNAL = PAMI,
VOLUME = "47",
YEAR = "2025",
NUMBER = "10",
MONTH = "October",
PAGES = "8598-8612",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243359"}
@article{bb248467,
AUTHOR = "Guo, H. and Fan, W. and Wei, B. and Zhu, J.F. and Tian, J. and Yi, C.Z. and Jiang, F.",
TITLE = "AD-DINO: Attention-Dynamic DINO for Distance-Aware Embodied Reference
Understanding",
JOURNAL = CirSysVideo,
VOLUME = "35",
YEAR = "2025",
NUMBER = "10",
MONTH = "October",
PAGES = "10238-10249",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243360"}
@article{bb248468,
AUTHOR = "Ke, J.C. and Wen, J. and Wang, H.T. and Cheng, W.H. and Wang, J.",
TITLE = "Multi-Perspective Cross-Modal Object Encoding for Referring
Expression Comprehension",
JOURNAL = IP,
VOLUME = "34",
YEAR = "2025",
PAGES = "6911-6924",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243361"}
@article{bb248469,
AUTHOR = "Li, J. and Wen, Z. and Zhang, Y. and Wang, W.X. and Cai, Y.X. and Zhang, T.X. and He, X.J. and Liu, J.",
TITLE = "Generalized referring expression segmentation driven by
instance-oriented queries",
JOURNAL = PR,
VOLUME = "172",
YEAR = "2026",
PAGES = "112524",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243362"}
@article{bb248470,
AUTHOR = "Liu, X.Y. and Liu, T. and Huang, S. and Xin, Y. and Hu, Y. and Qin, L. and Wang, D.L. and Wu, Y.Y. and Chen, H.G.",
TITLE = "M2IST: Multi-Modal Interactive Side-Tuning for Efficient Referring
Expression Comprehension",
JOURNAL = CirSysVideo,
VOLUME = "36",
YEAR = "2026",
NUMBER = "2",
MONTH = "February",
PAGES = "1341-1354",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243363"}
@article{bb248471,
AUTHOR = "Li, R.F. and Lu, M.C. and Lin, P.Y. and Yu, Z.H. and Ma, Z.Y.",
TITLE = "Improving Scene Knowledge Referring Expression Comprehension With
Large Language Models",
JOURNAL = MultMedMag,
VOLUME = "33",
YEAR = "2026",
NUMBER = "1",
MONTH = "January",
PAGES = "72-80",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243364"}
@article{bb248472,
AUTHOR = "Zhang, Z. and Guan, Z. and Zhao, T.C. and Shen, H.Z. and Cai, Y.X. and Su, Z.G. and Shang, Y.H. and Liu, Z.J. and Yin, J.W. and Li, X.",
TITLE = "Geo-R1: Improving few-shot geospatial referring expression
understanding with reinforcement fine-tuning",
JOURNAL = PandRS,
VOLUME = "237",
YEAR = "2026",
PAGES = "113-129",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243365"}
@article{bb248473,
AUTHOR = "Cheng, W.X. and Dai, M. and Yang, W.K.",
TITLE = "PLRVG: Progressive layer-wise refinement for visual grounding via
deep-to-shallow decoding",
JOURNAL = PR,
VOLUME = "179",
YEAR = "2026",
PAGES = "113555",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243366"}
@article{bb248474,
AUTHOR = "Yang, F. and Zhu, Y. and Zhan, Y.F. and Zhao, H.Y. and Li, X. and Wang, Y.W. and Tang, M. and Ning, X. and Wang, J.Q.",
TITLE = "Seg-LLaVA: Empowering pixel-level understanding with large vision
language model",
JOURNAL = PR,
VOLUME = "179",
YEAR = "2026",
PAGES = "113560",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243367"}
@article{bb248475,
AUTHOR = "Wu, C.L. and Chen, Q. and Ji, J.Y. and Liu, Y.H. and Ma, Y.W. and Sun, X.S. and Cao, L.J.",
TITLE = "3D-STMN++: Leveraging semantic proxies to enhance superpoint-text
matching for 3D Referring Expression Segmentation",
JOURNAL = PR,
VOLUME = "179",
YEAR = "2026",
PAGES = "113854",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243368"}
@article{bb248476,
AUTHOR = "Wang, K.Y. and Wu, G. and Fu, X. and Wang, X. and Liu, K. and Lu, X. and Ge, C.J. and Zhai, W. and Zha, Z.J.",
TITLE = "SkyFind: A Large-Scale Benchmark Unveiling Referring Expression
Comprehension for UAV",
JOURNAL = PAMI,
VOLUME = "48",
YEAR = "2026",
NUMBER = "8",
MONTH = "August",
PAGES = "9859-9875",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243369"}
@article{bb248477,
AUTHOR = "Li, R. and Zhuo, W. and Zheng, S.Y. and Wu, Z.H. and Shen, L.L.",
TITLE = "Zero-shot referring expression comprehension via guidance of
Multimodal Large Language Models",
JOURNAL = PR,
VOLUME = "180",
YEAR = "2026",
PAGES = "114223",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243370"}
@inproceedings{bb248478,
AUTHOR = "Chen, J. and Wei, F.Y. and Zhao, J.J. and Song, S. and Wu, B.H. and Peng, Z.X. and Chan, S.H.G. and Zhang, H.Y.",
TITLE = "Revisiting Referring Expression Comprehension Evaluation in the Era
of Large Multimodal Models",
BOOKTITLE = "AIBench25",
YEAR = "2025",
PAGES = "513-524",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243371"}
@inproceedings{bb248479,
AUTHOR = "Wang, Z.C. and Pan, Z.Y. and Peng, Z. and Cheng, J. and Xiao, L.W. and Jiang, W. and Cao, Z.G.",
TITLE = "Exploring Contextual Attribute Density in Referring Expression
Counting",
BOOKTITLE = CVPR25,
YEAR = "2025",
PAGES = "19587-19596",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243372"}
@inproceedings{bb248480,
AUTHOR = "Chen, X. and Luo, Y.X. and Luo, G. and Ji, J.Y. and Ding, H.H. and Zhou, Y.",
TITLE = "DViN: Dynamic Visual Routing Network for Weakly Supervised Referring
Expression Comprehension",
BOOKTITLE = CVPR25,
YEAR = "2025",
PAGES = "14347-14357",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243373"}
@inproceedings{bb248481,
AUTHOR = "Wang, S.J. and Kim, D. and Taalimi, A. and Sun, C. and Kuo, W.C.",
TITLE = "Learning Visual Grounding from Generative Vision and Language Model",
BOOKTITLE = WACV25,
YEAR = "2025",
PAGES = "8057-8067",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243374"}
@inproceedings{bb248482,
AUTHOR = "Wu, T.Y. and Huang, S.Y. and Wang, Y.C.A.F.",
TITLE = "Data-Efficient 3D Visual Grounding via Order-Aware Referring",
BOOKTITLE = WACV25,
YEAR = "2025",
PAGES = "3107-3117",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243375"}
@inproceedings{bb248483,
AUTHOR = "Chu, T.Y. and Lin, Y.X. and Huang, C.C. and Hua, K.L.",
TITLE = "Enhancing Anchor-based Weakly Supervised Referring Expression
Comprehension with Cross-modality Attention",
BOOKTITLE = ACCV24,
YEAR = "2024",
PAGES = "III: 131-147",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243376"}
@inproceedings{bb248484,
AUTHOR = "Nag, S. and Goswami, K. and Karanam, S.",
TITLE = "Safari: Adaptive Sequence Transformer for Weakly Supervised Referring
Expression Segmentation",
BOOKTITLE = ECCV24,
YEAR = "2024",
PAGES = "XLIV: 485-503",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243377"}
@inproceedings{bb248485,
AUTHOR = "Dai, S.Y. and Liu, J. and Cheung, N.M.",
TITLE = "Referring Expression Counting",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "16985-16995",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243378"}
@inproceedings{bb248486,
AUTHOR = "Han, Z. and Zhu, F.R. and Lao, Q. and Jiang, H.",
TITLE = "Zero-Shot Referring Expression Comprehension via Structural
Similarity Between Images and Captions",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "14364-14375",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243379"}
@inproceedings{bb248487,
AUTHOR = "Su, W. and Miao, P.H. and Dou, H.Z. and Li, X.",
TITLE = "ScanFormer: Referring Expression Comprehension by Iteratively
Scanning",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "13449-13458",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243380"}
@inproceedings{bb248488,
AUTHOR = "Yu, Z.H. and Li, R.",
TITLE = "Revisiting Counterfactual Problems in Referring Expression
Comprehension",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "13438-13448",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243381"}
@inproceedings{bb248489,
AUTHOR = "Li, X. and Qiu, K. and Wang, J.L. and Xu, X.H. and Singh, R. and Yamazaki, K. and Chen, H. and Huang, X.N. and Raj, B.",
TITLE = "R^2-Bench: Benchmarking the Robustness of Referring Perception Models
Under Perturbations",
BOOKTITLE = ECCV24,
YEAR = "2024",
PAGES = "IX: 211-230",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243382"}
@inproceedings{bb248490,
AUTHOR = "Chng, Y.X. and Zheng, H. and Han, Y.Z. and Qiu, X. and Huang, G.",
TITLE = "Mask Grounding for Referring Image Segmentation",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "26563-26573",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243383"}
@inproceedings{bb248491,
AUTHOR = "Shah, N.A. and VS, V. and Patel, V.M.",
TITLE = "LQMFormer: Language-Aware Query Mask Transformer for Referring Image
Segmentation",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "12903-12913",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243384"}
@inproceedings{bb248492,
AUTHOR = "Wang, W.X. and Yue, T.T. and Zhang, Y. and Guo, L.T. and He, X.J. and Wang, X.L. and Liu, J.",
TITLE = "Unveiling Parts Beyond Objects: Towards Finer-Granularity Referring
Expression Segmentation",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "12998-13008",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243385"}
@inproceedings{bb248493,
AUTHOR = "Wu, Y.X. and Zhang, Z. and Xie, C. and Zhu, F. and Zhao, R.",
TITLE = "Advancing Referring Expression Segmentation Beyond Single Image",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "2628-2638",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243386"}
@inproceedings{bb248494,
AUTHOR = "Kurita, S. and Katsura, N. and Onami, E.",
TITLE = "RefEgo: Referring Expression Comprehension Dataset from First-Person
Perception of Ego4D",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "15168-15178",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243387"}
@inproceedings{bb248495,
AUTHOR = "Qiao, Y.Y. and Qi, Y.K. and Yu, Z. and Liu, J. and Wu, Q.",
TITLE = "March in Chat: Interactive Prompting for Remote Embodied Referring
Expression",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "15712-15721",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243388"}
@inproceedings{bb248496,
AUTHOR = "Chen, Y. and Du, R. and Liang, K.M. and Ma, Z.Y.",
TITLE = "Self-Enhanced Training Framework for Referring Expression Grounding",
BOOKTITLE = ICIP23,
YEAR = "2023",
PAGES = "3060-3064",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243389"}
@inproceedings{bb248497,
AUTHOR = "Sun, J. and Luo, G. and Zhou, Y.Y. and Sun, X.S. and Jiang, G.N. and Wang, Z.Y. and Ji, R.R.",
TITLE = "RefTeacher: A Strong Baseline for Semi-Supervised Referring
Expression Comprehension",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "19144-19154",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243390"}
@inproceedings{bb248498,
AUTHOR = "Tang, J.J. and Zheng, G. and Shi, C. and Yang, S.",
TITLE = "Contrastive Grouping with Transformer for Referring Image
Segmentation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "23570-23580",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243391"}
@inproceedings{bb248499,
AUTHOR = "Liu, J. and Ding, H. and Cai, Z.W. and Zhang, Y.T. and Satzoda, R.K. and Mahadevan, V. and Manmatha, R.",
TITLE = "PolyFormer: Referring Image Segmentation as Sequential Polygon
Generation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "18653-18663",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803refex3.html#TT243392"}
Last update:Sep 30, 2026 at 11:45:00