@inproceedings{bb210800,
        AUTHOR = "Chen, Z.L. and Huang, X. and Guan, Q.L. and Lin, L. and Luo, W.Q.",
        TITLE = "A Retrospect to Multi-prompt Learning across Vision and Language",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "22133-22144",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205886"}

@inproceedings{bb210801,
        AUTHOR = "Derakhshani, M.M. and Sanchez, E. and Bulat, A. and da Costa, V.G.T. and Snoek, C.G.M. and Tzimiropoulos, G. and Martinez, B.",
        TITLE = "Bayesian Prompt Learning for Image-Language Model Generalization",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15191-15200",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205887"}

@inproceedings{bb210802,
        AUTHOR = "Cascante Bonilla, P. and Shehada, K. and Smith, J.S. and Doveh, S. and Kim, D.H. and Panda, R. and Varol, G. and Oliva, A. and Ordonez, V. and Feris, R.S. and Karlinsky, L.",
        TITLE = "Going Beyond Nouns With Vision & Language Models Using Synthetic
Data",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "20098-20108",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205888"}

@inproceedings{bb210803,
        AUTHOR = "Zara, G. and Conti, A. and Roy, S. and Lathuiliere, S. and Rota, P. and Ricci, E.",
        TITLE = "The Unreasonable Effectiveness of Large Language-Vision Models for
Source-free Video Domain Adaptation",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "10273-10283",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205889"}

@inproceedings{bb210804,
        AUTHOR = "Upadhyay, U. and Karthik, S. and Mancini, M. and Akata, Z.",
        TITLE = "ProbVLM: Probabilistic Adapter for Frozen Vison-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "1899-1910",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205890"}

@inproceedings{bb210805,
        AUTHOR = "Chen, Z.H. and Diao, S.Z. and Wang, B. and Li, G.B. and Wan, X.",
        TITLE = "Towards Unifying Medical Vision-and-Language Pre-training via Soft
Prompts",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "23346-23356",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205891"}

@inproceedings{bb210806,
        AUTHOR = "Bitton Guetta, N. and Bitton, Y. and Hessel, J. and Schmidt, L. and Elovici, Y. and Stanovsky, G. and Schwartz, R.",
        TITLE = "Breaking Common Sense: WHOOPS! A Vision-and-Language Benchmark of
Synthetic and Compositional Images",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2616-2627",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205892"}

@inproceedings{bb210807,
        AUTHOR = "Hu, Z.Y. and Li, Y. and Lyu, M.R. and Wang, L.W.",
        TITLE = "VL-PET: Vision-and-Language Parameter-Efficient Tuning via
Granularity Control",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2998-3008",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205893"}

@inproceedings{bb210808,
        AUTHOR = "Slyman, E. and Kahng, M. and Lee, S.",
        TITLE = "VLSlice: Interactive Vision-and-Language Slice Discovery",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15245-15255",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205894"}

@inproceedings{bb210809,
        AUTHOR = "Najibi, M. and Ji, J.W. and Zhou, Y. and Qi, C.R. and Yan, X.C. and Ettinger, S. and Anguelov, D.",
        TITLE = "Unsupervised 3D Perception with 2D Vision-Language Distillation for
Autonomous Driving",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "8568-8578",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205895"}

@inproceedings{bb210810,
        AUTHOR = "Zheng, K. and Wu, W. and Feng, R. and Zhu, K. and Liu, J.W. and Zhao, D.L. and Zha, Z.J. and Chen, W. and Shen, Y.J.",
        TITLE = "Regularized Mask Tuning: Uncovering Hidden Knowledge in Pre-trained
Vision-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "11629-11639",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205896"}

@inproceedings{bb210811,
        AUTHOR = "Wang, T. and Lin, K. and Li, L.J. and Lin, C.C. and Yang, Z.Y. and Zhang, H.W. and Liu, Z.C. and Wang, L.J.",
        TITLE = "Equivariant Similarity for Vision-Language Foundation Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "11964-11974",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205897"}

@inproceedings{bb210812,
        AUTHOR = "Xu, H. and Xie, S. and Huang, P.Y. and Yu, L.C. and Howes, R. and Ghosh, G. and Zettlemoyer, L. and Feichtenhofer, C.",
        TITLE = "CiT: Curation in Training for Effective Vision-Language Data",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15134-15143",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205898"}

@inproceedings{bb210813,
        AUTHOR = "Trager, M. and Perera, P. and Zancato, L. and Achille, A. and Bhatia, P. and Soatto, S.",
        TITLE = "Linear Spaces of Meanings: Compositional Structures in
Vision-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15349-15358",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205899"}

@inproceedings{bb210814,
        AUTHOR = "Chen, Y.S. and Song, Y.Z. and Yeo, C.Y. and Liu, B. and Fu, J.L. and Shuai, H.H.",
        TITLE = "SINC: Self-Supervised In-Context Learning for Vision-Language Tasks",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15384-15396",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205900"}

@inproceedings{bb210815,
        AUTHOR = "Wu, C.E. and Tian, Y. and Yu, H.C. and Wang, H. and Morgado, P. and Hu, Y.H. and Yang, L.J.",
        TITLE = "Why Is Prompt Tuning for Vision-Language Models Robust to Noisy
Labels?",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15442-15451",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205901"}

@inproceedings{bb210816,
        AUTHOR = "Ouali, Y. and Bulat, A. and Matinez, B. and Tzimiropoulos, G.",
        TITLE = "Black Box Few-Shot Adaptation for Vision-Language models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15488-15500",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205902"}

@inproceedings{bb210817,
        AUTHOR = "Kan, B. and Wang, T. and Lu, W.P. and Zhen, X.T. and Guan, W. and Zheng, F.",
        TITLE = "Knowledge-Aware Prompt Tuning for Generalizable Vision-Language
Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15624-15634",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205903"}

@inproceedings{bb210818,
        AUTHOR = "Zhai, J.T. and Zhang, Q. and Wu, T. and Chen, X.Y. and Liu, J.J. and Cheng, M.M.",
        TITLE = "SLAN: Self-Locator Aided Network for Vision-Language Understanding",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "21892-21901",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205904"}

@inproceedings{bb210819,
        AUTHOR = "Long, S. and Zhao, Z. and Yuan, J. and Tan, Z.C. and Liu, J.J. and Zhou, L.P. and Wang, S.S. and Wang, J.D.",
        TITLE = "Task-Oriented Multi-Modal Mutual Learning for Vision-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "21902-21912",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205905"}

@inproceedings{bb210820,
        AUTHOR = "Cho, E. and Kim, J. and Kim, H.W.J.",
        TITLE = "Distribution-Aware Prompt Tuning for Vision-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "21947-21956",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205906"}

@inproceedings{bb210821,
        AUTHOR = "Varma, M. and Delbrouck, J.B. and Hooper, S. and Chaudhari, A. and Langlotz, C.",
        TITLE = "ViLLA: Fine-Grained Vision-Language Representation Learning from
Real-World Data",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "22168-22178",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205907"}

@inproceedings{bb210822,
        AUTHOR = "Zhu, H.G. and Wei, Y.C. and Liang, X.D. and Zhang, C.J. and Zhao, Y.",
        TITLE = "CTP: Towards Vision-Language Continual Pretraining via Compatible
Momentum Contrast and Topology Preservation",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "22200-22210",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205908"}

@inproceedings{bb210823,
        AUTHOR = "Salin, E. and Ayache, S. and Favre, B.",
        TITLE = "Towards an Exhaustive Evaluation of Vision-Language Foundation Models",
        BOOKTITLE = MMFM23,
        YEAR = "2023",
        PAGES = "339-352",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205909"}

@inproceedings{bb210824,
        AUTHOR = "Hu, Z. and Zhu, X.L. and Tran, S. and Vidal, R. and Dhua, A.",
        TITLE = "ProVLA: Compositional Image Search with Progressive Vision-Language
Alignment and Multimodal Fusion",
        BOOKTITLE = CLVL23,
        YEAR = "2023",
        PAGES = "2764-2769",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205910"}

@inproceedings{bb210825,
        AUTHOR = "Hall, M. and Gustafson, L. and Adcock, A. and Misra, I. and Ross, C.",
        TITLE = "Vision-Language Models Performing Zero-Shot Tasks Exhibit Disparities
Between Gender Groups",
        BOOKTITLE = CLVL23,
        YEAR = "2023",
        PAGES = "2770-2777",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205911"}

@inproceedings{bb210826,
        AUTHOR = "Agnolucci, L. and Baldrati, A. and Todino, F. and Becattini, F. and Bertini, M. and del Bimbo, A.",
        TITLE = "ECO: Ensembling Context Optimization for Vision-Language Models",
        BOOKTITLE = CLVL23,
        YEAR = "2023",
        PAGES = "2803-2807",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205912"}

@inproceedings{bb210827,
        AUTHOR = "Palit, V. and Pandey, R. and Arora, A. and Liang, P.P.",
        TITLE = "Towards Vision-Language Mechanistic Interpretability: A Causal
Tracing Tool for BLIP",
        BOOKTITLE = CLVL23,
        YEAR = "2023",
        PAGES = "2848-2853",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205913"}

@inproceedings{bb210828,
        AUTHOR = "Sammani, F. and Deligiannis, N.",
        TITLE = "Uni-NLX: Unifying Textual Explanations for Vision and Vision-Language
Tasks",
        BOOKTITLE = VLAR23,
        YEAR = "2023",
        PAGES = "4636-4641",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205914"}

@inproceedings{bb210829,
        AUTHOR = "Lu, D. and Wang, Z.Q. and Wang, T. and Guan, W. and Gao, H. and Zheng, F.",
        TITLE = "Set-level Guidance Attack: Boosting Adversarial Transferability of
Vision-Language Pre-training Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "102-111",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205915"}

@inproceedings{bb210830,
        AUTHOR = "Lee, D.J. and Song, S. and Suh, J. and Choi, J. and Lee, S. and Kim, H.W.J.",
        TITLE = "Read-only Prompt Optimization for Vision-Language Few-shot Learning",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "1401-1411",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205916"}

@inproceedings{bb210831,
        AUTHOR = "Li, X. and Fang, Y.H. and Liu, M.H. and Ling, Z. and Tu, Z.W. and Su, H.",
        TITLE = "Distilling Large Vision-Language Model with Out-of-Distribution
Generalizability",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2492-2503",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205917"}

@inproceedings{bb210832,
        AUTHOR = "Li, J.C. and Gao, M. and Wei, L. and Tang, S.L. and Zhang, W.Q. and Li, M. and Ji, W. and Tian, Q. and Chua, T.S. and Zhuang, Y.T.",
        TITLE = "Gradient-Regulated Meta-Prompt Learning for Generalizable
Vision-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2551-2562",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205918"}

@inproceedings{bb210833,
        AUTHOR = "Bi, J.Y. and Cheng, D. and Yao, P. and Pang, B. and Zhan, Y.F. and Yang, C.G. and Wang, Y.J. and Sun, H. and Deng, W.W. and Zhang, Q.",
        TITLE = "VL-Match: Enhancing Vision-Language Pretraining with Token-Level and
Instance-Level Matching",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2584-2593",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205919"}

@inproceedings{bb210834,
        AUTHOR = "Udandarao, V. and Gupta, A. and Albanie, S.",
        TITLE = "SuS-X: Training-Free Name-Only Transfer of Vision-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2725-2736",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205920"}

@inproceedings{bb210835,
        AUTHOR = "Jiang, C. and Xu, H.Y. and Ye, W. and Ye, Q.H. and Li, C.L. and Yan, M. and Bi, B. and Zhang, S.K. and Huang, F. and Huang, S.",
        TITLE = "BUS: Efficient and Effective Vision-language Pre-training with
Bottom-Up Patch Summarization",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2888-2898",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205921"}

@inproceedings{bb210836,
        AUTHOR = "Shi, C. and Yang, S.",
        TITLE = "LoGoPrompt: Synthetic Text Images Can Be Good Visual Prompts for
Vision-Language Models",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "2920-2929",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205922"}

@inproceedings{bb210837,
        AUTHOR = "Wang, A.J.P. and Lin, K.Q. and Zhang, D.J. and Lei, S.W.X. and Shou, M.Z.",
        TITLE = "Too Large; Data Reduction for Vision-Language Pre-Training",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "3124-3134",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205923"}

@inproceedings{bb210838,
        AUTHOR = "Wang, W.H. and Yang, Z. and Xu, B. and Li, J. and Sun, Y.",
        TITLE = "ViLTA: Enhancing Vision-Language Pre-training through Textual
Augmentation",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "3135-3146",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205924"}

@inproceedings{bb210839,
        AUTHOR = "Wang, T.J.J. and Laaksonen, J. and Langer, T. and Arponen, H. and Bishop, T.E.",
        TITLE = "Learning by Hallucinating:
Vision-Language Pre-training with Weak Supervision",
        BOOKTITLE = WACV23,
        YEAR = "2023",
        PAGES = "1073-1083",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205925"}

@inproceedings{bb210840,
        AUTHOR = "Boecking, B. and Usuyama, N. and Bannur, S. and Castro, D.C. and Schwaighofer, A. and Hyland, S. and Wetscherek, M. and Naumann, T. and Nori, A. and Alvarez Valle, J. and Poon, H. and Oktay, O.",
        TITLE = "Making the Most of Text Semantics to Improve Biomedical Vision-Language
Processing",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXXVI:1-21",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205926"}

@inproceedings{bb210841,
        AUTHOR = "Cui, Q. and Zhou, B. and Guo, Y. and Yin, W.D. and Wu, H. and Yoshie, O. and Chen, Y.",
        TITLE = "Contrastive Vision-Language Pre-training with Limited Resources",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXXVI:236-253",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205927"}

@inproceedings{bb210842,
        AUTHOR = "Walmer, M. and Sikka, K. and Sur, I. and Shrivastava, A. and Jha, S.",
        TITLE = "Dual-Key Multimodal Backdoors for Visual Question Answering",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "15354-15364",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205928"}

@inproceedings{bb210843,
        AUTHOR = "Ding, Y. and Yu, J. and Liu, B. and Hu, Y. and Cui, M.X. and Wu, Q.",
        TITLE = "MuKEA: Multimodal Knowledge Extraction and Accumulation for
Knowledge-based Visual Question Answering",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "5079-5088",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205929"}

@inproceedings{bb210844,
        AUTHOR = "Gao, F. and Ping, Q. and Thattai, G. and Reganti, A. and Wu, Y.N. and Natarajan, P.",
        TITLE = "Transform-Retrieve-Generate: Natural Language-Centric
Outside-Knowledge Visual Question Answering",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "5057-5067",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205930"}

@inproceedings{bb210845,
        AUTHOR = "Aflalo, E. and Du, M. and Tseng, S.Y. and Liu, Y.F. and Wu, C. and Duan, N. and Lal, V.",
        TITLE = "VL-InterpreT: An Interactive Visualization Tool for Interpreting
Vision-Language Transformers",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "21374-21383",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205931"}

@inproceedings{bb210846,
        AUTHOR = "Hu, X.W. and Gan, Z. and Wang, J.F. and Yang, Z.Y. and Liu, Z.C. and Lu, Y. and Wang, L.J.",
        TITLE = "Scaling Up Vision-Language Pretraining for Image Captioning",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "17959-17968",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205932"}

@inproceedings{bb210847,
        AUTHOR = "Zhang, P.C. and Li, X.J. and Hu, X.W. and Yang, J.W. and Zhang, L. and Wang, L.J. and Choi, Y.J. and Gao, J.F.",
        TITLE = "VinVL: Revisiting Visual Representations in Vision-Language Models",
        BOOKTITLE = CVPR21,
        YEAR = "2021",
        PAGES = "5575-5584",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205933"}

@inproceedings{bb210848,
        AUTHOR = "Li, Z.W. and Stengel Eskin, E. and Zhang, Y.X. and Xie, C. and Tran, Q. and van Durme, B. and Yuille, A.L.",
        TITLE = "Calibrating Concepts and Operations:
Towards Symbolic Reasoning on Real Images",
        BOOKTITLE = ICCV21,
        YEAR = "2021",
        PAGES = "14890-14899",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205934"}

@inproceedings{bb210849,
        AUTHOR = "Yang, X. and Zhang, H.W. and Qi, G.J. and Cai, J.F.",
        TITLE = "Causal Attention for Vision-Language Tasks",
        BOOKTITLE = CVPR21,
        YEAR = "2021",
        PAGES = "9842-9852",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205935"}

@inproceedings{bb210850,
        AUTHOR = "Stefanini, M. and Cornia, M. and Baraldi, L. and Cucchiara, R.",
        TITLE = "A Novel Attention-based Aggregation Function to Combine Vision and
Language",
        BOOKTITLE = ICPR21,
        YEAR = "2021",
        PAGES = "1212-1219",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205936"}

@inproceedings{bb210851,
        AUTHOR = "Jain, V. and Lodhavia, J.",
        TITLE = "Automatic Question Tagging using k-Nearest Neighbors and Random
Forest",
        BOOKTITLE = ISCV20,
        YEAR = "2020",
        PAGES = "1-4",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205937"}

@inproceedings{bb210852,
        AUTHOR = "Zheng, W.B. and Yan, L. and Gou, C. and Wang, F.Y.",
        TITLE = "Webly Supervised Knowledge Embedding Model for Visual Reasoning",
        BOOKTITLE = CVPR20,
        YEAR = "2020",
        PAGES = "12442-12451",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205938"}

@inproceedings{bb210853,
        AUTHOR = "Nguyen, D.K. and Okatani, T.",
        TITLE = "Multi-Task Learning of Hierarchical Vision-Language Representation",
        BOOKTITLE = CVPR19,
        YEAR = "2019",
        PAGES = "10484-10493",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205939"}

@inproceedings{bb210854,
        AUTHOR = "Gupta, T. and Shih, K.J. and Singh, S. and Hoiem, D.",
        TITLE = "Aligned Image-Word Representations Improve Inductive Transfer Across
Vision-Language Tasks",
        BOOKTITLE = ICCV17,
        YEAR = "2017",
        PAGES = "4223-4232",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vlm3.html#TT205940"}

@article{bb210855,
        AUTHOR = "Wu, Y.C. and Yang, J.C.",
        TITLE = "A Robust Passage Retrieval Algorithm for Video Question Answering",
        JOURNAL = CirSysVideo,
        VOLUME = "18",
        YEAR = "2008",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "1411-1421",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205941"}

@inproceedings{bb210856,
        AUTHOR = "Wu, Y.C. and Lee, Y.S. and Yang, J.C. and Yen, S.J.",
        TITLE = "A New Passage Ranking Algorithm for Video Question Answering",
        BOOKTITLE = PSIVT06,
        YEAR = "2006",
        PAGES = "563-572",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205942"}

@article{bb210857,
        AUTHOR = "Li, G.D. and Li, H.J. and Ming, Z.Y. and Hong, R.C. and Tang, S. and Chua, T.S.",
        TITLE = "Question Answering over Community-Contributed Web Videos",
        JOURNAL = MultMedMag,
        VOLUME = "17",
        YEAR = "2010",
        NUMBER = "4",
        MONTH = "October",
        PAGES = "46-57",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205943"}

@inproceedings{bb210858,
        AUTHOR = "Song, Y.C. and Li, H.J.",
        TITLE = "Mash-Up Approach for Web Video Category Recommendation",
        BOOKTITLE = PSIVT10,
        YEAR = "2010",
        PAGES = "197-202",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205944"}

@article{bb210859,
        AUTHOR = "Guo, Z.Y. and Zhao, Z. and Jin, W. and Wei, Z.C. and Yang, M. and Wang, N.N. and Yuan, N.J.",
        TITLE = "Multi-Turn Video Question Generation via Reinforced Multi-Choice
Attention Network",
        JOURNAL = CirSysVideo,
        VOLUME = "31",
        YEAR = "2021",
        NUMBER = "5",
        PAGES = "1697-1710",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205945"}

@article{bb210860,
        AUTHOR = "Xue, H.Y. and Chu, W. and Zhao, Z. and Cai, D.",
        TITLE = "A Better Way to Attend: Attention With Trees for Video Question
Answering",
        JOURNAL = IP,
        VOLUME = "27",
        YEAR = "2018",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "5563-5574",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205946"}

@article{bb210861,
        AUTHOR = "Xue, H.Y. and Zhao, Z. and Cai, D.",
        TITLE = "Unifying the Video and Question Attentions for Open-Ended Video
Question Answering",
        JOURNAL = IP,
        VOLUME = "26",
        YEAR = "2017",
        NUMBER = "12",
        MONTH = "December",
        PAGES = "5656-5666",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205947"}

@article{bb210862,
        AUTHOR = "Zhao, Z. and Xiao, S.W. and Song, Z. and Lu, C.J. and Xiao, J. and Zhuang, Y.T.",
        TITLE = "Open-Ended Video Question Answering via Multi-Modal Conditional
Adversarial Networks",
        JOURNAL = IP,
        VOLUME = "29",
        YEAR = "2020",
        PAGES = "3859-3870",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205948"}

@article{bb210863,
        AUTHOR = "Zhao, Z. and Zhang, Z. and Xiao, S.W. and Xiao, Z.X. and Yan, X.H. and Yu, J. and Cai, D. and Wu, F.",
        TITLE = "Long-Form Video Question Answering via Dynamic Hierarchical
Reinforced Networks",
        JOURNAL = IP,
        VOLUME = "28",
        YEAR = "2019",
        NUMBER = "12",
        MONTH = "December",
        PAGES = "5939-5952",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205949"}

@article{bb210864,
        AUTHOR = "Yu, T. and Yu, J. and Yu, Z. and Huang, Q.M. and Tian, Q.",
        TITLE = "Long-Term Video Question Answering via Multimodal Hierarchical Memory
Attentive Networks",
        JOURNAL = CirSysVideo,
        VOLUME = "31",
        YEAR = "2021",
        NUMBER = "3",
        MONTH = "March",
        PAGES = "931-944",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205950"}

@article{bb210865,
        AUTHOR = "Jang, Y. and Song, Y. and Kim, C.D. and Yu, Y. and Kim, Y. and Kim, G.",
        TITLE = "Video Question Answering with Spatio-Temporal Reasoning",
        JOURNAL = IJCV,
        VOLUME = "127",
        YEAR = "2019",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "1385-1412",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205951"}

@inproceedings{bb210866,
        AUTHOR = "Jang, Y. and Song, Y. and Yu, Y. and Kim, Y. and Kim, G.",
        TITLE = "TGIF-QA:
Toward Spatio-Temporal Reasoning in Visual Question Answering",
        BOOKTITLE = CVPR17,
        YEAR = "2017",
        PAGES = "1359-1367",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205952"}

@article{bb210867,
        AUTHOR = "Yu, T. and Yu, J. and Yu, Z. and Tao, D.",
        TITLE = "Compositional Attention Networks With Two-Stream Fusion for Video
Question Answering",
        JOURNAL = IP,
        VOLUME = "29",
        YEAR = "2020",
        NUMBER = "",
        PAGES = "1204-1218",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205953"}

@article{bb210868,
        AUTHOR = "Wang, W.N. and Huang, Y. and Wang, L.",
        TITLE = "Long video question answering: A Matching-guided Attention Model",
        JOURNAL = PR,
        VOLUME = "102",
        YEAR = "2020",
        PAGES = "107248",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205954"}

@article{bb210869,
        AUTHOR = "Zhang, W. and Tang, S. and Cao, Y. and Pu, S. and Wu, F. and Zhuang, Y.",
        TITLE = "Frame Augmented Alternating Attention Network for Video Question
Answering",
        JOURNAL = MultMed,
        VOLUME = "22",
        YEAR = "2020",
        NUMBER = "4",
        MONTH = "April",
        PAGES = "1032-1041",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205955"}

@article{bb210870,
        AUTHOR = "Chen, J. and Shao, J. and He, C.",
        TITLE = "Movie fill in the blank by joint learning from video and text with
adaptive temporal attention",
        JOURNAL = PRL,
        VOLUME = "132",
        YEAR = "2020",
        PAGES = "62-68",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205956"}

@article{bb210871,
        AUTHOR = "Wang, A. and Luu, A.T. and Foo, C. and Zhu, H. and Tay, Y. and Chandrasekhar, V.",
        TITLE = "Holistic Multi-Modal Memory Network for Movie Question Answering",
        JOURNAL = IP,
        VOLUME = "29",
        YEAR = "2020",
        NUMBER = "1",
        PAGES = "489-499",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205957"}

@article{bb210872,
        AUTHOR = "Yuan, Z.Q. and Sun, S.Y. and Duan, L.X. and Li, C.S. and Wu, X. and Xu, C.S.",
        TITLE = "Adversarial Multimodal Network for Movie Story Question Answering",
        JOURNAL = MultMed,
        VOLUME = "23",
        YEAR = "2021",
        PAGES = "1744-1756",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205958"}

@article{bb210873,
        AUTHOR = "Gu, M. and Zhao, Z. and Jin, W. and Hong, R. and Wu, F.",
        TITLE = "Graph-Based Multi-Interaction Network for Video Question Answering",
        JOURNAL = IP,
        VOLUME = "30",
        YEAR = "2021",
        PAGES = "2758-2770",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205959"}

@article{bb210874,
        AUTHOR = "Xie, Z. and Wu, K.W. and Zhang, X.Y. and Yang, X.M. and Hou, J.K.",
        TITLE = "Learning continuous temporal embedding of videos using pattern theory",
        JOURNAL = PRL,
        VOLUME = "146",
        YEAR = "2021",
        PAGES = "222-229",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205960"}

@article{bb210875,
        AUTHOR = "Liu, Y. and Zhang, X.M. and Zhang, Q.Y. and Li, C.Z. and Huang, F. and Tang, X.H. and Li, Z.J.",
        TITLE = "Dual self-attention with co-attention networks for visual question
answering",
        JOURNAL = PR,
        VOLUME = "117",
        YEAR = "2021",
        PAGES = "107956",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205961"}

@article{bb210876,
        AUTHOR = "Liu, Y. and Zhang, X.M. and Huang, F. and Shen, S.X. and Tian, P. and Li, L. and Li, Z.J.",
        TITLE = "Dynamic Self-Attention with Vision Synchronization Networks for Video
Question Answering",
        JOURNAL = PR,
        VOLUME = "132",
        YEAR = "2022",
        PAGES = "108959",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205962"}

@article{bb210877,
        AUTHOR = "Liu, Y. and Zhang, X.M. and Huang, F. and Zhang, B. and Li, Z.J.",
        TITLE = "Cross-Attentional Spatio-Temporal Semantic Graph Networks for Video
Question Answering",
        JOURNAL = IP,
        VOLUME = "31",
        YEAR = "2022",
        PAGES = "1684-1696",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205963"}

@article{bb210878,
        AUTHOR = "Jin, W. and Zhao, Z. and Cao, X.C. and Zhu, J.M. and He, X.Q. and Zhuang, Y.T.",
        TITLE = "Adaptive Spatio-Temporal Graph Enhanced Vision-Language
Representation for Video QA",
        JOURNAL = IP,
        VOLUME = "30",
        YEAR = "2021",
        PAGES = "5477-5489",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205964"}

@article{bb210879,
        AUTHOR = "Gao, L. and Chen, T.M. and Li, X.P. and Zeng, P.P. and Zhao, L. and Li, Y.F.",
        TITLE = "Generalized pyramid co-attention with learnable aggregation net for
video question answering",
        JOURNAL = PR,
        VOLUME = "120",
        YEAR = "2021",
        PAGES = "108145",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205965"}

@article{bb210880,
        AUTHOR = "Le, T.M. and Le, V. and Venkatesh, S. and Tran, T.",
        TITLE = "Hierarchical Conditional Relation Networks for Multimodal Video
Question Answering",
        JOURNAL = IJCV,
        VOLUME = "129",
        YEAR = "2021",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "3027-3050",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205966"}

@inproceedings{bb210881,
        AUTHOR = "Le, T.M. and Le, V. and Venkatesh, S. and Tran, T.",
        TITLE = "Hierarchical Conditional Relation Networks for Video Question
Answering",
        BOOKTITLE = CVPR20,
        YEAR = "2020",
        PAGES = "9969-9978",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205967"}

@article{bb210882,
        AUTHOR = "Su, H.T. and Chang, C.H. and Shen, P.W. and Wang, Y.S. and Chang, Y.L. and Chang, Y.C. and Cheng, P.J. and Hsu, W.H.",
        TITLE = "End-to-End Video Question-Answer Generation With Generator-Pretester
Network",
        JOURNAL = CirSysVideo,
        VOLUME = "31",
        YEAR = "2021",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "4497-4507",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205968"}

@article{bb210883,
        AUTHOR = "Gao, L.L. and Lei, Y. and Zeng, P.P. and Song, J.K. and Wang, M. and Shen, H.T.",
        TITLE = "Hierarchical Representation Network With Auxiliary Tasks for Video
Captioning and Video Question Answering",
        JOURNAL = IP,
        VOLUME = "31",
        YEAR = "2022",
        PAGES = "202-215",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205969"}

@article{bb210884,
        AUTHOR = "Zhang, J.P. and Shao, J. and Cao, R. and Gao, L.L. and Xu, X. and Shen, H.T.",
        TITLE = "Action-Centric Relation Transformer Network for Video Question
Answering",
        JOURNAL = CirSysVideo,
        VOLUME = "32",
        YEAR = "2022",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "63-74",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205970"}

@article{bb210885,
        AUTHOR = "Zhang, H. and Sun, A. and Jing, W. and Zhen, L.L. and Zhou, J.T.Y. and Goh, R.S.M.",
        TITLE = "Natural Language Video Localization: A Revisit in Span-Based Question
Answering Framework",
        JOURNAL = PAMI,
        VOLUME = "44",
        YEAR = "2022",
        NUMBER = "8",
        MONTH = "August",
        PAGES = "4252-4266",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205971"}

@article{bb210886,
        AUTHOR = "Wang, J.Y. and Bao, B.K. and Xu, C.S.",
        TITLE = "DualVGR: A Dual-Visual Graph Reasoning Unit for Video Question
Answering",
        JOURNAL = MultMed,
        VOLUME = "24",
        YEAR = "2022",
        PAGES = "3369-3380",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205972"}

@article{bb210887,
        AUTHOR = "Zeng, P.P. and Zhang, H.N. and Gao, L. and Song, J.K. and Shen, H.T.",
        TITLE = "Video Question Answering With Prior Knowledge and Object-Sensitive
Learning",
        JOURNAL = IP,
        VOLUME = "31",
        YEAR = "2022",
        PAGES = "5936-5948",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205973"}

@article{bb210888,
        AUTHOR = "Gan, Z. and Li, L.J. and Li, C.Y. and Wang, L.J. and Liu, Z.C. and Gao, J.F.",
        TITLE = "Vision-Language Pre-Training:
Basics, Recent Advances, and Future Trends",
        JOURNAL = FTCGV,
        VOLUME = "14",
        YEAR = "2022",
        NUMBER = "3-4",
        PAGES = "163-352",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205974"}

@article{bb210889,
        AUTHOR = "Zhang, F. and Wang, R. and Zhou, F. and Luo, Y.M.",
        TITLE = "ERM: Energy-Based Refined-Attention Mechanism for Video Question
Answering",
        JOURNAL = CirSysVideo,
        VOLUME = "33",
        YEAR = "2023",
        NUMBER = "3",
        MONTH = "March",
        PAGES = "1454-1467",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205975"}

@article{bb210890,
        AUTHOR = "Yang, J. and Jang, H. and Yu, K.",
        TITLE = "Analyzing Geographic Questions Using Embedding-based Topic Modeling",
        JOURNAL = IJGI,
        VOLUME = "12",
        YEAR = "2023",
        NUMBER = "2",
        PAGES = "xx-yy",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205976"}

@inproceedings{bb210891,
        AUTHOR = "Zhao, S.W. and Liu, Y.Y. and Du, S. and Tian, Z.Q. and Qu, T. and Xu, L.H.",
        TITLE = "CMFG: Cross-model Fine-grained Feature Interaction for Text-video
Retrieval",
        BOOKTITLE = MMMod23,
        YEAR = "2023",
        PAGES = "II: 435-445",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205977"}

@article{bb210892,
        AUTHOR = "Luo, H.N. and Lin, G.S. and Yao, Y.Z. and Liu, F.Y. and Liu, Z.C. and Tang, Z.M.",
        TITLE = "Depth and Video Segmentation Based Visual Attention for Embodied
Question Answering",
        JOURNAL = PAMI,
        VOLUME = "45",
        YEAR = "2023",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "6807-6819",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205978"}

@inproceedings{bb210893,
        AUTHOR = "Luo, H.N. and Lin, G.S. and Liu, Z.C. and Liu, F.Y. and Tang, Z.M. and Yao, Y.Z.",
        TITLE = "SegEQA: Video Segmentation Based Visual Attention for Embodied
Question Answering",
        BOOKTITLE = ICCV19,
        YEAR = "2019",
        PAGES = "9666-9675",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205979"}

@article{bb210894,
        AUTHOR = "Zhang, X. and Zhang, F.F. and Xu, C.S.",
        TITLE = "Reducing Vision-Answer Biases for Multiple-Choice VQA",
        JOURNAL = IP,
        VOLUME = "32",
        YEAR = "2023",
        PAGES = "4621-4634",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205980"}

@article{bb210895,
        AUTHOR = "Xiao, J.B. and Zhou, P. and Yao, A. and Li, Y.C. and Hong, R.C. and Yan, S.C. and Chua, T.S.",
        TITLE = "Contrastive Video Question Answering via Video Graph Transformer",
        JOURNAL = PAMI,
        VOLUME = "45",
        YEAR = "2023",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "13265-13280",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205981"}

@inproceedings{bb210896,
        AUTHOR = "Xiao, J.B. and Zhou, P. and Chua, T.S. and Yan, S.C.",
        TITLE = "Video Graph Transformer for Video Question Answering",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXXVI:39-58",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205982"}

@article{bb210897,
        AUTHOR = "Shen, W.X. and Song, J. and Zhu, X. and Li, G. and Shen, H.T.",
        TITLE = "End-to-End Pre-Training With Hierarchical Matching and Momentum
Contrast for Text-Video Retrieval",
        JOURNAL = IP,
        VOLUME = "32",
        YEAR = "2023",
        PAGES = "5017-5030",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205983"}

@article{bb210898,
        AUTHOR = "Jiang, J.J. and Liu, Z. and Zheng, N.N.",
        TITLE = "LiVLR: A Lightweight Visual-Linguistic Reasoning Framework for Video
Question Answering",
        JOURNAL = MultMed,
        VOLUME = "25",
        YEAR = "2023",
        PAGES = "5002-5013",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205984"}

@article{bb210899,
        AUTHOR = "Xu, F.F. and Zhu, Y. and Wang, C. and Cao, Y.Z. and Zhong, Z. and Li, X.M.",
        TITLE = "Spatio-Temporal Two-stage Fusion for video question answering",
        JOURNAL = CVIU,
        VOLUME = "237",
        YEAR = "2023",
        PAGES = "103821",
        BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803vidq2.html#TT205985"}

Last update:Jan 30, 2024 at 20:33:16