@article{bb294100,
        AUTHOR = "Mezai, L. and Hachouf, F.",
        TITLE = "Score-Level Fusion of Face and Voice Using Particle Swarm
Optimization and Belief Functions",
        JOURNAL = HMS,
        VOLUME = "45",
        YEAR = "2015",
        NUMBER = "6",
        MONTH = "December",
        PAGES = "761-772",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288725"}

@article{bb294101,
        AUTHOR = "Wu, P. and Liu, H. and Li, X. and Fan, T. and Zhang, X.",
        TITLE = "A Novel Lip Descriptor for Audio-Visual Keyword Spotting Based on
Adaptive Decision Fusion",
        JOURNAL = MultMed,
        VOLUME = "18",
        YEAR = "2016",
        NUMBER = "3",
        MONTH = "March",
        PAGES = "326-338",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288726"}

@article{bb294102,
        AUTHOR = "Dilpazir, H. and Muhammad, Z. and Minhas, Q. and Ahmed, F. and Malik, H. and Mahmood, H.",
        TITLE = "Multivariate mutual information for audio video fusion",
        JOURNAL = SIViP,
        VOLUME = "10",
        YEAR = "2016",
        NUMBER = "7",
        MONTH = "October",
        PAGES = "1265-1272",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288727"}

@article{bb294103,
        AUTHOR = "Beyan, C. and Capozzi, F. and Becchio, C. and Murino, V.",
        TITLE = "Prediction of the Leadership Style of an Emergent Leader Using Audio
and Visual Nonverbal Features",
        JOURNAL = MultMed,
        VOLUME = "20",
        YEAR = "2018",
        NUMBER = "2",
        MONTH = "February",
        PAGES = "441-456",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288728"}

@article{bb294104,
        AUTHOR = "Fernandez Lopez, A. and Sukno, F.M.",
        TITLE = "Survey on automatic lip-reading in the era of deep learning",
        JOURNAL = IVC,
        VOLUME = "78",
        YEAR = "2018",
        PAGES = "53-72",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288729"}

@article{bb294105,
        AUTHOR = "Stafylakis, T. and Khan, M.H. and Tzimiropoulos, G.",
        TITLE = "Pushing the boundaries of audiovisual word recognition using Residual
Networks and LSTMs",
        JOURNAL = CVIU,
        VOLUME = "176-177",
        YEAR = "2018",
        PAGES = "22-32",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288730"}

@inproceedings{bb294106,
        AUTHOR = "Stafylakis, T. and Tzimiropoulos, G.",
        TITLE = "Zero-Shot Keyword Spotting for Visual Speech Recognition In-the-wild",
        BOOKTITLE = ECCV18,
        YEAR = "2018",
        PAGES = "II: 536-552",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288731"}

@article{bb294107,
        AUTHOR = "Liu, X. and Geng, J.J. and Ling, H.B. and Cheung, Y.M.",
        TITLE = "Attention guided deep audio-face fusion for efficient speaker naming",
        JOURNAL = PR,
        VOLUME = "88",
        YEAR = "2019",
        PAGES = "557-568",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288732"}

@article{bb294108,
        AUTHOR = "Tsiami, A. and Koutras, P. and Katsamanis, A. and Vatakis, A. and Maragos, P.",
        TITLE = "A behaviorally inspired fusion approach for computational audiovisual
saliency modeling",
        JOURNAL = SP:IC,
        VOLUME = "76",
        YEAR = "2019",
        PAGES = "186-200",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288733"}

@article{bb294109,
        AUTHOR = "Hsiao, S. and Sun, H. and Hsieh, M. and Tsai, M. and Tsao, Y. and Lee, C.",
        TITLE = "Toward Automating Oral Presentation Scoring During Principal
Certification Program Using Audio-Video Low-Level Behavior Profiles",
        JOURNAL = AffCom,
        VOLUME = "10",
        YEAR = "2019",
        NUMBER = "4",
        MONTH = "October",
        PAGES = "552-567",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288734"}

@article{bb294110,
        AUTHOR = "Ma, Y. and Hong, H. and Li, H. and Zhao, H. and Li, Y.S. and Sun, L. and Gu, C. and Zhu, X.H.",
        TITLE = "Non-Contact Speech Recovery Technology Using a 24 GHz Portable
Auditory Radar and Webcam",
        JOURNAL = RS,
        VOLUME = "12",
        YEAR = "2020",
        NUMBER = "4",
        PAGES = "xx-yy",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288735"}

@inproceedings{bb294111,
        AUTHOR = "Xu, B. and Wang, J. and Lu, C. and Guo, Y.",
        TITLE = "Watch to Listen Clearly: Visual Speech Enhancement Driven
Multi-modality Speech Recognition",
        BOOKTITLE = WACV20,
        YEAR = "2020",
        PAGES = "1626-1635",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288736"}

@article{bb294112,
        AUTHOR = "Tao, F. and Busso, C.",
        TITLE = "End-to-End Audiovisual Speech Recognition System With Multitask
Learning",
        JOURNAL = MultMed,
        VOLUME = "23",
        YEAR = "2021",
        PAGES = "1-11",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288737"}

@article{bb294113,
        AUTHOR = "Xu, J.H. and Zhang, B. and Wang, Z.Y. and Wang, Y. and Chen, F. and Gao, J.B. and Feng, D.D.",
        TITLE = "Affective Audio Annotation of Public Speeches with Convolutional
Clustering Neural Network",
        JOURNAL = AffCom,
        VOLUME = "13",
        YEAR = "2022",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "238-249",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288738"}

@article{bb294114,
        AUTHOR = "Afouras, T. and Chung, J.S. and Senior, A. and Vinyals, O. and Zisserman, A.",
        TITLE = "Deep Audio-Visual Speech Recognition",
        JOURNAL = PAMI,
        VOLUME = "44",
        YEAR = "2022",
        NUMBER = "12",
        MONTH = "December",
        PAGES = "8717-8727",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288739"}

@inproceedings{bb294115,
        AUTHOR = "Rahimi, A. and Afouras, T. and Zisserman, A.",
        TITLE = "Reading to Listen at the Cocktail Party:
Multi-Modal Speech Separation",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "10483-10492",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288740"}

@article{bb294116,
        AUTHOR = "Narain, J. and Johnson, K.T. and Quatieri, T.F. and Picard, R.W. and Maes, P.",
        TITLE = "Modeling Real-World Affective and Communicative Nonverbal
Vocalizations From Minimally Speaking Individuals",
        JOURNAL = AffCom,
        VOLUME = "13",
        YEAR = "2022",
        NUMBER = "4",
        MONTH = "October",
        PAGES = "2238-2253",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288741"}

@article{bb294117,
        AUTHOR = "Gong, Y. and Liu, A.H. and Rouditchenko, A. and Glass, J.",
        TITLE = "UAVM: Towards Unifying Audio and Visual Models",
        JOURNAL = SPLetters,
        VOLUME = "29",
        YEAR = "2022",
        PAGES = "2437-2441",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288742"}

@inproceedings{bb294118,
        AUTHOR = "Oya, T. and Iwase, S. and Morishima, S.",
        TITLE = "The Sound of Bounding-Boxes",
        BOOKTITLE = "ICPR22",
        YEAR = "2022",
        PAGES = "9-15",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288743"}

@article{bb294119,
        AUTHOR = "Sen, T.K. and Naven, G. and Gerstner, L. and Bagley, D. and Baten, R.A. and Rahman, W. and Hasan, M.K. and Haut, K. and Mamun, A.A. and Samrose, S. and Solbu, A. and Barnes, R.E. and Frank, M.G. and Hoque, E.",
        TITLE = "DBATES: Dataset for Discerning Benefits of Audio, Textual, and Facial
Expression Features in Competitive Debate Speeches",
        JOURNAL = AffCom,
        VOLUME = "14",
        YEAR = "2023",
        NUMBER = "2",
        MONTH = "April",
        PAGES = "1028-1043",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288744"}

@article{bb294120,
        AUTHOR = "Sharma, G. and Dhall, A. and Cai, J.F.",
        TITLE = "Audio-Visual Automatic Group Affect Analysis",
        JOURNAL = AffCom,
        VOLUME = "14",
        YEAR = "2023",
        NUMBER = "2",
        MONTH = "April",
        PAGES = "1056-1069",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288745"}

@article{bb294121,
        AUTHOR = "Cheng, W.L. and Tang, W. and Huang, Y. and Luo, Y.W. and Wang, L.",
        TITLE = "A Reconstruction-Based Visual-Acoustic-Semantic Embedding Method for
Speech-Image Retrieval",
        JOURNAL = MultMed,
        VOLUME = "25",
        YEAR = "2023",
        PAGES = "4067-4080",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288746"}

@article{bb294122,
        AUTHOR = "Kefalas, T. and Fotiadou, E. and Georgopoulos, M. and Panagakis, Y. and Ma, P.C. and Petridis, S. and Stafylakis, T. and Pantic, M.",
        TITLE = "KAN-AV dataset for audio-visual face and speech analysis in the wild",
        JOURNAL = IVC,
        VOLUME = "140",
        YEAR = "2023",
        PAGES = "104839",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288747"}

@article{bb294123,
        AUTHOR = "Wang, X.M. and Mi, J.C. and Li, B.Q. and Zhao, Y.X. and Meng, J.X.",
        TITLE = "CATNet: Cross-modal fusion for audio-visual speech recognition",
        JOURNAL = PRL,
        VOLUME = "178",
        YEAR = "2024",
        PAGES = "216-222",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288748"}

@article{bb294124,
        AUTHOR = "Zhu, D.D. and Zhang, K.W. and Zhang, N. and Zhou, Q.Q. and Min, X.K. and Zhai, G.T. and Yang, X.K.",
        TITLE = "Unified Audio-Visual Saliency Model for Omnidirectional Videos With
Spatial Audio",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "764-775",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288749"}

@article{bb294125,
        AUTHOR = "Zhu, Y.X. and Duan, H.Y. and Zhang, K.W. and Zhu, Y.C. and Zhu, X. and Teng, L. and Min, X.K. and Zhai, G.T.",
        TITLE = "How Does Audio Influence Visual Attention in Omnidirectional Videos?
Database and Model",
        JOURNAL = IP,
        VOLUME = "34",
        YEAR = "2025",
        PAGES = "3447-3462",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288750"}

@inproceedings{bb294126,
        AUTHOR = "Li, J. and Zhai, G.T. and Zhu, Y.C. and Zhou, J. and Zhang, X.P.",
        TITLE = "How Sound Affects Visual Attention in Omnidirectional Videos",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "3066-3070",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288751"}

@article{bb294127,
        AUTHOR = "Qian, X.Y. and Xue, W. and Zhang, Q. and Tao, R.J. and Li, H.Z.",
        TITLE = "Deep Cross-Modal Retrieval Between Spatial Image and Acoustic Speech",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "4480-4489",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288752"}

@article{bb294128,
        AUTHOR = "Xie, J.W. and Liu, Z. and Li, G.Y. and Song, Y.J.",
        TITLE = "Audio-visual saliency prediction with multisensory perception and
integration",
        JOURNAL = IVC,
        VOLUME = "143",
        YEAR = "2024",
        PAGES = "104955",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288753"}

@article{bb294129,
        AUTHOR = "Sun, X. and Wang, X. and Liu, Q. and Zhou, X.",
        TITLE = "Multi-Level Signal Fusion for Enhanced Weakly-Supervised Audio-Visual
Video Parsing",
        JOURNAL = SPLetters,
        VOLUME = "31",
        YEAR = "2024",
        PAGES = "1149-1153",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288754"}

@article{bb294130,
        AUTHOR = "Han, H.C. and Zheng, Q.H. and Luo, M.N. and Miao, K.Y. and Tian, F. and Chen, Y.",
        TITLE = "Noise-Tolerant Learning for Audio-Visual Action Recognition",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "7761-7774",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288755"}

@article{bb294131,
        AUTHOR = "Xiao, Y.W. and Liu, X.M. and Zhu, A. and Huang, J.",
        TITLE = "Relational-branchformer: Novel framework for audio-visual speech
recognition",
        JOURNAL = IVC,
        VOLUME = "149",
        YEAR = "2024",
        PAGES = "105182",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288756"}

@article{bb294132,
        AUTHOR = "Li, W.R. and Wang, P.H. and Xiong, R.Q. and Fan, X.P.",
        TITLE = "Spiking Tucker Fusion Transformer for Audio-Visual Zero-Shot Learning",
        JOURNAL = IP,
        VOLUME = "33",
        YEAR = "2024",
        PAGES = "4840-4852",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288757"}

@article{bb294133,
        AUTHOR = "Li, W.R. and Wang, P.H. and Wang, X.T. and Zuo, W.M. and Fan, X.P. and Tian, Y.H.",
        TITLE = "Multi-Timescale Motion-Decoupled Spiking Transformer for Audio-Visual
Zero-Shot Learning",
        JOURNAL = CirSysVideo,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "10772-10786",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288758"}

@article{bb294134,
        AUTHOR = "Li, K. and Xie, F. and Chen, H. and Yuan, K. and Hu, X.L.",
        TITLE = "An Audio-Visual Speech Separation Model Inspired by
Cortico-Thalamo-Cortical Circuits",
        JOURNAL = PAMI,
        VOLUME = "46",
        YEAR = "2024",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "6637-6651",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288759"}

@article{bb294135,
        AUTHOR = "Zhou, J.X. and Guo, D. and Zhong, Y.R. and Wang, M.",
        TITLE = "Advancing Weakly-Supervised Audio-Visual Video Parsing via Segment-Wise
Pseudo Labeling",
        JOURNAL = IJCV,
        VOLUME = "132",
        YEAR = "2024",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "5308-5329",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288760"}

@inproceedings{bb294136,
        AUTHOR = "Zhou, J.X. and Guo, D. and Mao, Y.X. and Zhong, Y.R. and Chang, X.J. and Wang, M.",
        TITLE = "Label-anticipated Event Disentanglement for Audio-visual Video Parsing",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "X: 35-51",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288761"}

@article{bb294137,
        AUTHOR = "Liu, J. and Chen, S. and He, X.J. and Guo, L.T. and Zhu, X.X. and Wang, W.N. and Tang, J.H.",
        TITLE = "VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and
Dataset",
        JOURNAL = PAMI,
        VOLUME = "47",
        YEAR = "2025",
        NUMBER = "2",
        MONTH = "February",
        PAGES = "708-724",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288762"}

@article{bb294138,
        AUTHOR = "Steinmetz, N. and Balal, N.",
        TITLE = "Feasibility Study of Real-Time Speech Detection and Characterization
Using Millimeter-Wave Micro-Doppler Radar",
        JOURNAL = RS,
        VOLUME = "17",
        YEAR = "2025",
        NUMBER = "1",
        PAGES = "91",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288763"}

@article{bb294139,
        AUTHOR = "Chen, T.X. and Tan, Z.T. and Gong, T. and Chu, Q. and Wu, Y. and Liu, B. and Yu, N.H. and Lu, L. and Ye, J.P.",
        TITLE = "Bootstrapping Audio-Visual Video Segmentation by Strengthening Audio
Cues",
        JOURNAL = CirSysVideo,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "3",
        MONTH = "March",
        PAGES = "2398-2409",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288764"}

@article{bb294140,
        AUTHOR = "Gao, J.Y. and Chen, M.Y. and Xu, C.S.",
        TITLE = "Learning Probabilistic Presence-Absence Evidence for
Weakly-Supervised Audio-Visual Event Perception",
        JOURNAL = PAMI,
        VOLUME = "47",
        YEAR = "2025",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "4787-4802",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288765"}

@inproceedings{bb294141,
        AUTHOR = "Gao, J.Y. and Chen, M.Y. and Xu, C.S.",
        TITLE = "Collecting Cross-Modal Presence-Absence Evidence for
Weakly-Supervised Audio-Visual Event Perception",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "18827-18836",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288766"}

@article{bb294142,
        AUTHOR = "Li, K.W. and Chen, H. and Du, J. and Zhou, H.S. and Siniscalchi, S.M. and Niu, S.T. and Xiong, S.F.",
        TITLE = "Lightweight Audio-Visual Wake Word Spotting With Diverse Acoustic
Knowledge Distillation",
        JOURNAL = CirSysVideo,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "7",
        MONTH = "July",
        PAGES = "7308-7320",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288767"}

@article{bb294143,
        AUTHOR = "Vilaca, L. and Yu, Y. and Viana, P.",
        TITLE = "A Survey of Recent Advances and Challenges in Deep Audio-Visual
Correlation Learning",
        JOURNAL = Surveys,
        VOLUME = "57",
        YEAR = "2025",
        NUMBER = "12",
        MONTH = "July",
        PAGES = "xx-yy",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288768"}

@article{bb294144,
        AUTHOR = "Zhu, C.Z. and Shao, J.L. and Lin, J.X. and Wang, Y.J. and Wang, J. and Tang, J.H. and Li, K.",
        TITLE = "fMRI2GES: Co-Speech Gesture Reconstruction From fMRI Signal With Dual
Brain Decoding Alignment",
        JOURNAL = CirSysVideo,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "9",
        MONTH = "September",
        PAGES = "9017-9029",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288769"}

@article{bb294145,
        AUTHOR = "Zhou, D.L. and Zhang, Y.K. and Wu, J.H. and Zhang, X.Y. and Xie, L. and Yin, E.",
        TITLE = "AVE Speech: A Comprehensive Multimodal Dataset for Speech Recognition
Integrating Audio, Visual, and Electromyographic Signals",
        JOURNAL = HMS,
        VOLUME = "55",
        YEAR = "2025",
        NUMBER = "4",
        MONTH = "August",
        PAGES = "559-568",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288770"}

@article{bb294146,
        AUTHOR = "Attia, D. and Benazza Benyahia, A.",
        TITLE = "Recognizing of Vocal Fold Disorders From High Speed Video:
Use of Spatio-Temporal Deep Neural Networks",
        JOURNAL = IJIST,
        VOLUME = "35",
        YEAR = "2025",
        NUMBER = "5",
        PAGES = "e70170",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288771"}

@inproceedings{bb294147,
        AUTHOR = "Park, E.",
        TITLE = "Prompt the Missing: Efficient and Robust Audio-Visual Classification
Under Uncertain Modalities",
        BOOKTITLE = TrustworthyOpen25,
        YEAR = "2025",
        PAGES = "1645-1653",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288772"}

@article{bb294148,
        AUTHOR = "Cokelek, M. and Ozsoy, H. and Imamoglu, N. and Ozcinar, C. and Ayhan, I. and Erdem, E. and Erdem, A.",
        TITLE = "Spherical Vision Transformers for Audio-Visual Saliency Prediction in
360° Videos",
        JOURNAL = PAMI,
        VOLUME = "48",
        YEAR = "2026",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "329-345",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288773"}

@inproceedings{bb294149,
        AUTHOR = "Chao, F.Y. and Ozcinar, C. and Zhang, L. and Hamidouche, W. and Deforges, O. and Smolic, A.",
        TITLE = "Towards Audio-Visual Saliency Prediction for Omnidirectional Video
with Spatial Audio",
        BOOKTITLE = VCIP20,
        YEAR = "2020",
        PAGES = "355-358",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288774"}

@article{bb294150,
        AUTHOR = "Qi, M.S. and Lv, C.S. and Ma, H.D.",
        TITLE = "Robust Disentangled Counterfactual Learning for Physical Audiovisual
Commonsense Reasoning",
        JOURNAL = PAMI,
        VOLUME = "48",
        YEAR = "2026",
        NUMBER = "3",
        MONTH = "March",
        PAGES = "2514-2527",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288775"}

@article{bb294151,
        AUTHOR = "Kim, M. and Jung, J.W. and Rha, H. and Maiti, S. and Arora, S. and Chang, X.K. and Watanabe, S. and Ro, Y.M.",
        TITLE = "TMT: Tri-Modal Translation Between Speech, Image, and Text by
Processing Different Modalities as Different Languages",
        JOURNAL = MultMed,
        VOLUME = "28",
        YEAR = "2026",
        PAGES = "1976-1988",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288776"}

@article{bb294152,
        AUTHOR = "Tu, G. and Jing, R. and Luo, X. and Cambria, E. and Li, W.J. and Xu, R.F.",
        TITLE = "Is multimodal conversational emotion recognition satisfactory?
Exploring the gaps in performance, generalization, and confidence",
        JOURNAL = PR,
        VOLUME = "175",
        YEAR = "2026",
        PAGES = "113087",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288777"}

@article{bb294153,
        AUTHOR = "Parida, K.K. and Srivastava, S. and Sharma, G.",
        TITLE = "Noise Aware Audio-Visual Speech Denoising",
        JOURNAL = MultMed,
        VOLUME = "28",
        YEAR = "2026",
        PAGES = "2915-2924",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288778"}

@article{bb294154,
        AUTHOR = "Fu, D.J. and Cheng, X.Z. and Chen, J.Y. and Jin, T. and Zhang, Z.F.",
        TITLE = "Emphasizing Domain Differences Through Interactive-Augmented Prompts
in Continual Audio-Visual Speech Recognition",
        JOURNAL = IP,
        VOLUME = "35",
        YEAR = "2026",
        PAGES = "4269-4279",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288779"}

@article{bb294155,
        AUTHOR = "Mao, A. and Yan, J.B. and Fang, Y.M. and Cai, C.",
        TITLE = "Audio-visual saliency prediction based on joint adversarial learning
and Co-Attention mechanism",
        JOURNAL = PR,
        VOLUME = "179",
        YEAR = "2026",
        PAGES = "113548",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288780"}

@article{bb294156,
        AUTHOR = "Lu, F.H. and Yang, T. and Zhu, Z.Q. and Huang, Y. and Gao, S.Q. and Luo, Y. and Wang, Z.X. and Li, Q. and Sun, Q.Y. and Li, J.X.",
        TITLE = "MINA: Multimodal intention analysis of social media posts via
LLM-guided audio-visual-text reasoning",
        JOURNAL = PR,
        VOLUME = "179",
        YEAR = "2026",
        PAGES = "113700",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288781"}

@article{bb294157,
        AUTHOR = "Zhu, X.X. and E, X.S. and Nappi, M. and Rida, I. and Chen, H.",
        TITLE = "FedBayesMamba: Uncertainty-aware federated learning for multimodal
and audio-visual sequential modeling with selective state space
models",
        JOURNAL = PR,
        VOLUME = "180",
        YEAR = "2026",
        PAGES = "114240",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288782"}

@inproceedings{bb294158,
        AUTHOR = "Lee, K. and Zhang, Y. and Duan, Z.Y.",
        TITLE = "Audio Visual Segmentation through Text Embeddings",
        BOOKTITLE = ICIP25,
        YEAR = "2025",
        PAGES = "2510-2515",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288783"}

@inproceedings{bb294159,
        AUTHOR = "Luo, S.T. and Yang, S. and Shan, S.G. and Chen, X.L.",
        TITLE = "Dynamic Visual Speaking Patterns: You Are the Way You Speak",
        BOOKTITLE = FG25,
        YEAR = "2025",
        PAGES = "1-11",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288784"}

@inproceedings{bb294160,
        AUTHOR = "Huang, S. and Wu, J.X. and Wei, X.Y. and Cai, Y. and Jiang, D.M. and Wang, Y.W.",
        TITLE = "Sound Bridge: Associating Egocentric and Exocentric Videos via Audio
Cues",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "28942-28951",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288785"}

@inproceedings{bb294161,
        AUTHOR = "Du, H.H. and Li, G.Y. and Zhou, C. and Zhang, C.J. and Zhao, A. and Hu, D.",
        TITLE = "Crab: A Unified Audio-Visual Scene Understanding Model with Explicit
Cooperation",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "18804-18814",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288786"}

@inproceedings{bb294162,
        AUTHOR = "Shaar, E. and Shaulov, A. and Chechik, G. and Wolf, L.B.",
        TITLE = "Adapting to the Unknown: Training-Free Audio-Visual Event Perception
with Dynamic Thresholds",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "3142-3151",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288787"}

@inproceedings{bb294163,
        AUTHOR = "Wu, X.C. and Sun, H. and Wang, Y.F. and Nie, J.Y. and Zhang, J. and Wang, Y.B. and Xue, J.X. and He, L.",
        TITLE = "AVF-MAE++: Scaling Affective Video Facial Masked Autoencoders via
Efficient Audio-Visual Self-Supervised Learning",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "9142-9153",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288788"}

@inproceedings{bb294164,
        AUTHOR = "Lai, Y.H. and Ebbers, J. and Wang, Y.C.A.F. and Germain, F. and Jones, M.J. and Chatterjee, M.",
        TITLE = "UWAV: Uncertainty-Weighted Weakly-Supervised Audio-Visual Video
Parsing",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "13561-13570",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288789"}

@inproceedings{bb294165,
        AUTHOR = "Guo, R. and Ying, X.H. and Chen, Y. and Niu, D. and Li, G.Y. and Qu, L. and Qi, Y.Y. and Zhou, J.X. and Xing, B. and Yue, W.Z. and Shi, J. and Wang, Q. and Zhang, P.L. and Liang, B.",
        TITLE = "Audio-Visual Instance Segmentation",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "13550-13560",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288790"}

@inproceedings{bb294166,
        AUTHOR = "Zhang, Y.H. and Yang, S. and Shan, S.G. and Chen, X.L.",
        TITLE = "ES3: Evolving Self-Supervised Learning of Robust Audio-Visual Speech
Representations",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "27059-27069",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288791"}

@inproceedings{bb294167,
        AUTHOR = "Xiong, J.W. and Zhang, P. and You, T. and Li, C.Y. and Huang, W. and Zha, Y.F.",
        TITLE = "DiffSal: Joint Audio and Video Learning for Diffusion Saliency
Prediction",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "27263-27273",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288792"}

@inproceedings{bb294168,
        AUTHOR = "Li, X. and Wang, J.L. and Xu, X.H. and Peng, X.L. and Singh, R. and Lu, Y. and Raj, B.",
        TITLE = "QDFormer: Towards Robust Audiovisual Segmentation in Complex
Environments with Quantization-based Semantic Decomposition",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "3402-3413",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288793"}

@inproceedings{bb294169,
        AUTHOR = "Singh, N. and Wu, C.W. and Orife, I. and Kalayeh, M.",
        TITLE = "Looking Similar, Sounding Different: Leveraging Counterfactual
Cross-Modal Pairs for Audiovisual Representation Learning",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "26897-26908",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288794"}

@inproceedings{bb294170,
        AUTHOR = "Jia, W.Q. and Liu, M. and Jiang, H. and Ananthabhotla, I. and Rehg, J.M. and Ithapu, V.K. and Gao, R.H.",
        TITLE = "The Audio-Visual Conversational Graph: From an Egocentric-Exocentric
Perspective",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "26386-26395",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288795"}

@inproceedings{bb294171,
        AUTHOR = "Chen, Y.H. and Liu, Y. and Wang, H. and Liu, F. and Wang, C. and Frazer, H. and Carneiro, G.",
        TITLE = "Unraveling Instance Associations: A Closer Look for Audio-Visual
Segmentation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "26487-26497",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288796"}

@inproceedings{bb294172,
        AUTHOR = "Guo, Y.X. and Sun, S.Y. and Ma, S. and Zheng, K. and Bao, X.Y. and Ma, S.J. and Zou, W. and Zheng, Y.",
        TITLE = "CrossMAE: Cross-Modality Masked Autoencoders for Region-Aware
Audio-Visual Pre-Training",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "26711-26721",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288797"}

@inproceedings{bb294173,
        AUTHOR = "Mo, S.T. and Morgado, P.",
        TITLE = "Unveiling the Power of Audio-Visual Early Fusion Transformers with
Dense Interactions Through Masked Modeling",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "27176-27186",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288798"}

@inproceedings{bb294174,
        AUTHOR = "Wang, K. and Tian, Y.P. and Hatzinakos, D.",
        TITLE = "Towards Efficient Audio-Visual Learners via Empowering Pre-trained
Vision Transformers with Cross-Modal Adaptation",
        BOOKTITLE = WhatNext24,
        YEAR = "2024",
        PAGES = "1837-1846",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288799"}

@inproceedings{bb294175,
        AUTHOR = "Ryumina, E. and Markitantov, M. and Ryumin, D. and Kaya, H. and Karpov, A.",
        TITLE = "Zero-Shot Audio-Visual Compound Expression Recognition Method based
on Emotion Probability Fusion",
        BOOKTITLE = ABAW24,
        YEAR = "2024",
        PAGES = "4752-4760",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288800"}

@inproceedings{bb294176,
        AUTHOR = "Mahmud, T. and Mo, S.T. and Tian, Y.P. and Marculescu, D.",
        TITLE = "MA-AVT: Modality Alignment for Parameter-Efficient Audio-Visual
Transformers",
        BOOKTITLE = ECV24,
        YEAR = "2024",
        PAGES = "7996-8005",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288801"}

@inproceedings{bb294177,
        AUTHOR = "Yang, Z.Y. and Lin, J.G. and Chen, P.H. and Cherian, A. and Marks, T.K. and Le Roux, J. and Gan, C.",
        TITLE = "RILA: Reflective and Imaginative Language Agent for Zero-Shot
Semantic Audio-Visual Navigation",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "16251-16261",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288802"}

@inproceedings{bb294178,
        AUTHOR = "Dai, Y.S. and Chen, H. and Du, J. and Wang, R. and Chen, S.H. and Wang, H.T. and Lee, C.H.",
        TITLE = "A Study of Dropout-Induced Modality Bias on Robustness to Missing
Video Frames for Audio-Visual Speech Recognition",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "27435-27445",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288803"}

@inproceedings{bb294179,
        AUTHOR = "Galland, L. and Pelachaud, C. and Pecune, F.",
        TITLE = "Seeing and Hearing What Has Not Been Said: A multimodal client
behavior classifier in Motivational Interviewing with interpretable
fusion",
        BOOKTITLE = FG24,
        YEAR = "2024",
        PAGES = "1-9",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288804"}

@inproceedings{bb294180,
        AUTHOR = "Praveen, R.G. and Alam, J.",
        TITLE = "Audio-Visual Person Verification Based on Recursive Fusion of Joint
Cross-Attention",
        BOOKTITLE = FG24,
        YEAR = "2024",
        PAGES = "1-5",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288805"}

@inproceedings{bb294181,
        AUTHOR = "Praveen, R.G. and Alam, J.",
        TITLE = "Dynamic Cross Attention for Audio-Visual Person Verification",
        BOOKTITLE = FG24,
        YEAR = "2024",
        PAGES = "1-5",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288806"}

@inproceedings{bb294182,
        AUTHOR = "He, Y.H. and Shin, S. and Cherian, A. and Trigoni, N. and Markham, A.",
        TITLE = "Sound3DVDet: 3D Sound Source Detection using Multiview Microphone
Array and RGB Images",
        BOOKTITLE = WACV24,
        YEAR = "2024",
        PAGES = "5484-5495",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288807"}

@inproceedings{bb294183,
        AUTHOR = "Ghaleb, E. and Burenko, I. and Rasenberg, M. and Pouw, W. and Uhrig, P. and Holler, J. and Toni, I. and Ozyurek, A. and Fernandez, R.",
        TITLE = "Co-Speech Gesture Detection through Multi-Phase Sequence Labeling",
        BOOKTITLE = WACV24,
        YEAR = "2024",
        PAGES = "3995-4003",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288808"}

@inproceedings{bb294184,
        AUTHOR = "Xu, Y.T. and Hu, C.H. and Lee, G.H.",
        TITLE = "Rethink Cross-Modal Fusion in Weakly-Supervised Audio-Visual Video
Parsing",
        BOOKTITLE = WACV24,
        YEAR = "2024",
        PAGES = "5603-5612",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288809"}

@inproceedings{bb294185,
        AUTHOR = "Rachavarapu, K.K. and Ramakrishnan, K. and Rajagopalan, A. N.",
        TITLE = "Weakly-Supervised Audio-Visual Video Parsing with Prototype-Based
Pseudo-Labeling",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "18952-18962",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288810"}

@inproceedings{bb294186,
        AUTHOR = "Rachavarapu, K.K. and Rajagopalan, A.N.",
        TITLE = "Boosting Positive Segments for Weakly-Supervised Audio-Visual Video
Parsing",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "10158-10168",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288811"}

@inproceedings{bb294187,
        AUTHOR = "Chen, J. and Wang, W.G. and Liu, S. and Li, H.S. and Yang, Y.",
        TITLE = "Omnidirectional Information Gathering for Knowledge Transfer-based
Audio-Visual Navigation",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "10959-10969",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288812"}

@inproceedings{bb294188,
        AUTHOR = "Cheng, X.Z. and Jin, T. and Huang, R.J. and Li, L.J. and Lin, W. and Wang, Z. and Wang, Y. and Liu, H.D. and Yin, A.X. and Zhao, Z.",
        TITLE = "MixSpeech: Cross-Modality Self-Learning with Audio-Visual Stream
Mixup for Visual Speech Translation and Recognition",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "15689-15699",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288813"}

@inproceedings{bb294189,
        AUTHOR = "Georgescu, M.I. and Fonseca, E. and Ionescu, R.T. and Lucic, M. and Schmid, C. and Arnab, A.",
        TITLE = "Audiovisual Masked Autoencoders",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "16098-16108",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288814"}

@inproceedings{bb294190,
        AUTHOR = "Chen, M.F. and Su, K. and Shlizerman, E.",
        TITLE = "Be Everywhere - Hear Everything (BEE): Audio Scene Reconstruction by
Sparse Audio-Visual Samples",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "7819-7828",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288815"}

@inproceedings{bb294191,
        AUTHOR = "Xie, H.X. and Lee, M.X. and Chen, T.J. and Chen, H.J. and Liu, H.I. and Shuai, H.H. and Cheng, W.H.",
        TITLE = "Most Important Person-guided Dual-branch Cross-Patch Attention for
Group Affect Recognition",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "20541-20551",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288816"}

@inproceedings{bb294192,
        AUTHOR = "Djilali, Y.A.D. and Narayan, S. and Boussaid, H. and Almazrouei, E. and Debbah, M.",
        TITLE = "Lip2Vec: Efficient and Robust Visual Speech Recognition via
Latent-to-Latent Visual to Audio Representation Mapping",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "13744-13755",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288817"}

@inproceedings{bb294193,
        AUTHOR = "Chen, G.Y. and Zhang, D. and Liu, T. and Du, X.Y.",
        TITLE = "Local-Global Contrast for Learning Voice-Face Representations",
        BOOKTITLE = ICIP23,
        YEAR = "2023",
        PAGES = "51-55",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288818"}

@inproceedings{bb294194,
        AUTHOR = "Hong, J. and Kim, M. and Choi, J. and Ro, Y.M.",
        TITLE = "Watch or Listen: Robust Audio-Visual Speech Recognition with Visual
Corruption Modeling and Reliability Scoring",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "18783-18794",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288819"}

@inproceedings{bb294195,
        AUTHOR = "Porgali, B. and Albiero, V. and Ryda, J. and Ferrer, C.C. and Hazirbas, C.",
        TITLE = "The Casual Conversations v2 Dataset: A diverse, large benchmark for
measuring fairness and robustness in audio/vision/speech models",
        BOOKTITLE = FaDE-TCV23,
        YEAR = "2023",
        PAGES = "10-17",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288820"}

@inproceedings{bb294196,
        AUTHOR = "Xiong, J.W. and Wang, G. and Zhang, P. and Huang, W. and Zha, Y.F. and Zhai, G.T.",
        TITLE = "CASP-Net: Rethinking Video Saliency Prediction from an Audio-Visual
Consistency Perceptual Perspective",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "6441-6450",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288821"}

@inproceedings{bb294197,
        AUTHOR = "Liao, J.H. and Duan, H.H. and Feng, K.H. and Zhao, W.B. and Yang, Y.B. and Chen, L.Y.",
        TITLE = "A Light Weight Model for Active Speaker Detection",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "22932-22941",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288822"}

@inproceedings{bb294198,
        AUTHOR = "Seo, P.H. and Nagrani, A. and Schmid, C.",
        TITLE = "AVFormer: Injecting Vision into Frozen Speech Models for Zero-Shot
AV-ASR",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "22922-22931",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288823"}

@inproceedings{bb294199,
        AUTHOR = "Feng, D. and Yang, S. and Shan, S.G. and Chen, X.L.",
        TITLE = "Audio-Driven Deformation Flow for Effective Lip Reading",
        BOOKTITLE = "ICPR22",
        YEAR = "2022",
        PAGES = "274-280",
        BIBSOURCE = "http://www.visionbib.com/bibliography/people916.html#TT288824"}

Last update:Aug 19, 2026 at 13:26:35