@article{bb136400,
AUTHOR = "Li, L.H. and Tang, S. and Zhang, Y.D. and Deng, L.X. and Tian, Q.",
TITLE = "GLA: Global-Local Attention for Image Description",
JOURNAL = MultMed,
VOLUME = "20",
YEAR = "2018",
NUMBER = "3",
MONTH = "March",
PAGES = "726-737",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132392"}
@article{bb136401,
AUTHOR = "Wu, C.L. and Wei, Y.W. and Chu, X.L. and Su, F. and Wang, L.Q.",
TITLE = "Modeling visual and word-conditional semantic attention for image
captioning",
JOURNAL = SP:IC,
VOLUME = "67",
YEAR = "2018",
PAGES = "100-107",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132393"}
@article{bb136402,
AUTHOR = "Xu, N. and Liu, A.A. and Liu, J. and Nie, W.Z. and Su, Y.T.",
TITLE = "Scene graph captioner:
Image captioning based on structural visual representation",
JOURNAL = JVCIR,
VOLUME = "58",
YEAR = "2019",
PAGES = "477-485",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132394"}
@article{bb136403,
AUTHOR = "Ding, S.T. and Qu, S. and Xi, Y.L. and Sangaiah, A.K. and Wan, S.H.",
TITLE = "Image caption generation with high-level image features",
JOURNAL = PRL,
VOLUME = "123",
YEAR = "2019",
PAGES = "89-95",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132395"}
@article{bb136404,
AUTHOR = "Zhang, Z.J. and Wu, Q. and Wang, Y. and Chen, F.",
TITLE = "High-Quality Image Captioning With Fine-Grained and Semantic-Guided
Visual Attention",
JOURNAL = MultMed,
VOLUME = "21",
YEAR = "2019",
NUMBER = "7",
MONTH = "July",
PAGES = "1681-1693",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132396"}
@inproceedings{bb136405,
AUTHOR = "Zhang, Z.J. and Wu, Q. and Wang, Y. and Chen, F.",
TITLE = "Fine-Grained and Semantic-Guided Visual Attention for Image
Captioning",
BOOKTITLE = WACV18,
YEAR = "2018",
PAGES = "1709-1717",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132397"}
@article{bb136406,
AUTHOR = "Tan, J.H. and Chan, C.S. and Chuah, J.H.",
TITLE = "COMIC: Toward A Compact Image Captioning Model With Attention",
JOURNAL = MultMed,
VOLUME = "21",
YEAR = "2019",
NUMBER = "10",
MONTH = "October",
PAGES = "2686-2696",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132398"}
@article{bb136407,
AUTHOR = "Yang, L. and Hu, H.F.",
TITLE = "Visual Skeleton and Reparative Attention for Part-of-Speech image
captioning system",
JOURNAL = CVIU,
VOLUME = "189",
YEAR = "2019",
PAGES = "102819",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132399"}
@article{bb136408,
AUTHOR = "Wang, J.B. and Wang, W. and Wang, L. and Wang, Z.Y. and Feng, D.D. and Tan, T.N.",
TITLE = "Learning Visual Relationship and Context-Aware Attention for Image
Captioning",
JOURNAL = PR,
VOLUME = "98",
YEAR = "2020",
PAGES = "107075",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132400"}
@article{bb136409,
AUTHOR = "Wei, H.Y. and Li, Z.X. and Zhang, C.L. and Ma, H.F.",
TITLE = "The synergy of double attention: Combine sentence-level and
word-level attention for image captioning",
JOURNAL = CVIU,
VOLUME = "201",
YEAR = "2020",
PAGES = "103068",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132401"}
@article{bb136410,
AUTHOR = "Ji, J.Z. and Du, Z.R. and Zhang, X.D.",
TITLE = "Divergent-convergent attention for image captioning",
JOURNAL = PR,
VOLUME = "115",
YEAR = "2021",
PAGES = "107928",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132402"}
@article{bb136411,
AUTHOR = "Wei, Y.W. and Wu, C.L. and Jia, Z.Y. and Hu, X.F. and Guo, S. and Shi, H.T.",
TITLE = "Past is important: Improved image captioning by looking back in time",
JOURNAL = SP:IC,
VOLUME = "94",
YEAR = "2021",
PAGES = "116183",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132403"}
@article{bb136412,
AUTHOR = "Zhang, Z.J. and Wu, Q. and Wang, Y. and Chen, F.",
TITLE = "Exploring region relationships implicitly:
Image captioning with visual relationship attention",
JOURNAL = IVC,
VOLUME = "109",
YEAR = "2021",
PAGES = "104146",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132404"}
@article{bb136413,
AUTHOR = "Zhang, Z.J. and Wu, Q. and Wang, Y. and Chen, F.",
TITLE = "Exploring Pairwise Relationships Adaptively From Linguistic Context
in Image Captioning",
JOURNAL = MultMed,
VOLUME = "24",
YEAR = "2022",
PAGES = "3101-3113",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132405"}
@article{bb136414,
AUTHOR = "Zhong, X. and Nie, G.Z. and Huang, W.X. and Liu, W.X. and Ma, B. and Lin, C.W.",
TITLE = "Attention-guided image captioning with adaptive global and local
feature fusion",
JOURNAL = JVCIR,
VOLUME = "78",
YEAR = "2021",
PAGES = "103138",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132406"}
@article{bb136415,
AUTHOR = "Wan, B.Y. and Jiang, W.H. and Fang, Y.M. and Zhu, M.W. and Li, Q. and Liu, Y.",
TITLE = "Revisiting image captioning via maximum discrepancy competition",
JOURNAL = PR,
VOLUME = "122",
YEAR = "2022",
PAGES = "108358",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132407"}
@article{bb136416,
AUTHOR = "Chen, T.Y. and Li, Z.X. and Wu, J.L. and Ma, H.F. and Su, B.P.",
TITLE = "Improving image captioning with Pyramid Attention and SC-GAN",
JOURNAL = IVC,
VOLUME = "117",
YEAR = "2022",
PAGES = "104340",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132408"}
@article{bb136417,
AUTHOR = "Zhou, Y.J. and Long, J.F. and Xu, S.P. and Shang, L.",
TITLE = "Attribute-driven image captioning via soft-switch pointer",
JOURNAL = PRL,
VOLUME = "152",
YEAR = "2021",
PAGES = "34-41",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132409"}
@article{bb136418,
AUTHOR = "Wang, Q.Z. and Wan, J. and Chan, A.B.",
TITLE = "On Diversity in Image Captioning: Metrics and Methods",
JOURNAL = PAMI,
VOLUME = "44",
YEAR = "2022",
NUMBER = "2",
MONTH = "February",
PAGES = "1035-1049",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132410"}
@article{bb136419,
AUTHOR = "Wang, J.N. and Xu, W.J. and Wang, Q.Z. and Chan, A.B.",
TITLE = "On Distinctive Image Captioning via Comparing and Reweighting",
JOURNAL = PAMI,
VOLUME = "45",
YEAR = "2023",
NUMBER = "2",
MONTH = "February",
PAGES = "2088-2103",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132411"}
@article{bb136420,
AUTHOR = "Wang, J.N. and Xu, W.J. and Wang, Q.Z. and Chan, A.B.",
TITLE = "Group-Based Distinctive Image Captioning with Memory Difference
Encoding and Attention",
JOURNAL = IJCV,
VOLUME = "133",
YEAR = "2025",
NUMBER = "4",
MONTH = "April",
PAGES = "1435-1455",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132412"}
@inproceedings{bb136421,
AUTHOR = "Wang, J.N. and Xu, W.J. and Wang, Q.Z. and Chan, A.B.",
TITLE = "Compare and Reweight:
Distinctive Image Captioning Using Similar Images Sets",
BOOKTITLE = ECCV20,
YEAR = "2020",
PAGES = "I:370-386",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132413"}
@article{bb136422,
AUTHOR = "Liu, M.F. and Hu, H.J. and Li, L.J. and Yu, Y. and Guan, W.L.",
TITLE = "Chinese Image Caption Generation via Visual Attention and Topic
Modeling",
JOURNAL = Cyber,
VOLUME = "52",
YEAR = "2022",
NUMBER = "2",
MONTH = "February",
PAGES = "1247-1257",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132414"}
@article{bb136423,
AUTHOR = "Li, X. and Zhang, W.K. and Sun, X. and Gao, X.",
TITLE = "Without detection: Two-step clustering features with local-global
attention for image captioning",
JOURNAL = IET-CV,
VOLUME = "16",
YEAR = "2022",
NUMBER = "3",
PAGES = "280-294",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132415"}
@article{bb136424,
AUTHOR = "Yu, L.T. and Zhang, J. and Wu, Q.",
TITLE = "Dual Attention on Pyramid Feature Maps for Image Captioning",
JOURNAL = MultMed,
VOLUME = "24",
YEAR = "2022",
NUMBER = "2022",
PAGES = "1775-1786",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132416"}
@article{bb136425,
AUTHOR = "Shao, X.J. and Xiang, Z.L. and Li, Y.X. and Zhang, M.J.",
TITLE = "Variational joint self-attention for image captioning",
JOURNAL = IET-IPR,
VOLUME = "16",
YEAR = "2022",
NUMBER = "8",
PAGES = "2075-2086",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132417"}
@article{bb136426,
AUTHOR = "Ma, Y.W. and Ji, J.Y. and Sun, X.S. and Zhou, Y. and Ji, R.R.",
TITLE = "Towards local visual modeling for image captioning",
JOURNAL = PR,
VOLUME = "138",
YEAR = "2023",
PAGES = "109420",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132418"}
@article{bb136427,
AUTHOR = "Barati, A. and Farsi, H. and Mohamadzadeh, S.",
TITLE = "Integration of the latent variable knowledge into deep image
captioning with Bayesian modeling",
JOURNAL = IET-IPR,
VOLUME = "17",
YEAR = "2023",
NUMBER = "7",
PAGES = "2256-2271",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132419"}
@article{bb136428,
AUTHOR = "Ji, J.Y. and Huang, X.Y. and Sun, X.S. and Zhou, Y. and Luo, G. and Cao, L.J. and Liu, J.Z. and Shao, L. and Ji, R.R.",
TITLE = "Multi-Branch Distance-Sensitive Self-Attention Network for Image
Captioning",
JOURNAL = MultMed,
VOLUME = "25",
YEAR = "2023",
PAGES = "3962-3974",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132420"}
@article{bb136429,
AUTHOR = "Cornia, M. and Baraldi, L. and Tal, A. and Cucchiara, R.",
TITLE = "Fully-attentive iterative networks for region-based controllable
image and video captioning",
JOURNAL = CVIU,
VOLUME = "237",
YEAR = "2023",
PAGES = "103857",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132421"}
@article{bb136430,
AUTHOR = "Song, L.F. and Li, F. and Wang, Y. and Liu, Y. and Wang, Y.H. and Xiang, S.M.",
TITLE = "Image captioning: Semantic selection unit with stacked residual
attention",
JOURNAL = IVC,
VOLUME = "144",
YEAR = "2024",
PAGES = "104965",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132422"}
@article{bb136431,
AUTHOR = "Du, R. and Zhang, W.K. and Li, S. and Chen, J.L. and Guo, Z.",
TITLE = "Spatial guided image captioning: Guiding attention with object's
spatial interaction",
JOURNAL = IET-IPR,
VOLUME = "18",
YEAR = "2024",
NUMBER = "12",
PAGES = "3368-3380",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132423"}
@article{bb136432,
AUTHOR = "Zhang, X.D. and Jia, A. and Ji, J.Z. and Qu, L.Q. and Ye, Q.X.",
TITLE = "Intra- and Inter-Head Orthogonal Attention for Image Captioning",
JOURNAL = IP,
VOLUME = "34",
YEAR = "2025",
PAGES = "594-607",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132424"}
@article{bb136433,
AUTHOR = "Song, L.F. and Wang, Y. and Shi, L. and Yu, J.Z. and Li, F. and Xiang, S.M.",
TITLE = "Transformer with token attention and attribute prediction for image
captioning",
JOURNAL = PRL,
VOLUME = "188",
YEAR = "2025",
PAGES = "74-80",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132425"}
@article{bb136434,
AUTHOR = "Parseh, M.J. and Ghadiri, S.",
TITLE = "Graph-based image captioning with semantic and spatial features",
JOURNAL = SP:IC,
VOLUME = "133",
YEAR = "2025",
PAGES = "117273",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132426"}
@inproceedings{bb136435,
AUTHOR = "Sui, J.H. and Yu, H.M. and Liang, X.Y. and Ping, P.",
TITLE = "Image Caption Method Based on Graph Attention Network with Global
Context",
BOOKTITLE = ICIVC22,
YEAR = "2022",
PAGES = "480-487",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132427"}
@inproceedings{bb136436,
AUTHOR = "Popattia, M. and Rafi, M. and Qureshi, R. and Nawaz, S.",
TITLE = "Guiding Attention using Partial-Order Relationships for Image
Captioning",
BOOKTITLE = MULA22,
YEAR = "2022",
PAGES = "4670-4679",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132428"}
@inproceedings{bb136437,
AUTHOR = "Deb, T. and Sadmanee, A. and Bhaumik, K.K. and Ali, A.A. and Amin, M.A. and Rahman, A.K.M.M.",
TITLE = "Variational Stacked Local Attention Networks for Diverse Video
Captioning",
BOOKTITLE = WACV22,
YEAR = "2022",
PAGES = "2493-2502",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132429"}
@inproceedings{bb136438,
AUTHOR = "Li, Z. and Tran, Q. and Mai, L. and Lin, Z. and Yuille, A.L.",
TITLE = "Context-Aware Group Captioning via Self-Attention and Contrastive
Features",
BOOKTITLE = CVPR20,
YEAR = "2020",
PAGES = "3437-3447",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132430"}
@inproceedings{bb136439,
AUTHOR = "Guo, L. and Liu, J. and Zhu, X. and Yao, P. and Lu, S. and Lu, H.",
TITLE = "Normalized and Geometry-Aware Self-Attention Network for Image
Captioning",
BOOKTITLE = CVPR20,
YEAR = "2020",
PAGES = "10324-10333",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132431"}
@inproceedings{bb136440,
AUTHOR = "Pan, Y. and Yao, T. and Li, Y. and Mei, T.",
TITLE = "X-Linear Attention Networks for Image Captioning",
BOOKTITLE = CVPR20,
YEAR = "2020",
PAGES = "10968-10977",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132432"}
@inproceedings{bb136441,
AUTHOR = "Park, G. and Han, C. and Kim, D. and Yoon, W.J.",
TITLE = "MHSAN: Multi-Head Self-Attention Network for Visual Semantic
Embedding",
BOOKTITLE = WACV20,
YEAR = "2020",
PAGES = "1507-1515",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132433"}
@inproceedings{bb136442,
AUTHOR = "He, S. and Tavakoli, H.R. and Borji, A. and Pugeault, N.",
TITLE = "Human Attention in Image Captioning: Dataset and Analysis",
BOOKTITLE = ICCV19,
YEAR = "2019",
PAGES = "8528-8537",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132434"}
@inproceedings{bb136443,
AUTHOR = "Huang, L. and Wang, W. and Chen, J. and Wei, X.",
TITLE = "Attention on Attention for Image Captioning",
BOOKTITLE = ICCV19,
YEAR = "2019",
PAGES = "4633-4642",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132435"}
@inproceedings{bb136444,
AUTHOR = "Wei, H.Y. and Li, Z.X. and Zhang, C.L.",
TITLE = "Image Captioning Based on Visual and Semantic Attention",
BOOKTITLE = MMMod20,
YEAR = "2020",
PAGES = "I:151-162",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132436"}
@inproceedings{bb136445,
AUTHOR = "Fukui, H. and Hirakawa, T. and Yamashita, T. and Fujiyoshi, H.",
TITLE = "Attention Branch Network: Learning of Attention Mechanism for Visual
Explanation",
BOOKTITLE = CVPR19,
YEAR = "2019",
PAGES = "10697-10706",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132437"}
@inproceedings{bb136446,
AUTHOR = "Huang, Y. and Li, C. and Li, T. and Wan, W. and Chen, J.",
TITLE = "Image Captioning with Attribute Refinement",
BOOKTITLE = ICIP19,
YEAR = "2019",
PAGES = "1820-1824",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132438"}
@inproceedings{bb136447,
AUTHOR = "Shi, J. and Li, Y. and Wang, S.",
TITLE = "Cascade Attention: Multiple Feature Based Learning for Image
Captioning",
BOOKTITLE = ICIP19,
YEAR = "2019",
PAGES = "1970-1974",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132439"}
@inproceedings{bb136448,
AUTHOR = "Xiao, H. and Shi, J.",
TITLE = "A Novel Attribute Selection Mechanism for Video Captioning",
BOOKTITLE = ICIP19,
YEAR = "2019",
PAGES = "619-623",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132440"}
@inproceedings{bb136449,
AUTHOR = "Wang, Q.Z. and Chan, A.B.",
TITLE = "Gated Hierarchical Attention for Image Captioning",
BOOKTITLE = ACCV18,
YEAR = "2018",
PAGES = "IV:21-37",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132441"}
@inproceedings{bb136450,
AUTHOR = "Wang, W.X. and Chen, Z.H. and Hu, H.F.",
TITLE = "Multivariate Attention Network for Image Captioning",
BOOKTITLE = ACCV18,
YEAR = "2018",
PAGES = "VI:587-602",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132442"}
@inproceedings{bb136451,
AUTHOR = "Ghanimifard, M. and Dobnik, S.",
TITLE = "Knowing When to Look for What and Where: Evaluating Generation of
Spatial Descriptions with Adaptive Attention",
BOOKTITLE = VL18,
YEAR = "2018",
PAGES = "IV:153-161",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132443"}
@inproceedings{bb136452,
AUTHOR = "Khademi, M. and Schulte, O.",
TITLE = "Image Caption Generation with Hierarchical Contextual Visual Spatial
Attention",
BOOKTITLE = Cognitive18,
YEAR = "2018",
PAGES = "2024-20248",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132444"}
@inproceedings{bb136453,
AUTHOR = "Wang, F. and Gong, X. and Huang, L.",
TITLE = "Time-Dependent Pre-attention Model for Image Captioning",
BOOKTITLE = ICPR18,
YEAR = "2018",
PAGES = "3297-3302",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132445"}
@inproceedings{bb136454,
AUTHOR = "Chen, S. and Zhao, Q.",
TITLE = "Boosted Attention: Leveraging Human Attention for Image Captioning",
BOOKTITLE = ECCV18,
YEAR = "2018",
PAGES = "XI: 72-88",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132446"}
@inproceedings{bb136455,
AUTHOR = "Fang, F. and Wang, H. and Tang, P.",
TITLE = "Image Captioning with Word Level Attention",
BOOKTITLE = ICIP18,
YEAR = "2018",
PAGES = "1278-1282",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132447"}
@inproceedings{bb136456,
AUTHOR = "Zhu, Z. and Xue, Z. and Yuan, Z.",
TITLE = "Topic-Guided Attention for Image Captioning",
BOOKTITLE = ICIP18,
YEAR = "2018",
PAGES = "2615-2619",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132448"}
@inproceedings{bb136457,
AUTHOR = "Pedersoli, M. and Lucas, T. and Schmid, C. and Verbeek, J.",
TITLE = "Areas of Attention for Image Captioning",
BOOKTITLE = ICCV17,
YEAR = "2017",
PAGES = "1251-1259",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132449"}
@inproceedings{bb136458,
AUTHOR = "Tavakoliy, H.R. and Shetty, R. and Borji, A. and Laaksonen, J.",
TITLE = "Paying Attention to Descriptions Generated by Image Captioning Models",
BOOKTITLE = ICCV17,
YEAR = "2017",
PAGES = "2506-2515",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132450"}
@inproceedings{bb136459,
AUTHOR = "Lu, J. and Xiong, C. and Parikh, D. and Socher, R.",
TITLE = "Knowing When to Look: Adaptive Attention via a Visual Sentinel for
Image Captioning",
BOOKTITLE = CVPR17,
YEAR = "2017",
PAGES = "3242-3250",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132451"}
@inproceedings{bb136460,
AUTHOR = "Chen, L. and Zhang, H. and Xiao, J. and Nie, L. and Shao, J. and Liu, W. and Chua, T.S.",
TITLE = "SCA-CNN: Spatial and Channel-Wise Attention in Convolutional Networks
for Image Captioning",
BOOKTITLE = CVPR17,
YEAR = "2017",
PAGES = "6298-6306",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132452"}
@inproceedings{bb136461,
AUTHOR = "Zanfir, M. and Marinoiu, E. and Sminchisescu, C.",
TITLE = "Spatio-Temporal Attention Models for Grounded Video Captioning",
BOOKTITLE = ACCV16,
YEAR = "2016",
PAGES = "IV: 104-119",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132453"}
@inproceedings{bb136462,
AUTHOR = "Chen, T.H. and Zeng, K.H. and Hsu, W.T. and Sun, M.",
TITLE = "Video Captioning via Sentence Augmentation and Spatio-Temporal
Attention",
BOOKTITLE = Assist16,
YEAR = "2016",
PAGES = "I: 269-286",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132454"}
@inproceedings{bb136463,
AUTHOR = "Chen, T.L. and Zhang, Z.P. and You, Q.Z. and Fang, C. and Wang, Z.W. and Jin, H.L. and Luo, J.B.",
TITLE = "'Factual' or 'Emotional':
Stylized Image Captioning with Adaptive Learning and Attention",
BOOKTITLE = ECCV18,
YEAR = "2018",
PAGES = "X: 527-543",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132455"}
@inproceedings{bb136464,
AUTHOR = "You, Q.Z. and Jin, H.L. and Wang, Z.W. and Fang, C. and Luo, J.B.",
TITLE = "Image Captioning with Semantic Attention",
BOOKTITLE = CVPR16,
YEAR = "2016",
PAGES = "4651-4659",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607attic3.html#TT132456"}
@article{bb136465,
AUTHOR = "Lu, X. and Wang, B. and Zheng, X. and Li, X.",
TITLE = "Exploring Models and Data for Remote Sensing Image Caption Generation",
JOURNAL = GeoRS,
VOLUME = "56",
YEAR = "2018",
NUMBER = "4",
MONTH = "April",
PAGES = "2183-2195",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132457"}
@article{bb136466,
AUTHOR = "Zhang, X.R. and Wang, X. and Tang, X. and Zhou, H.Y. and Li, C.",
TITLE = "Description Generation for Remote Sensing Images Using Attribute
Attention Mechanism",
JOURNAL = RS,
VOLUME = "11",
YEAR = "2019",
NUMBER = "6",
PAGES = "xx-yy",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132458"}
@article{bb136467,
AUTHOR = "Zhang, Z.Y. and Diao, W.H. and Zhang, W.K. and Yan, M.L. and Gao, X. and Sun, X.",
TITLE = "LAM: Remote Sensing Image Captioning with Label-Attention Mechanism",
JOURNAL = RS,
VOLUME = "11",
YEAR = "2019",
NUMBER = "20",
PAGES = "xx-yy",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132459"}
@article{bb136468,
AUTHOR = "Fu, K. and Li, Y. and Zhang, W.K. and Yu, H.F. and Sun, X.",
TITLE = "Boosting Memory with a Persistent Memory Mechanism for Remote Sensing
Image Captioning",
JOURNAL = RS,
VOLUME = "12",
YEAR = "2020",
NUMBER = "11",
PAGES = "xx-yy",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132460"}
@article{bb136469,
AUTHOR = "Lu, X. and Wang, B. and Zheng, X.",
TITLE = "Sound Active Attention Framework for Remote Sensing Image Captioning",
JOURNAL = GeoRS,
VOLUME = "58",
YEAR = "2020",
NUMBER = "3",
MONTH = "March",
PAGES = "1985-2000",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132461"}
@article{bb136470,
AUTHOR = "Li, Y.Y. and Fang, S.K. and Jiao, L.C. and Liu, R.J. and Shang, R.H.",
TITLE = "A Multi-Level Attention Model for Remote Sensing Image Captions",
JOURNAL = RS,
VOLUME = "12",
YEAR = "2020",
NUMBER = "6",
PAGES = "xx-yy",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132462"}
@article{bb136471,
AUTHOR = "Li, X.L. and Zhang, X.T. and Huang, W. and Wang, Q.",
TITLE = "Truncation Cross Entropy Loss for Remote Sensing Image Captioning",
JOURNAL = GeoRS,
VOLUME = "59",
YEAR = "2021",
NUMBER = "6",
MONTH = "June",
PAGES = "5246-5257",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132463"}
@article{bb136472,
AUTHOR = "Sumbul, G. and Nayak, S. and Demir, B.",
TITLE = "SD-RSIC: Summarization-Driven Deep Remote Sensing Image Captioning",
JOURNAL = GeoRS,
VOLUME = "59",
YEAR = "2021",
NUMBER = "8",
MONTH = "August",
PAGES = "6922-6934",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132464"}
@article{bb136473,
AUTHOR = "Wang, Q. and Huang, W. and Zhang, X.T. and Li, X.L.",
TITLE = "Word-Sentence Framework for Remote Sensing Image Captioning",
JOURNAL = GeoRS,
VOLUME = "59",
YEAR = "2021",
NUMBER = "12",
MONTH = "December",
PAGES = "10532-10543",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132465"}
@article{bb136474,
AUTHOR = "Yang, Q.Q. and Ni, Z.H. and Ren, P.",
TITLE = "Meta captioning:
A meta learning based remote sensing image captioning framework",
JOURNAL = PandRS,
VOLUME = "186",
YEAR = "2022",
PAGES = "190-200",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132466"}
@article{bb136475,
AUTHOR = "Liu, Z.Y. and Dong, A.M. and Yu, J.G. and Han, Y.B. and Zhou, Y. and Zhao, K.",
TITLE = "Scene classification for remote sensing images with self-attention
augmented CNN",
JOURNAL = IET-IPR,
VOLUME = "16",
YEAR = "2022",
NUMBER = "11",
PAGES = "3085-3096",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132467"}
@article{bb136476,
AUTHOR = "Zhou, H.N. and Du, X.P. and Xia, L. and Li, S.",
TITLE = "Self-Learning for Few-Shot Remote Sensing Image Captioning",
JOURNAL = RS,
VOLUME = "14",
YEAR = "2022",
NUMBER = "18",
PAGES = "xx-yy",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132468"}
@article{bb136477,
AUTHOR = "Wang, Q. and Huang, W. and Zhang, X.T. and Li, X.L.",
TITLE = "GLCM: Global-Local Captioning Model for Remote Sensing Image
Captioning",
JOURNAL = Cyber,
VOLUME = "53",
YEAR = "2023",
NUMBER = "11",
MONTH = "November",
PAGES = "6910-6922",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132469"}
@article{bb136478,
AUTHOR = "Yang, T. and Zhou, Q. and Wang, Q.",
TITLE = "DIA: Deriving linguistic information from auxiliary languages for
remote sensing image captioning",
JOURNAL = PR,
VOLUME = "171",
YEAR = "2026",
PAGES = "112209",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132470"}
@article{bb136479,
AUTHOR = "Cheng, Q. and Xu, Y.Q. and Huang, Z.Y.",
TITLE = "VCC-DiffNet: Visual Conditional Control Diffusion Network for Remote
Sensing Image Captioning",
JOURNAL = RS,
VOLUME = "16",
YEAR = "2024",
NUMBER = "16",
PAGES = "2961",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132471"}
@article{bb136480,
AUTHOR = "Li, Y.P. and Zhang, X.R. and Zhang, T.Y. and Wang, G.C. and Wang, X.L. and Li, S.",
TITLE = "A Patch-Level Region-Aware Module with a Multi-Label Framework for
Remote Sensing Image Captioning",
JOURNAL = RS,
VOLUME = "16",
YEAR = "2024",
NUMBER = "21",
PAGES = "3987",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132472"}
@article{bb136481,
AUTHOR = "Zhang, K. and Li, P. and Wang, J.Q.",
TITLE = "A Review of Deep Learning-Based Remote Sensing Image Caption:
Methods, Models, Comparisons and Future Directions",
JOURNAL = RS,
VOLUME = "16",
YEAR = "2024",
NUMBER = "21",
PAGES = "4113",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132473"}
@article{bb136482,
AUTHOR = "Leng, G. and Xiong, Y.J. and Qiu, C.P. and Guo, C.Z.",
TITLE = "Discrete diffusion models with Refined Language-Image Pre-trained
representations for remote sensing image captioning",
JOURNAL = PRL,
VOLUME = "186",
YEAR = "2024",
PAGES = "164-169",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132474"}
@article{bb136483,
AUTHOR = "Li, Y.P. and Zhang, X.R. and Wang, G.C. and Zhang, T.Y.",
TITLE = "Exploring Difference Semantic Prior Guidance for Remote Sensing Image
Change Captioning",
JOURNAL = RS,
VOLUME = "18",
YEAR = "2026",
NUMBER = "2",
PAGES = "232",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132475"}
@article{bb136484,
AUTHOR = "Wang, S.A. and Ye, X.T. and Gu, Y. and Wang, J.H. and Meng, Y. and Tian, J.X. and Hou, B. and Jiao, L.C.",
TITLE = "Multi-Label Semantic Feature Fusion for Remote Sensing Image
Captioning",
JOURNAL = PandRS,
VOLUME = "184",
YEAR = "2022",
PAGES = "1-18",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132476"}
@article{bb136485,
AUTHOR = "Han, X. and Wu, Z.J. and Li, Y.P. and Zhang, X.R. and Wang, G.C. and Hou, B.",
TITLE = "CSSA: A Cross-Modal Spatial-Semantic Alignment Framework for Remote
Sensing Image Captioning",
JOURNAL = RS,
VOLUME = "18",
YEAR = "2026",
NUMBER = "3",
PAGES = "522",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132477"}
@article{bb136486,
AUTHOR = "Zhang, X.R. and Li, Y.P. and Wang, X. and Liu, F.X. and Wu, Z.J. and Cheng, X. and Jiao, L.C.",
TITLE = "Multi-Source Interactive Stair Attention for Remote Sensing Image
Captioning",
JOURNAL = RS,
VOLUME = "15",
YEAR = "2023",
NUMBER = "3",
PAGES = "xx-yy",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132478"}
@article{bb136487,
AUTHOR = "Li, Y.P. and Zhang, X.R. and Cheng, X. and Tang, X. and Jiao, L.C.",
TITLE = "Learning Consensus-Aware Semantic Knowledge for Remote Sensing Image
Captioning",
JOURNAL = PR,
VOLUME = "145",
YEAR = "2024",
PAGES = "109893",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132479"}
@article{bb136488,
AUTHOR = "Guo, Z. and Liu, H.M. and Ren, Z. and Jiao, L.C. and Gou, S.P. and Li, R.M.",
TITLE = "Attribute-Based Learning for Remote Sensing Image Captioning in
Unseen Scenes",
JOURNAL = RS,
VOLUME = "17",
YEAR = "2025",
NUMBER = "7",
PAGES = "1237",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132480"}
@article{bb136489,
AUTHOR = "Mehmood, M. and Shahzad, A. and Hussain, F. and Caceres Najarro, L.A. and Usman, M.",
TITLE = "Remote Sensing Image Captioning via Self-Supervised DINOv3 and
Transformer Fusion",
JOURNAL = RS,
VOLUME = "18",
YEAR = "2026",
NUMBER = "6",
PAGES = "846",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132481"}
@inproceedings{bb136490,
AUTHOR = "Wei, Y.C. and Li, L. and Geng, S.L.",
TITLE = "Remote Sensing Image Captioning Using Hire-MLP",
BOOKTITLE = CVIDL23,
YEAR = "2023",
PAGES = "109-112",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132482"}
@inproceedings{bb136491,
AUTHOR = "Chavhan, R. and Banerjee, B. and Zhu, X.X. and Chaudhuri, S.",
TITLE = "A Novel Actor Dual-Critic Model for Remote Sensing Image Captioning",
BOOKTITLE = ICPR21,
YEAR = "2021",
PAGES = "4918-4925",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607rsic2.html#TT132483"}
@article{bb136492,
AUTHOR = "Nakayama, H. and Harada, T. and Kuniyoshi, Y.",
TITLE = "Dense Sampling Low-Level Statistics of Local Features",
JOURNAL = IEICE,
VOLUME = "E93-D",
YEAR = "2010",
NUMBER = "7",
MONTH = "July",
PAGES = "1727-1736",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132484"}
@inproceedings{bb136493,
AUTHOR = "Kuniyoshi, Y. and Harada, T. and Nakayama, H.",
TITLE = "Dense Sampling Low-Level Statistics of Local Features",
BOOKTITLE = CIVR09,
YEAR = "2009",
PAGES = "Article No 17",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132484"}
@inproceedings{bb136494,
AUTHOR = "Nakayama, H. and Harada, T. and Kuniyoshi, Y.",
TITLE = "Global Gaussian approach for scene categorization using information
geometry",
BOOKTITLE = CVPR10,
YEAR = "2010",
PAGES = "2336-2343",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132485"}
@inproceedings{bb136495,
AUTHOR = "Nakayama, H. and Harada, T. and Kuniyoshi, Y.",
TITLE = "AI Goggles: Real-time Description and Retrieval in the Real World with
Online Learning",
BOOKTITLE = CRV09,
YEAR = "2009",
PAGES = "184-191",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132486"}
@inproceedings{bb136496,
AUTHOR = "Ushiku, Y. and Yamaguchi, M. and Mukuta, Y. and Harada, T.",
TITLE = "Common Subspace for Model and Similarity:
Phrase Learning for Caption Generation from Images",
BOOKTITLE = ICCV15,
YEAR = "2015",
PAGES = "2668-2676",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132487"}
@inproceedings{bb136497,
AUTHOR = "Harada, T. and Nakayama, H. and Kuniyoshi, Y.",
TITLE = "Improving Local Descriptors by Embedding Global and Local Spatial
Information",
BOOKTITLE = ECCV10,
YEAR = "2010",
PAGES = "IV: 736-749",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132488"}
@inproceedings{bb136498,
AUTHOR = "Nakayama, H. and Harada, T. and Kuniyoshi, Y.",
TITLE = "Evaluation of dimensionality reduction methods for image
auto-annotation",
BOOKTITLE = BMVC10,
YEAR = "2010",
PAGES = "xx-yy",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132489"}
@inproceedings{bb136499,
AUTHOR = "Jin, J. and Nakayama, H.",
TITLE = "Annotation order matters:
Recurrent Image Annotator for arbitrary length image tagging",
BOOKTITLE = ICPR16,
YEAR = "2016",
PAGES = "2452-2457",
BIBSOURCE = "http://www.visionbib.com/bibliography/match607ian2.html#TT132490"}
Last update:May 24, 2026 at 14:46:09