@inproceedings{bb130900,
        AUTHOR = "Niu, Z.X. and Zhou, M. and Wang, L. and Gao, X.B. and Hua, G.",
        TITLE = "Hierarchical Multimodal LSTM for Dense Visual-Semantic Embedding",
        BOOKTITLE = ICCV17,
        YEAR = "2017",
        PAGES = "1899-1907",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607lscap4.html#TT126940"}

@inproceedings{bb130901,
        AUTHOR = "Tan, Y.H. and Chan, C.S.",
        TITLE = "phi-LSTM: A Phrase-Based Hierarchical LSTM Model for Image Captioning",
        BOOKTITLE = ACCV16,
        YEAR = "2016",
        PAGES = "V: 101-117",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607lscap4.html#TT126941"}

@inproceedings{bb130902,
        AUTHOR = "Wang, M. and Song, L. and Yang, X.K. and Luo, C.F.",
        TITLE = "A parallel-fusion RNN-LSTM architecture for image caption generation",
        BOOKTITLE = ICIP16,
        YEAR = "2016",
        PAGES = "4448-4452",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607lscap4.html#TT126942"}

@article{bb130903,
        AUTHOR = "Verma, Y. and Jawahar, C.V.",
        TITLE = "A support vector approach for cross-modal search of images and texts",
        JOURNAL = CVIU,
        VOLUME = "154",
        YEAR = "2017",
        NUMBER = "1",
        PAGES = "48-63",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126943"}

@inproceedings{bb130904,
        AUTHOR = "Dutta, A. and Verma, Y. and Jawahar, C.V.",
        TITLE = "Recurrent Image Annotation with Explicit Inter-Label Dependencies",
        BOOKTITLE = ECCV20,
        YEAR = "2020",
        PAGES = "XXIX: 191-207",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126944"}

@article{bb130905,
        AUTHOR = "Xue, J.F. and Eguchi, K.",
        TITLE = "Video Data Modeling Using Sequential Correspondence Hierarchical
Dirichlet Processes",
        JOURNAL = IEICE,
        VOLUME = "E100-D",
        YEAR = "2017",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "33-41",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126945"}

@article{bb130906,
        AUTHOR = "Liu, A.A. and Xu, N. and Wong, Y.K. and Li, J. and Su, Y.T. and Kankanhalli, M.",
        TITLE = "Hierarchical & multimodal video captioning: Discovering and
transferring multimodal knowledge for vision to language",
        JOURNAL = CVIU,
        VOLUME = "163",
        YEAR = "2017",
        NUMBER = "1",
        PAGES = "113-125",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126946"}

@article{bb130907,
        AUTHOR = "Guan, J.N. and Wang, E.",
        TITLE = "Repeated review based image captioning for image evidence review",
        JOURNAL = SP:IC,
        VOLUME = "63",
        YEAR = "2018",
        PAGES = "141-148",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126947"}

@article{bb130908,
        AUTHOR = "Hu, M. and Yang, Y. and Shen, F. and Zhang, L. and Shen, H.T. and Li, X.",
        TITLE = "Robust Web Image Annotation via Exploring Multi-Facet and Structural
Knowledge",
        JOURNAL = IP,
        VOLUME = "26",
        YEAR = "2017",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "4871-4884",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126948"}

@article{bb130909,
        AUTHOR = "Gil Gonzalez, J. and Alvarez Meza, A. and Orozco Gutierrez, A.",
        TITLE = "Learning from multiple annotators using kernel alignment",
        JOURNAL = PRL,
        VOLUME = "116",
        YEAR = "2018",
        PAGES = "150-156",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126949"}

@article{bb130910,
        AUTHOR = "Zheng, H. and Wu, J.H. and Liang, R. and Li, Y. and Li, X.Z.",
        TITLE = "Multi-task learning for captioning images with novel words",
        JOURNAL = IET-CV,
        VOLUME = "13",
        YEAR = "2019",
        NUMBER = "3",
        MONTH = "April",
        PAGES = "294-301",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126950"}

@article{bb130911,
        AUTHOR = "Park, C.C. and Kim, B. and Kim, G.",
        TITLE = "Towards Personalized Image Captioning via Multimodal Memory Networks",
        JOURNAL = PAMI,
        VOLUME = "41",
        YEAR = "2019",
        NUMBER = "4",
        MONTH = "April",
        PAGES = "999-1012",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126951"}

@inproceedings{bb130912,
        AUTHOR = "Park, C.C. and Kim, B. and Kim, G.",
        TITLE = "Attend to You: Personalized Image Captioning with Context Sequence
Memory Networks",
        BOOKTITLE = CVPR17,
        YEAR = "2017",
        PAGES = "6432-6440",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126952"}

@article{bb130913,
        AUTHOR = "Yang, M. and Zhao, W. and Xu, W. and Feng, Y. and Zhao, Z. and Chen, X. and Lei, K.",
        TITLE = "Multitask Learning for Cross-Domain Image Captioning",
        JOURNAL = MultMed,
        VOLUME = "21",
        YEAR = "2019",
        NUMBER = "4",
        MONTH = "April",
        PAGES = "1047-1061",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126953"}

@article{bb130914,
        AUTHOR = "Yu, N. and Hu, X. and Song, B. and Yang, J. and Zhang, J.",
        TITLE = "Topic-Oriented Image Captioning Based on Order-Embedding",
        JOURNAL = IP,
        VOLUME = "28",
        YEAR = "2019",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "2743-2754",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126954"}

@article{bb130915,
        AUTHOR = "Li, X. and Xu, C. and Wang, X. and Lan, W. and Jia, Z. and Yang, G. and Xu, J.",
        TITLE = "COCO-CN for Cross-Lingual Image Tagging, Captioning, and Retrieval",
        JOURNAL = MultMed,
        VOLUME = "21",
        YEAR = "2019",
        NUMBER = "9",
        MONTH = "September",
        PAGES = "2347-2360",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126955"}

@article{bb130916,
        AUTHOR = "Tian, C. and Tian, M. and Jiang, M.M. and Liu, H. and Deng, D.H.",
        TITLE = "How much do cross-modal related semantics benefit image captioning by
weighting attributes and re-ranking sentences?",
        JOURNAL = PRL,
        VOLUME = "125",
        YEAR = "2019",
        PAGES = "639-645",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126956"}

@article{bb130917,
        AUTHOR = "Niu, Y. and Lu, Z. and Wen, J. and Xiang, T. and Chang, S.",
        TITLE = "Multi-Modal Multi-Scale Deep Learning for Large-Scale Image
Annotation",
        JOURNAL = IP,
        VOLUME = "28",
        YEAR = "2019",
        NUMBER = "4",
        MONTH = "April",
        PAGES = "1720-1731",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126957"}

@article{bb130918,
        AUTHOR = "Huang, Y. and Chen, J. and Ouyang, W. and Wan, W. and Xue, Y.",
        TITLE = "Image Captioning With End-to-End Attribute Detection and Subsequent
Attributes Prediction",
        JOURNAL = IP,
        VOLUME = "29",
        YEAR = "2020",
        PAGES = "4013-4026",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126958"}

@article{bb130919,
        AUTHOR = "Zhao, W. and Wu, X. and Luo, J.",
        TITLE = "Cross-Domain Image Captioning via Cross-Modal Retrieval and Model
Adaptation",
        JOURNAL = IP,
        VOLUME = "30",
        YEAR = "2021",
        PAGES = "1180-1192",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126959"}

@article{bb130920,
        AUTHOR = "Wang, H. and Du, Y.T. and Zhang, G.X. and Cai, Z.M. and Su, C.",
        TITLE = "Learning Fundamental Visual Concepts Based on Evolved Multi-Edge
Concept Graph",
        JOURNAL = MultMed,
        VOLUME = "23",
        YEAR = "2021",
        PAGES = "4400-4413",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126960"}

@article{bb130921,
        AUTHOR = "Zhang, J. and Mei, K. and Zheng, Y. and Fan, J.",
        TITLE = "Integrating Part of Speech Guidance for Image Captioning",
        JOURNAL = MultMed,
        VOLUME = "23",
        YEAR = "2021",
        PAGES = "92-104",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126961"}

@article{bb130922,
        AUTHOR = "Kim, D.J. and Oh, T.H. and Choi, J. and Kweon, I.S.",
        TITLE = "Dense Relational Image Captioning via Multi-Task Triple-Stream
Networks",
        JOURNAL = PAMI,
        VOLUME = "44",
        YEAR = "2022",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "7348-7362",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126962"}

@inproceedings{bb130923,
        AUTHOR = "Kim, D.J. and Choi, J. and Oh, T.H. and Kweon, I.S.",
        TITLE = "Dense Relational Captioning: Triple-Stream Networks for
Relationship-Based Captioning",
        BOOKTITLE = CVPR19,
        YEAR = "2019",
        PAGES = "6264-6273",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126963"}

@article{bb130924,
        AUTHOR = "Nguyen, T.S. and Fernando, B.",
        TITLE = "Effective Multimodal Encoding for Image Paragraph Captioning",
        JOURNAL = IP,
        VOLUME = "31",
        YEAR = "2022",
        PAGES = "6381-6395",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126964"}

@article{bb130925,
        AUTHOR = "Duan, Y.Q. and Wang, Z. and Li, Y. and Wang, J.Y.",
        TITLE = "Cross-domain multi-style merge for image captioning",
        JOURNAL = CVIU,
        VOLUME = "228",
        YEAR = "2023",
        PAGES = "103617",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126965"}

@article{bb130926,
        AUTHOR = "Wu, X.X. and Li, T.",
        TITLE = "Sentimental Visual Captioning using Multimodal Transformer",
        JOURNAL = IJCV,
        VOLUME = "131",
        YEAR = "2023",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "1073-1090",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126966"}

@article{bb130927,
        AUTHOR = "Ding, Z.W. and Lan, G.L. and Song, Y.Z. and Yang, Z.W.",
        TITLE = "SGIR: Star Graph-Based Interaction for Efficient and Robust
Multimodal Representation",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "4217-4229",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126967"}

@article{bb130928,
        AUTHOR = "Zhao, W.T. and Wu, X.X.",
        TITLE = "Boosting Entity-Aware Image Captioning With Multi-Modal Knowledge
Graph",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "2659-2670",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126968"}

@article{bb130929,
        AUTHOR = "Gao, J.L. and Li, J. and Jia, C.M. and Wang, S.S. and Ma, S.W. and Gao, W.",
        TITLE = "Cross Modal Compression With Variable Rate Prompt",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "3444-3456",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126969"}

@article{bb130930,
        AUTHOR = "Gao, J.L. and Jia, C.M. and Huang, Z.M. and Wang, S.S. and Ma, S.W. and Gao, W.",
        TITLE = "Rate-Distortion Optimized Cross Modal Compression With Multiple
Domains",
        JOURNAL = CirSysVideo,
        VOLUME = "34",
        YEAR = "2024",
        NUMBER = "8",
        MONTH = "August",
        PAGES = "6978-6992",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126970"}

@article{bb130931,
        AUTHOR = "Cao, S. and An, G. and Cen, Y.G. and Yang, Z.Q. and Lin, W.S.",
        TITLE = "CAST: Cross-Modal Retrieval and Visual Conditioning for image
captioning",
        JOURNAL = PR,
        VOLUME = "153",
        YEAR = "2024",
        PAGES = "110555",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126971"}

@article{bb130932,
        AUTHOR = "Song, Z.J. and Hu, Z.Z. and Zhou, Y. and Zhao, Y. and Hong, R.C. and Wang, M.",
        TITLE = "Embedded Heterogeneous Attention Transformer for Cross-Lingual Image
Captioning",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "9008-9020",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126972"}

@article{bb130933,
        AUTHOR = "Li, Y. and Ji, J.Y. and Sun, X.S. and Zhou, Y. and Luo, Y.P. and Ji, R.R.",
        TITLE = "M3ixup: A multi-modal data augmentation approach for image captioning",
        JOURNAL = PR,
        VOLUME = "158",
        YEAR = "2025",
        PAGES = "110941",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126973"}

@article{bb130934,
        AUTHOR = "Deng, H.Y. and Xie, Y.S. and Wang, Q. and Wang, J.J. and Ruan, W.J. and Liu, W. and Liu, Y.J.",
        TITLE = "CDKM: Common and Distinct Knowledge Mining Network With Content
Interaction for Dense Captioning",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "10462-10473",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126974"}

@article{bb130935,
        AUTHOR = "Zhang, G.Q. and Kan, S.C. and Shi, L. and Xu, W. and An, G. and Cen, Y.G.",
        TITLE = "Cross-scene visual context parsing with large vision-language model",
        JOURNAL = PR,
        VOLUME = "166",
        YEAR = "2025",
        PAGES = "111641",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126975"}

@inproceedings{bb130936,
        AUTHOR = "Chen, L. and Li, J.S. and Dong, X.Y. and Zhang, P. and He, C.H. and Wang, J.Q. and Zhao, F. and Lin, D.",
        TITLE = "ShareGPT4V: Improving Large Multi-modal Models with Better Captions",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XVII: 370-387",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126976"}

@inproceedings{bb130937,
        AUTHOR = "Jin, B. and Zheng, Y.P. and Li, P.F. and Li, W. and Zheng, Y.H. and Hu, S. and Liu, X.Y. and Zhu, J. and Yan, Z.J. and Sun, H.Y. and Zhan, K. and Jia, P. and Long, X.X. and Chen, Y.L. and Zhao, H.",
        TITLE = "Tod3cap: Towards 3d Dense Captioning in Outdoor Scenes",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XVIII: 367-384",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126977"}

@inproceedings{bb130938,
        AUTHOR = "Kim, M.J. and Lim, H.S. and Lee, S. and Kim, B. and Kim, G.",
        TITLE = "Bi-directional Contextual Attention for 3d Dense Captioning",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XVIII: 385-401",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126978"}

@inproceedings{bb130939,
        AUTHOR = "Zhao, Y.Z. and Liu, Y. and Guo, Z. and Wu, W.J. and Gong, C. and Ye, Q.X. and Wan, F.",
        TITLE = "Controlcap: Controllable Region-level Captioning",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XXXVIII: 21-38",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126979"}

@inproceedings{bb130940,
        AUTHOR = "Wang, Z. and Jiang, X.Y. and Xiao, J. and Chen, T. and Chen, L.",
        TITLE = "Decap: Towards Generalized Explicit Caption Editing via Diffusion
Mechanism",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XLIII: 365-381",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126980"}

@inproceedings{bb130941,
        AUTHOR = "Mao, S.Q. and Zhang, C.Y. and Su, H. and Song, H. and Shalyminov, I. and Cai, W.D.",
        TITLE = "Controllable Contextualized Image Captioning: Directing the Visual
Narrative Through User-defined Highlights",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "L: 464-481",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126981"}

@inproceedings{bb130942,
        AUTHOR = "Sarto, S. and Cornia, M. and Baraldi, L. and Cucchiara, R.",
        TITLE = "Bridge: Bridging Gaps in Image Captioning Evaluation with Stronger
Visual Cues",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "LXXVIII: 70-87",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126982"}

@inproceedings{bb130943,
        AUTHOR = "Matsuda, K. and Wada, Y. and Sugiura, K.",
        TITLE = "DENEB: A Hallucination-robust Automatic Evaluation Metric for Image
Captioning",
        BOOKTITLE = ACCV24,
        YEAR = "2024",
        PAGES = "III: 166-182",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126983"}

@inproceedings{bb130944,
        AUTHOR = "Hu, J.C. and Cavicchioli, R. and Capotondi, A.",
        TITLE = "A Request for Clarity over the End of Sequence Token in the
Self-critical Sequence Training",
        BOOKTITLE = CIAP23,
        YEAR = "2023",
        PAGES = "I:39-50",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126984"}

@inproceedings{bb130945,
        AUTHOR = "Hu, W.Z. and Wang, L.X. and Xu, L.F.",
        TITLE = "Spatial-Semantic Attention for Grounded Image Captioning",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "61-65",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126985"}

@inproceedings{bb130946,
        AUTHOR = "Sharif, N. and Jalwana, M.A.A.K. and Bennamoun, M. and Liu, W. and Shah, S.A.A.",
        TITLE = "Leveraging Linguistically-aware Object Relations and NASNet for Image
Captioning",
        BOOKTITLE = IVCNZ20,
        YEAR = "2020",
        PAGES = "1-6",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126986"}

@inproceedings{bb130947,
        AUTHOR = "Kuo, C.W. and Kira, Z.",
        TITLE = "Beyond a Pre-Trained Object Detector: Cross-Modal Textual and Visual
Context for Image Captioning",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "17948-17958",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126987"}

@inproceedings{bb130948,
        AUTHOR = "Zhou, M.Y. and Zhou, L.W. and Wang, S.H. and Cheng, Y. and Li, L.J. and Yu, Z. and Liu, J.J.",
        TITLE = "UC2: Universal Cross-lingual Cross-modal Vision-and-Language
Pre-training",
        BOOKTITLE = CVPR21,
        YEAR = "2021",
        PAGES = "4153-4163",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126988"}

@inproceedings{bb130949,
        AUTHOR = "Laina, I. and Rupprecht, C. and Navab, N.",
        TITLE = "Towards Unsupervised Image Captioning With Shared Multimodal
Embeddings",
        BOOKTITLE = ICCV19,
        YEAR = "2019",
        PAGES = "7413-7423",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126989"}

@inproceedings{bb130950,
        AUTHOR = "Akbari, H. and Karaman, S. and Bhargava, S. and Chen, B. and Vondrick, C. and Chang, S.F.",
        TITLE = "Multi-Level Multimodal Common Semantic Space for Image-Phrase Grounding",
        BOOKTITLE = CVPR19,
        YEAR = "2019",
        PAGES = "12468-12478",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126990"}

@inproceedings{bb130951,
        AUTHOR = "Chen, T.H. and Liao, Y.H. and Chuang, C.Y. and Hsu, W.T. and Fu, J. and Sun, M.",
        TITLE = "Show, Adapt and Tell:
Adversarial Training of Cross-Domain Image Captioner",
        BOOKTITLE = ICCV17,
        YEAR = "2017",
        PAGES = "521-530",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126991"}

@inproceedings{bb130952,
        AUTHOR = "Pini, S. and Cornia, M. and Baraldi, L. and Cucchiara, R.",
        TITLE = "Towards Video Captioning with Naming:
A Novel Dataset and a Multi-modal Approach",
        BOOKTITLE = CIAP17,
        YEAR = "2017",
        PAGES = "II:384-395",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126992"}

@inproceedings{bb130953,
        AUTHOR = "Pan, J.Y. and Yang, H.J. and Faloutsos, C.",
        TITLE = "MMSS: Graph-based Multi-modal Story-oriented Video Summarization and
Retrieval",
        BOOKTITLE = CMU-CS-TR,
        YEAR = "2004",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126993"}

@inproceedings{bb130954,
        AUTHOR = "Pan, J.Y. and Yang, H.J. and Faloutsos, C. and Duygulu, P.",
        TITLE = "GCap: Graph-based Automatic Image Captioning",
        BOOKTITLE = MMDE04,
        YEAR = "2004",
        PAGES = "146",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126994"}

@inproceedings{bb130955,
        AUTHOR = "Pan, J.Y.",
        TITLE = "Advanced Tools for Video and Multimedia Mining",
        BOOKTITLE = CMU-CS,
        YEAR = "2006",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126995"}

@inproceedings{bb130956,
        AUTHOR = "Pan, J.Y.",
        TITLE = "Advanced Tools for Video and Multimedia Mining",
        BOOKTITLE = Ph.D.,
        YEAR = "2006",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607mmic3.html#TT126995"}

@article{bb130957,
        AUTHOR = "Yu, J. and Li, J. and Yu, Z. and Huang, Q.",
        TITLE = "Multimodal Transformer With Multi-View Visual Representation for
Image Captioning",
        JOURNAL = CirSysVideo,
        VOLUME = "30",
        YEAR = "2020",
        NUMBER = "12",
        MONTH = "December",
        PAGES = "4467-4480",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT126996"}

@article{bb130958,
        AUTHOR = "Zhang, Y. and Shi, X.Y. and Mi, S. and Yang, X.",
        TITLE = "Image captioning with transformer and knowledge graph",
        JOURNAL = PRL,
        VOLUME = "143",
        YEAR = "2021",
        PAGES = "43-49",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT126997"}

@article{bb130959,
        AUTHOR = "Yan, C.G. and Hao, Y.M. and Li, L. and Yin, J. and Liu, A. and Mao, Z. and Chen, Z.Y. and Gao, X.Y.",
        TITLE = "Task-Adaptive Attention for Image Captioning",
        JOURNAL = CirSysVideo,
        VOLUME = "32",
        YEAR = "2022",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "43-51",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT126998"}

@article{bb130960,
        AUTHOR = "Ren, Z.H. and Gou, S.P. and Guo, Z. and Mao, S.S. and Li, R.M.",
        TITLE = "A Mask-Guided Transformer Network with Topic Token for Remote Sensing
Image Captioning",
        JOURNAL = RS,
        VOLUME = "14",
        YEAR = "2022",
        NUMBER = "12",
        PAGES = "xx-yy",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT126999"}

@article{bb130961,
        AUTHOR = "Ji, J.Y. and Ma, Y.W. and Sun, X.S. and Zhou, Y. and Wu, Y.J. and Ji, R.R.",
        TITLE = "Knowing What to Learn: A Metric-Oriented Focal Mechanism for Image
Captioning",
        JOURNAL = IP,
        VOLUME = "31",
        YEAR = "2022",
        PAGES = "4321-4335",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127000"}

@article{bb130962,
        AUTHOR = "Li, X. and Zhang, W.K. and Sun, X. and Gao, X.",
        TITLE = "Semantic-meshed and content-guided transformer for image captioning",
        JOURNAL = IET-CV,
        VOLUME = "16",
        YEAR = "2022",
        NUMBER = "5",
        PAGES = "431-444",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127001"}

@article{bb130963,
        AUTHOR = "Xian, T.T. and Li, Z.X. and Tang, Z.J. and Ma, H.F.",
        TITLE = "Adaptive Path Selection for Dynamic Image Captioning",
        JOURNAL = CirSysVideo,
        VOLUME = "32",
        YEAR = "2022",
        NUMBER = "9",
        MONTH = "September",
        PAGES = "5762-5775",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127002"}

@article{bb130964,
        AUTHOR = "Cao, S. and An, G. and Zheng, Z.X. and Wang, Z.Y.",
        TITLE = "Vision-Enhanced and Consensus-Aware Transformer for Image Captioning",
        JOURNAL = CirSysVideo,
        VOLUME = "32",
        YEAR = "2022",
        NUMBER = "10",
        MONTH = "October",
        PAGES = "7005-7018",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127003"}

@article{bb130965,
        AUTHOR = "Jiang, W.T. and Zhou, W. and Hu, H.F.",
        TITLE = "Double-Stream Position Learning Transformer Network for Image
Captioning",
        JOURNAL = CirSysVideo,
        VOLUME = "32",
        YEAR = "2022",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "7706-7718",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127004"}

@article{bb130966,
        AUTHOR = "Li, J.C. and Zhou, W. and Wang, K. and Hu, H.F.",
        TITLE = "Triple-Stream Commonsense Circulation Transformer Network for Image
Captioning",
        JOURNAL = CVIU,
        VOLUME = "249",
        YEAR = "2024",
        PAGES = "104165",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127005"}

@article{bb130967,
        AUTHOR = "Hu, J.T. and Yang, Y. and Yao, L. and An, Y.Z. and Pan, L.",
        TITLE = "Position-guided transformer for image captioning",
        JOURNAL = IVC,
        VOLUME = "128",
        YEAR = "2022",
        PAGES = "104575",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127006"}

@article{bb130968,
        AUTHOR = "Wang, Z.G. and Shi, S. and Zhai, Z.R. and Wu, Y. and Yang, R.",
        TITLE = "ArCo: Attention-reinforced transformer with contrastive learning for
image captioning",
        JOURNAL = IVC,
        VOLUME = "128",
        YEAR = "2022",
        PAGES = "104570",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127007"}

@article{bb130969,
        AUTHOR = "Li, Z.X. and Wei, J. and Huang, F.C. and Ma, H.F.",
        TITLE = "Modeling graph-structured contexts for image captioning",
        JOURNAL = IVC,
        VOLUME = "129",
        YEAR = "2023",
        PAGES = "104591",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127008"}

@article{bb130970,
        AUTHOR = "Zhang, J. and Xie, Y.S. and Ding, W.C. and Wang, Z.",
        TITLE = "Cross on Cross Attention: Deep Fusion Transformer for Image
Captioning",
        JOURNAL = CirSysVideo,
        VOLUME = "33",
        YEAR = "2023",
        NUMBER = "8",
        MONTH = "August",
        PAGES = "4257-4268",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127009"}

@article{bb130971,
        AUTHOR = "Lim, J.H. and Chan, C.S.",
        TITLE = "Mask-guided network for image captioning",
        JOURNAL = PRL,
        VOLUME = "173",
        YEAR = "2023",
        PAGES = "79-86",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127010"}

@article{bb130972,
        AUTHOR = "Li, Z.X. and Su, Q. and Chen, T.Y.",
        TITLE = "External knowledge-assisted Transformer for image captioning",
        JOURNAL = IVC,
        VOLUME = "140",
        YEAR = "2023",
        PAGES = "104864",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127011"}

@article{bb130973,
        AUTHOR = "Chen, J.Q.",
        TITLE = "Transform, contrast and tell:
Coherent entity-aware multi-image captioning",
        JOURNAL = CVIU,
        VOLUME = "238",
        YEAR = "2024",
        PAGES = "103878",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127012"}

@article{bb130974,
        AUTHOR = "Yang, X.B. and Tian, X. and Wu, J.S. and Yang, X.C. and Ma, S. and Qi, X. and Hou, Z.Q.",
        TITLE = "LLAFN-Generator: Learnable linear-attention with fast-normalization
for large-scale image captioning",
        JOURNAL = CVIU,
        VOLUME = "248",
        YEAR = "2024",
        PAGES = "104088",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127013"}

@article{bb130975,
        AUTHOR = "Yi, Y. and Liang, Y. and Kong, D. and Tang, Z.W. and Peng, J.B.",
        TITLE = "Tag-inferring and tag-guided Transformer for image captioning",
        JOURNAL = IET-CV,
        VOLUME = "18",
        YEAR = "2024",
        NUMBER = "6",
        PAGES = "801-812",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127014"}

@inproceedings{bb130976,
        AUTHOR = "Song, J.Y. and Pan, R.J. and Zhou, J. and Yang, H.",
        TITLE = "M-rat: a Multi-grained Retrieval Augmentation Transformer for Image
Captioning",
        BOOKTITLE = ACCV24,
        YEAR = "2024",
        PAGES = "III: 185-203",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127015"}

@inproceedings{bb130977,
        AUTHOR = "Caffagni, D. and Barraco, M. and Cornia, M. and Baraldi, L. and Cucchiara, R.",
        TITLE = "Synthcap: Augmenting Transformers with Synthetic Data for Image
Captioning",
        BOOKTITLE = CIAP23,
        YEAR = "2023",
        PAGES = "I:112-123",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127016"}

@inproceedings{bb130978,
        AUTHOR = "Lou, L.S. and Lu, K. and Xue, J.",
        TITLE = "Improved Transformer with Parallel Encoders for Image Captioning",
        BOOKTITLE = "ICPR22",
        YEAR = "2022",
        PAGES = "4072-4075",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127017"}

@inproceedings{bb130979,
        AUTHOR = "Wang, Y.H. and Shang, L.",
        TITLE = "Generating Spatial-aware Captions for TextCaps",
        BOOKTITLE = "ICPR22",
        YEAR = "2022",
        PAGES = "379-385",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127018"}

@inproceedings{bb130980,
        AUTHOR = "Feng, Y. and Maeda, K. and Ogawa, T. and Haseyama, M.",
        TITLE = "Human-Centric Image Retrieval with Gaze-Based Image Captioning",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "3828-3832",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127019"}

@inproceedings{bb130981,
        AUTHOR = "Yang, X. and Wang, Y. and Chen, H. and Li, J.",
        TITLE = "CSTNET: Enhancing Global-To-Local Interactions for Image Captioning",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "1861-1865",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127020"}

@inproceedings{bb130982,
        AUTHOR = "Nguyen, V.Q. and Suganuma, M. and Okatani, T.",
        TITLE = "GRIT: Faster and Better Image Captioning Transformer Using Dual Visual
Features",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXXVI:167-184",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127021"}

@inproceedings{bb130983,
        AUTHOR = "Vo, D.M. and Chen, H. and Sugimoto, A. and Nakayama, H.",
        TITLE = "NOC-REK: Novel Object Captioning with Retrieved Vocabulary from
External Knowledge",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "17979-17987",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127022"}

@inproceedings{bb130984,
        AUTHOR = "Yuan, Z.H. and Yan, X. and Liao, Y.H. and Guo, Y. and Li, G.B. and Cui, S.G. and Li, Z.",
        TITLE = "X-Trans2Cap:
Cross-Modal Knowledge Transfer using Transformer for 3D Dense Captioning",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "8553-8563",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127023"}

@inproceedings{bb130985,
        AUTHOR = "Liu, B. and Wang, D. and Yang, X. and Zhou, Y. and Yao, R. and Shao, Z.W. and Zhao, J.Q.",
        TITLE = "Show, Deconfound and Tell: Image Captioning with Causal Inference",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "18020-18029",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127024"}

@inproceedings{bb130986,
        AUTHOR = "Fang, Z.Y. and Wang, J.F. and Hu, X.W. and Liang, L. and Gan, Z. and Wang, L.J. and Yang, Y.Z. and Liu, Z.C.",
        TITLE = "Injecting Semantic Concepts into End-to-End Image Captioning",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "17988-17998",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127025"}

@inproceedings{bb130987,
        AUTHOR = "Li, Y. and Pan, Y.W. and Yao, T. and Mei, T.",
        TITLE = "Comprehending and Ordering Semantics for Image Captioning",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "17969-17978",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127026"}

@inproceedings{bb130988,
        AUTHOR = "Fei, Z.C. and Yan, X. and Wang, S.H. and Tian, Q.",
        TITLE = "DeeCap: Dynamic Early Exiting for Efficient Image Captioning",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "12206-12216",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127027"}

@inproceedings{bb130989,
        AUTHOR = "Wu, M.R. and Zhang, X.Y. and Sun, X.S. and Zhou, Y. and Chen, C. and Gu, J.X. and Sun, X. and Ji, R.R.",
        TITLE = "DIFNet: Boosting Visual Information Flow for Image Captioning",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "17999-18008",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127028"}

@inproceedings{bb130990,
        AUTHOR = "Rio Torto, I. and Cardoso, J.S. and Teixeira, L.F.",
        TITLE = "From Captions to Explanations: A Multimodal Transformer-based
Architecture for Natural Language Explanation Generation",
        BOOKTITLE = IbPRIA22,
        YEAR = "2022",
        PAGES = "54-65",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127029"}

@inproceedings{bb130991,
        AUTHOR = "Chen, H.S. and Wang, Y. and Yang, X. and Li, J.",
        TITLE = "Captioning Transformer With Scene Graph Guiding",
        BOOKTITLE = ICIP21,
        YEAR = "2021",
        PAGES = "2538-2542",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127030"}

@inproceedings{bb130992,
        AUTHOR = "Zhang, X.Y. and Sun, X.S. and Luo, Y.P. and Ji, J.Y. and Zhou, Y. and Wu, Y.J. and Huang, F.Y. and Ji, R.R.",
        TITLE = "RSTNet:
Captioning with Adaptive Attention on Visual and Non-Visual Words",
        BOOKTITLE = CVPR21,
        YEAR = "2021",
        PAGES = "15460-15469",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127031"}

@inproceedings{bb130993,
        AUTHOR = "He, S. and Liao, W.T. and Tavakoli, H.R. and Yang, M. and Rosenhahn, B. and Pugeault, N.",
        TITLE = "Image Captioning Through Image Transformer",
        BOOKTITLE = ACCV20,
        YEAR = "2020",
        PAGES = "IV:153-169",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127032"}

@inproceedings{bb130994,
        AUTHOR = "Cornia, M. and Stefanini, M. and Baraldi, L. and Cucchiara, R.",
        TITLE = "Meshed-Memory Transformer for Image Captioning",
        BOOKTITLE = CVPR20,
        YEAR = "2020",
        PAGES = "10575-10584",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127033"}

@inproceedings{bb130995,
        AUTHOR = "Tran, A. and Mathews, A. and Xie, L.",
        TITLE = "Transform and Tell: Entity-Aware News Image Captioning",
        BOOKTITLE = CVPR20,
        YEAR = "2020",
        PAGES = "13032-13042",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127034"}

@inproceedings{bb130996,
        AUTHOR = "Li, G. and Zhu, L. and Liu, P. and Yang, Y.",
        TITLE = "Entangled Transformer for Image Captioning",
        BOOKTITLE = ICCV19,
        YEAR = "2019",
        PAGES = "8927-8936",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607tic2.html#TT127035"}

@article{bb130997,
        AUTHOR = "Sharma, D. and Chattopadhyay, C.",
        TITLE = "High-level feature aggregation for fine-grained architectural floor
plan retrieval",
        JOURNAL = IET-CV,
        VOLUME = "12",
        YEAR = "2018",
        NUMBER = "5",
        MONTH = "August",
        PAGES = "702-709",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607seco3.html#TT127036"}

@inproceedings{bb130998,
        AUTHOR = "Sharma, D. and Chattopadhyay, C. and Harit, G.",
        TITLE = "A unified framework for semantic matching of architectural floorplans",
        BOOKTITLE = ICPR16,
        YEAR = "2016",
        PAGES = "2422-2427",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607seco3.html#TT127037"}

@article{bb130999,
        AUTHOR = "Ham, B. and Cho, M.S. and Schmid, C. and Ponce, J.",
        TITLE = "Proposal Flow: Semantic Correspondences from Object Proposals",
        JOURNAL = PAMI,
        VOLUME = "40",
        YEAR = "2018",
        NUMBER = "7",
        MONTH = "July",
        PAGES = "1711-1725",
        BIBSOURCE = "http://www.visionbib.com/bibliography/match607seco3.html#TT127038"}

Last update:Jul 15, 2025 at 15:29:22