{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T18:04:53Z","timestamp":1764785093004,"version":"3.41.0"},"reference-count":66,"publisher":"Association for Computing Machinery (ACM)","issue":"5s","license":[{"start":{"date-parts":[[2023,6,7]],"date-time":"2023-06-07T00:00:00Z","timestamp":1686096000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62172256, 62202278, 62202272"],"award-info":[{"award-number":["62172256, 62202278, 62202272"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"crossref","award":["ZR2019ZD06"],"award-info":[{"award-number":["ZR2019ZD06"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Major Program of the National Natural Science Foundation of China","award":["61991411"],"award-info":[{"award-number":["61991411"]}]},{"name":"Quan Cheng Laboratory"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["ACM Trans. Multimedia Comput. Commun. Appl."],"published-print":{"date-parts":[[2023,10,31]]},"abstract":"<jats:p>\n            Video captioning aims to automatically generate natural language sentences describing the content of a video. Although encoder-decoder-based models have achieved promising progress, it is still very challenging to effectively model the linguistic behavior of humans in generating video captions. In this paper, we propose a novel video captioning model by learning from\n            <jats:bold>gLobal sEntence and looking AheaD, LEAD<\/jats:bold>\n            for short. Specifically, LEAD consists of two modules: a\n            <jats:bold>Vision Module (VM)<\/jats:bold>\n            and a\n            <jats:bold>Language Module (LM)<\/jats:bold>\n            . Thereinto, VM is a novel attention network, which can map visual features to high-level language space and model entire sentences explicitly. LM can not only effectively make use of the information of the previous sequence when generating the current word, but also have a look at the future word. Therefore, based on VM and LM, LEAD can obtain global sentence information and future word information to make video captioning more like a fill-in-the-blank task than a word-by-word sentence generation. In addition, we also propose an autonomous strategy and a multi-stage training scheme to optimize the model, which can mitigate the problem of information leakage. Extensive experiments show that LEAD outperforms some state-of-the-art methods on MSR-VTT, MSVD, and VATEX, demonstrating the effectiveness of the proposed approach in video captioning. In addition, we release the code of our proposed model to be publicly available.\n            <jats:xref ref-type=\"fn\">\n              <jats:sup>1<\/jats:sup>\n            <\/jats:xref>\n          <\/jats:p>","DOI":"10.1145\/3587252","type":"journal-article","created":{"date-parts":[[2023,3,9]],"date-time":"2023-03-09T10:59:38Z","timestamp":1678359578000},"page":"1-20","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Video Captioning by Learning from Global Sentence and Looking Ahead"],"prefix":"10.1145","volume":"19","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7389-5883","authenticated-orcid":false,"given":"Tian-Zi","family":"Niu","sequence":"first","affiliation":[{"name":"The School of Software, Shandong University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3481-4892","authenticated-orcid":false,"given":"Zhen-Duo","family":"Chen","sequence":"additional","affiliation":[{"name":"The School of Software, Shandong University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6901-5476","authenticated-orcid":false,"given":"Xin","family":"Luo","sequence":"additional","affiliation":[{"name":"The School of Software, Shandong University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6790-2098","authenticated-orcid":false,"given":"Peng-Fei","family":"Zhang","sequence":"additional","affiliation":[{"name":"The School of Information Technology and Electrical Engineering, The University of Queensland, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9738-4949","authenticated-orcid":false,"given":"Zi","family":"Huang","sequence":"additional","affiliation":[{"name":"The School of Information Technology and Electrical Engineering, The University of Queensland, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9972-7370","authenticated-orcid":false,"given":"Xin-Shun","family":"Xu","sequence":"additional","affiliation":[{"name":"The School of Software, Shandong University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,6,7]]},"reference":[{"key":"e_1_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01277"},{"key":"e_1_3_2_3_2","volume-title":"Proceedings of the International Conference on Learning Representations","author":"Ballas Nicolas","year":"2016","unstructured":"Nicolas Ballas, Li Yao, Chris Pal, et al.2016. Delving deeper into convolutional networks for learning video representations. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_4_2","first-page":"65","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the Annual Meeting of the Association for Computational Linguistics. 65\u201372."},{"key":"e_1_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_6_2","first-page":"190","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"Chen David L.","year":"2011","unstructured":"David L. Chen and William B. Dolan. 2011. Collecting highly parallel data for paraphrase evaluation. In Proceedings of the Annual Meeting of the Association for Computational Linguistics. 190\u2013200."},{"key":"e_1_3_2_7_2","first-page":"1079","article-title":"Delving deeper into the decoder for video captioning","volume":"325","author":"Chen Haoran","year":"2020","unstructured":"Haoran Chen, Jianmin Li, and Xiaolin Hu. 2020. Delving deeper into the decoder for video captioning. Frontiers in Artificial Intelligence and Applications 325 (2020), 1079\u20131086.","journal-title":"Frontiers in Artificial Intelligence and Applications"},{"key":"e_1_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3539225"},{"key":"e_1_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018191"},{"key":"e_1_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093291"},{"key":"e_1_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3063423"},{"key":"e_1_3_2_12_2","first-page":"4171","volume-title":"Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, et al.2019. BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 4171\u20134186."},{"key":"e_1_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3550276"},{"key":"e_1_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00702"},{"key":"e_1_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2017.2729019"},{"key":"e_1_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.337"},{"key":"e_1_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00685"},{"key":"e_1_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3416290"},{"key":"e_1_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3460474"},{"key":"e_1_3_2_21_2","first-page":"1106","article-title":"ImageNet classification with deep convolutional neural networks","volume":"2","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey E. Hinton. 2012. ImageNet classification with deep convolutional neural networks. Advances in Neural Information Processing Systems 2 (2012), 1106\u20131114.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_22_2","first-page":"74","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. ROUGE: A package for automatic evaluation of summaries. In Proceedings of the Annual Meeting of the Association for Computational Linguistics. 74\u201381."},{"key":"e_1_3_2_23_2","first-page":"1865","article-title":"Prophet attention: Predicting attention with future attention","volume":"33","author":"Liu Fenglin","year":"2020","unstructured":"Fenglin Liu, Xuancheng Ren, Xian Wu, et al.2020. Prophet attention: Predicting attention with future attention. Advances in Neural Information Processing Systems 33 (2020), 1865\u20131876.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.24"},{"key":"e_1_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2940007"},{"key":"e_1_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503927"},{"key":"e_1_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.146"},{"key":"e_1_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01088"},{"key":"e_1_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.117"},{"key":"e_1_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.497"},{"key":"e_1_3_2_31_2","first-page":"311","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. BLEU: A method for automatic evaluation of machine translation. In Proceedings of the Annual Meeting of the Association for Computational Linguistics. 311\u2013318."},{"key":"e_1_3_2_32_2","volume-title":"Proceedings of the International Conference on Learning Representations","author":"Patrick Mandela","year":"2021","unstructured":"Mandela Patrick, Po-Yao Huang, Yuki Markus Asano, et al.2021. Support-set bottlenecks for video-text representation learning. In Proceedings of the International Conference on Learning Representations."},{"key":"e_1_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00854"},{"key":"e_1_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_35_2","first-page":"8748","volume-title":"Proceedings of the International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, et al.2021. Learning transferable visual models from natural language supervision. In Proceedings of the International Conference on Machine Learning. 8748\u20138763."},{"key":"e_1_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_37_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.131"},{"key":"e_1_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i3.16353"},{"key":"e_1_3_2_40_2","first-page":"802","article-title":"Convolutional LSTM network: A machine learning approach for precipitation nowcasting","volume":"28","author":"Shi Xingjian","year":"2015","unstructured":"Xingjian Shi, Zhourong Chen, Hao Wang, et al.2015. Convolutional LSTM network: A machine learning approach for precipitation nowcasting. Advances in Neural Information Processing Systems 28 (2015), 802\u2013810.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_41_2","doi-asserted-by":"publisher","DOI":"10.1145\/3546828"},{"key":"e_1_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2018.2851077"},{"key":"e_1_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"e_1_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/104"},{"key":"e_1_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/3303083"},{"key":"e_1_3_2_46_2","first-page":"1218","volume-title":"Proceedings of the International Conference on Computational Linguistics","author":"Thomason Jesse","year":"2014","unstructured":"Jesse Thomason, Subhashini Venugopalan, Sergio Guadarrama, et al.2014. Integrating language and vision to generate natural language descriptions of videos in the wild. In Proceedings of the International Conference on Computational Linguistics. 1218\u20131227."},{"key":"e_1_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00250"},{"key":"e_1_3_2_48_2","first-page":"5998","article-title":"Attention is all you need","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, et al.2017. Attention is all you need. Advances in Neural Information Processing Systems 30 (2017), 5998\u20136008.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"e_1_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00273"},{"key":"e_1_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00795"},{"key":"e_1_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1080\/0952813X.2021.1883745"},{"key":"e_1_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00443"},{"key":"e_1_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00468"},{"key":"e_1_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2956593"},{"key":"e_1_3_2_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3478024"},{"key":"e_1_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2846664"},{"key":"e_1_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16421"},{"key":"e_1_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.1145\/3386725"},{"key":"e_1_3_2_62_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.512"},{"key":"e_1_3_2_63_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00852"},{"key":"e_1_3_2_64_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00971"},{"key":"e_1_3_2_65_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01329"},{"key":"e_1_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01311"},{"key":"e_1_3_2_67_2","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3058626"}],"container-title":["ACM Transactions on Multimedia Computing, Communications, and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3587252","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3587252","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:47:16Z","timestamp":1750178836000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3587252"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,7]]},"references-count":66,"journal-issue":{"issue":"5s","published-print":{"date-parts":[[2023,10,31]]}},"alternative-id":["10.1145\/3587252"],"URL":"https:\/\/doi.org\/10.1145\/3587252","relation":{},"ISSN":["1551-6857","1551-6865"],"issn-type":[{"type":"print","value":"1551-6857"},{"type":"electronic","value":"1551-6865"}],"subject":[],"published":{"date-parts":[[2023,6,7]]},"assertion":[{"value":"2022-10-17","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2023-03-07","order":1,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2023-06-07","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}