{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T10:28:55Z","timestamp":1763202535333,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NExT Research Centre"},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U20B2063"],"award-info":[{"award-number":["U20B2063"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["BX2021055, 2021M690027"],"award-info":[{"award-number":["BX2021055, 2021M690027"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475173","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T20:00:05Z","timestamp":1634587205000},"page":"5110-5118","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Multi-Perspective Video Captioning"],"prefix":"10.1145","author":[{"given":"Yi","family":"Bin","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xindi","family":"Shang","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Peng","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yujuan","family":"Ding","sequence":"additional","affiliation":[{"name":"Hong Kong Polytechnic University, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"volume-title":"Spice: Semantic propositional image caption evaluation. In ECCV. 382--398.","year":"2016","author":"Anderson Peter","key":"e_1_3_2_1_1_1"},{"volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","year":"2016","author":"Ba Jimmy Lei","key":"e_1_3_2_1_2_1"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Lorenzo Baraldi Costantino Grana and Rita Cucchiara. 2017. Hierarchical boundary-aware neural encoder for video captioning. In CVPR. 1657--1666.  Lorenzo Baraldi Costantino Grana and Rita Cucchiara. 2017. Hierarchical boundary-aware neural encoder for video captioning. In CVPR. 1657--1666.","DOI":"10.1109\/CVPR.2017.339"},{"key":"e_1_3_2_1_4_1","first-page":"2631","article-title":"Describing video with attention-based bidirectional LSTM","volume":"49","author":"Bin Yi","year":"2018","journal-title":"TCYB"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123391"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Moitreya Chatterjee and Alexander G Schwing. 2018. Diverse and coherent paragraph generation from images. In ECCV. 729--744.  Moitreya Chatterjee and Alexander G Schwing. 2018. Diverse and coherent paragraph generation from images. In ECCV. 729--744.","DOI":"10.1007\/978-3-030-01216-8_45"},{"volume-title":"Groupcap: Group-based image captioning with structured relevance and diversity constraints. In CVPR. 1345--1353.","year":"2018","author":"Chen Fuhai","key":"e_1_3_2_1_7_1"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123420"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Tianlang Chen Zhongping Zhang Quanzeng You Chen Fang Zhaowen Wang Hailin Jin and Jiebo Luo. 2018b. \u201cFactual\u201dor\u201cEmotional\u201d: Stylized Image Captioning with Adaptive Learning and Attention. In ECCV. 519--535.  Tianlang Chen Zhongping Zhang Quanzeng You Chen Fang Zhaowen Wang Hailin Jin and Jiebo Luo. 2018b. \u201cFactual\u201dor\u201cEmotional\u201d: Stylized Image Captioning with Adaptive Learning and Attention. In ECCV. 519--535.","DOI":"10.1007\/978-3-030-01249-6_32"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/W14-3348"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.5555\/3157096.3157211"},{"volume-title":"Stylenet: Generating attractive visual captions with styles. In CVPR. 3137--3146.","year":"2017","author":"Gan Chuang","key":"e_1_3_2_1_12_1"},{"key":"e_1_3_2_1_13_1","unstructured":"Longteng Guo Jing Liu Peng Yao Jiangwei Li and Hanqing Lu. 2019. MSCap: Multi-Style Image Captioning With Unpaired Stylized Text. In CVPR. 4204--4213.  Longteng Guo Jing Liu Peng Yao Jiangwei Li and Hanqing Lu. 2019. MSCap: Multi-Style Image Captioning With Unpaired Stylized Text. In CVPR. 4204--4213."},{"key":"e_1_3_2_1_14_1","unstructured":"Kaiming He Georgia Gkioxari Piotr Doll\u00e1r and Ross Girshick. 2017. Mask r-cnn. In ICCV. 2961--2969.  Kaiming He Georgia Gkioxari Piotr Doll\u00e1r and Ross Girshick. 2017. Mask r-cnn. In ICCV. 2961--2969."},{"key":"e_1_3_2_1_15_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778."},{"volume-title":"Densecap: Fully convolutional localization networks for dense captioning. In CVPR. 4565--4574.","year":"2016","author":"Johnson Justin","key":"e_1_3_2_1_16_1"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Ranjay Krishna Kenji Hata Frederic Ren Li Fei-Fei and Juan Carlos Niebles. 2017a. Dense-captioning events in videos. In ICCV. 706--715.  Ranjay Krishna Kenji Hata Frederic Ren Li Fei-Fei and Juan Carlos Niebles. 2017a. Dense-captioning events in videos. In ICCV. 706--715.","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/2891460.2891535"},{"volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74--81.","year":"2004","author":"Lin Chin-Yew","key":"e_1_3_2_1_20_1"},{"key":"e_1_3_2_1_21_1","unstructured":"Jiasen Lu Caiming Xiong Devi Parikh and Richard Socher. 2017. Knowing when to look: Adaptive attention via a visual sentinel for image captioning. In CVPR. 375--383.  Jiasen Lu Caiming Xiong Devi Parikh and Richard Socher. 2017. Knowing when to look: Adaptive attention via a visual sentinel for image captioning. In CVPR. 375--383."},{"volume-title":"Steven Bethard, and David McClosky.","year":"2014","author":"Manning Christopher D","key":"e_1_3_2_1_22_1"},{"volume-title":"Senticap: Generating image descriptions with sentiments. In AAAI .","year":"2016","author":"Mathews Alexander Patrick","key":"e_1_3_2_1_23_1"},{"key":"e_1_3_2_1_24_1","unstructured":"Pingbo Pan Zhongwen Xu Yi Yang Fei Wu and Yueting Zhuang. 2016. Hierarchical recurrent neural encoder for video representation with application to captioning. In CVPR. 1029--1038.  Pingbo Pan Zhongwen Xu Yi Yang Fei Wu and Yueting Zhuang. 2016. Hierarchical recurrent neural encoder for video representation with application to captioning. In CVPR. 1029--1038."},{"key":"e_1_3_2_1_25_1","unstructured":"Yingwei Pan Ting Yao Houqiang Li and Tao Mei. 2017. Video captioning with transferred semantic attributes. In CVPR. 6504--6512.  Yingwei Pan Ting Yao Houqiang Li and Tao Mei. 2017. Video captioning with transferred semantic attributes. In CVPR. 6504--6512."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Jae Sung Park Marcus Rohrbach Trevor Darrell and Anna Rohrbach. 2019. Adversarial inference for multi-sentence video description. In CVPR. 6598--6608.  Jae Sung Park Marcus Rohrbach Trevor Darrell and Anna Rohrbach. 2019. Adversarial inference for multi-sentence video description. In CVPR. 6598--6608.","DOI":"10.1109\/CVPR.2019.00676"},{"volume-title":"MRA-Net: Improving VQA via Multi-modal Relation Attention Network. TPAMI","year":"2020","author":"Peng Liang","key":"e_1_3_2_1_28_1"},{"key":"e_1_3_2_1_29_1","unstructured":"D Raj Reddy et al. 1977. Speech Understanding Systems: A Summary of Results of the Five-Year Research Effort. Department of Computer Science.  D Raj Reddy et al. 1977. Speech Understanding Systems: A Summary of Results of the Five-Year Research Effort. Department of Computer Science."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3323873.3325056"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Zhiqiang Shen Jianguo Li Zhou Su Minjun Li Yurong Chen Yu-Gang Jiang and Xiangyang Xue. 2017. Weakly supervised dense video captioning. In CVPR. 1916--1924.  Zhiqiang Shen Jianguo Li Zhou Su Minjun Li Yurong Chen Yu-Gang Jiang and Xiangyang Xue. 2017. Weakly supervised dense video captioning. In CVPR. 1916--1924.","DOI":"10.1109\/CVPR.2017.548"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Kurt Shuster Samuel Humeau Hexiang Hu Antoine Bordes and Jason Weston. 2019. Engaging image captioning via personality. In CVPR. 12516--12526.  Kurt Shuster Samuel Humeau Hexiang Hu Antoine Bordes and Jason Weston. 2019. Engaging image captioning via personality. In CVPR. 12516--12526.","DOI":"10.1109\/CVPR.2019.01280"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2812802"},{"volume-title":"Cider: Consensus-based image description evaluation. In CVPR. 4566--4575.","year":"2015","author":"Vedantam Ramakrishna","key":"e_1_3_2_1_34_1"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Ashwin K Vijayakumar Michael Cogswell Ramprasaath R Selvaraju Qing Sun Stefan Lee David Crandall and Dhruv Batra. 2018. Diverse beam search for improved description of complex scenes. In AAAI .  Ashwin K Vijayakumar Michael Cogswell Ramprasaath R Selvaraju Qing Sun Stefan Lee David Crandall and Dhruv Batra. 2018. Diverse beam search for improved description of complex scenes. In AAAI .","DOI":"10.1609\/aaai.v32i1.12340"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Jingwen Wang Wenhao Jiang Lin Ma Wei Liu and Yong Xu. 2018a. Bidirectional attentive fusion with context gating for dense video captioning. In CVPR. 7190--7198.  Jingwen Wang Wenhao Jiang Lin Ma Wei Liu and Yong Xu. 2018a. Bidirectional attentive fusion with context gating for dense video captioning. In CVPR. 7190--7198.","DOI":"10.1109\/CVPR.2018.00751"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295326"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240583"},{"volume-title":"Universal Weighting Metric Learning for Cross-Modal Retrieval. TPAMI","year":"2021","author":"Wei Jiwei","key":"e_1_3_2_1_40_1"},{"volume-title":"Diverse Video Captioning Through Latent Variable Expansion with Conditional GAN. arXiv preprint arXiv:1910.12019","year":"2019","author":"Xiao Huanhou","key":"e_1_3_2_1_41_1"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.5555\/3045118.3045336"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2855422"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.512"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Luowei Zhou Yannis Kalantidis Xinlei Chen Jason J Corso and Marcus Rohrbach. 2019. Grounded video description. In CVPR. 6578--6587.  Luowei Zhou Yannis Kalantidis Xinlei Chen Jason J Corso and Marcus Rohrbach. 2019. Grounded video description. In CVPR. 6578--6587.","DOI":"10.1109\/CVPR.2019.00674"}],"event":{"name":"MM '21: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Virtual Event China","acronym":"MM '21"},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475173","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475173","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:47Z","timestamp":1750193327000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475173"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":45,"alternative-id":["10.1145\/3474085.3475173","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475173","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}