{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T13:57:12Z","timestamp":1760709432847,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,10,23]],"date-time":"2017-10-23T00:00:00Z","timestamp":1508716800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1509206, 61472276"],"award-info":[{"award-number":["U1509206, 61472276"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2017,10,23]]},"DOI":"10.1145\/3123266.3127904","type":"proceedings-article","created":{"date-parts":[[2017,10,20]],"date-time":"2017-10-20T13:04:26Z","timestamp":1508504666000},"page":"1877-1882","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Multirate Multimodal Video Captioning"],"prefix":"10.1145","author":[{"given":"Ziwei","family":"Yang","sequence":"first","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youjiang","family":"Xu","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huiyun","family":"Wang","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Wang","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yahong","family":"Han","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2017,10,23]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Lorenzo Baraldi Costantino Grana and Rita Cucchiara. 2017. Hierarchical Boundary-Aware Neural Encoder for Video Captioning CVPR. Lorenzo Baraldi Costantino Grana and Rita Cucchiara. 2017. Hierarchical Boundary-Aware Neural Encoder for Video Captioning CVPR.","DOI":"10.1109\/CVPR.2017.339"},{"volume-title":"Microsoft COCO captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325","year":"2015","author":"Chen Xinlei","key":"e_1_3_2_1_2_1"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Liu Chenxi Mao Junhua Sha Fei and Yuille Alan. 2017. Attention Correctness in Neural Image Captioning. AAAI. Liu Chenxi Mao Junhua Sha Fei and Yuille Alan. 2017. Attention Correctness in Neural Image Captioning. AAAI.","DOI":"10.1609\/aaai.v31i1.11197"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2984064"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967242"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_38"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2984065"},{"key":"e_1_3_2_1_10_1","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks Advances in neural information processing systems. 1097--1105. Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks Advances in neural information processing systems. 1097--1105."},{"volume-title":"Meteor universal: Language specific translation evaluation for any target language. ACL","year":"2014","author":"Alon Lavie Michael Denkowski","key":"e_1_3_2_1_11_1"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806314"},{"volume-title":"et almbox","year":"2016","author":"Li Yi","key":"e_1_3_2_1_13_1"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Pingbo Pan Zhongwen Xu Yi Yang Fei Wu and Yueting Zhuang. 2016 b. Hierarchical recurrent neural encoder for video representation with application to captioning CVPR. 1029--1038. Pingbo Pan Zhongwen Xu Yi Yang Fei Wu and Yueting Zhuang. 2016 b. Hierarchical recurrent neural encoder for video representation with application to captioning CVPR. 1029--1038.","DOI":"10.1109\/CVPR.2016.117"},{"key":"e_1_3_2_1_15_1","unstructured":"Yingwei Pan Tao Mei Ting Yao Houqiang Li and Yong Rui. 2016 a. Jointly Modeling Embedding and Translation to Bridge Video and Language Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). Yingwei Pan Tao Mei Ting Yao Houqiang Li and Yong Rui. 2016 a. Jointly Modeling Embedding and Translation to Bridge Video and Language Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"volume-title":"Video Captioning with Transferred Semantic Attributes Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","year":"2017","author":"Pan Yingwei","key":"e_1_3_2_1_16_1"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2984066"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2984062"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"volume-title":"Cider: Consensus-based image description evaluation CVPR. 4566--4575.","year":"2015","author":"Vedantam Ramakrishna","key":"e_1_3_2_1_21_1"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Subhashini Venugopalan Huijuan Xu Jeff Donahue Marcus Rohrbach Raymond Mooney and Kate Saenko. 2015 b. Translating videos to natural language using deep recurrent neural networks. NAACL-HLT. Subhashini Venugopalan Huijuan Xu Jeff Donahue Marcus Rohrbach Raymond Mooney and Kate Saenko. 2015 b. Translating videos to natural language using deep recurrent neural networks. NAACL-HLT.","DOI":"10.3115\/v1\/N15-1173"},{"volume-title":"MSR-VTT: A Large Video Description Dataset for Bridging Video and Language Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","year":"2016","author":"Xu Jun","key":"e_1_3_2_1_24_1"},{"key":"e_1_3_2_1_25_1","volume-title":"Attend and Tell: Neural Image Caption Generation with Visual Attention ICML","volume":"14","author":"Xu Kelvin","year":"2015"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123327"},{"key":"e_1_3_2_1_27_1","unstructured":"Zhilin Yang Ye Yuan Yuexin Wu William W Cohen and Ruslan R Salakhutdinov. 2016. Review networks for caption generation. In NIPS. 2361--2369. Zhilin Yang Ye Yuan Yuexin Wu William W Cohen and Ruslan R Salakhutdinov. 2016. Review networks for caption generation. In NIPS. 2361--2369."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.512"},{"key":"e_1_3_2_1_29_1","unstructured":"Quanzeng You Hailin Jin Zhaowen Wang Chen Fang and Jiebo Luo. 2016. Image captioning with semantic attention. In CVPR. 4651--4659. Quanzeng You Hailin Jin Zhaowen Wang Chen Fang and Jiebo Luo. 2016. Image captioning with semantic attention. In CVPR. 4651--4659."},{"key":"e_1_3_2_1_30_1","unstructured":"Linchao Zhu Zhongwen Xu and Yi Yang. 2017. Bidirectional Multirate Reconstruction for Temporal Modeling in Videos CVPR. Linchao Zhu Zhongwen Xu and Yi Yang. 2017. Bidirectional Multirate Reconstruction for Temporal Modeling in Videos CVPR."}],"event":{"name":"MM '17: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Mountain View California USA","acronym":"MM '17"},"container-title":["Proceedings of the 25th ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3123266.3127904","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3123266.3127904","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T16:46:08Z","timestamp":1750956368000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3123266.3127904"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,10,23]]},"references-count":30,"alternative-id":["10.1145\/3123266.3127904","10.1145\/3123266"],"URL":"https:\/\/doi.org\/10.1145\/3123266.3127904","relation":{},"subject":[],"published":{"date-parts":[[2017,10,23]]},"assertion":[{"value":"2017-10-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}