{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,27]],"date-time":"2025-10-27T10:51:35Z","timestamp":1761562295236,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,10,19]],"date-time":"2017-10-19T00:00:00Z","timestamp":1508371200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Natural Science Foundation of China (NSFC)","award":["U1509206, 61472276"],"award-info":[{"award-number":["U1509206, 61472276"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2017,10,19]]},"DOI":"10.1145\/3123266.3123327","type":"proceedings-article","created":{"date-parts":[[2017,10,20]],"date-time":"2017-10-20T13:04:26Z","timestamp":1508504666000},"page":"146-153","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":66,"title":["Catching the Temporal Regions-of-Interest for Video Captioning"],"prefix":"10.1145","author":[{"given":"Ziwei","family":"Yang","sequence":"first","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yahong","family":"Han","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2017,10,19]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Dzmitry Bahdanau Kyunghyun Cho and Yoshua Bengio. 2014. Neural machine translation by jointly learning to align and translate ICLR. Dzmitry Bahdanau Kyunghyun Cho and Yoshua Bengio. 2014. Neural machine translation by jointly learning to align and translate ICLR."},{"key":"e_1_3_2_1_2_1","unstructured":"Nicolas Ballas Li Yao Chris Pal and Aaron Courville. 2016. Delving deeper into convolutional networks for learning video representations ICLR. Nicolas Ballas Li Yao Chris Pal and Aaron Courville. 2016. Delving deeper into convolutional networks for learning video representations ICLR."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Lorenzo Baraldi Costantino Grana and Rita Cucchiara. 2017. Hierarchical Boundary-Aware Neural Encoder for Video Captioning CVPR. Lorenzo Baraldi Costantino Grana and Rita Cucchiara. 2017. Hierarchical Boundary-Aware Neural Encoder for Video Captioning CVPR.","DOI":"10.1109\/CVPR.2017.339"},{"volume-title":"Collecting highly parallel data for paraphrase evaluation Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies-Volume 1","author":"Chen David L","key":"e_1_3_2_1_4_1"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Long Chen Hanwang Zhang Jun Xiao Liqiang Nie Jian Shao Wei Liu and Tat-Seng Chua. 2017. SCA-CNN: Spatial and Channel-wise Attention in Convolutional Networks for Image Captioning CVPR. Long Chen Hanwang Zhang Jun Xiao Liqiang Nie Jian Shao Wei Liu and Tat-Seng Chua. 2017. SCA-CNN: Spatial and Channel-wise Attention in Convolutional Networks for Image Captioning CVPR.","DOI":"10.1109\/CVPR.2017.667"},{"volume-title":"Microsoft COCO captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325","year":"2015","author":"Chen Xinlei","key":"e_1_3_2_1_6_1"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Liu Chenxi Mao Junhua Sha Fei and Yuille Alan. 2017. Attention Correctness in Neural Image Captioning. AAAI. Liu Chenxi Mao Junhua Sha Fei and Yuille Alan. 2017. Attention Correctness in Neural Image Captioning. AAAI.","DOI":"10.1609\/aaai.v31i1.11197"},{"volume-title":"Sergio Guadarrama, Marcus Rohrbach, Subhashini Venugopalan, Kate Saenko, and Trevor Darrell.","year":"2015","author":"Donahue Jeffrey","key":"e_1_3_2_1_8_1"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2984064"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.337"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967242"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2984070"},{"volume-title":"Generating Natural-Language Video Descriptions Using Text-Mined Knowledge AAAI","author":"Krishnamoorthy Niveda","key":"e_1_3_2_1_14_1"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995466"},{"volume-title":"Meteor universal: Language specific translation evaluation for any target language. ACL","year":"2014","author":"Alon Lavie Michael Denkowski","key":"e_1_3_2_1_16_1"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806314"},{"key":"e_1_3_2_1_18_1","unstructured":"Siming Li Girish Kulkarni Tamara L Berg Alexander C Berg and Yejin Choi. 2011. Composing simple image descriptions using web-scale n-grams CoNLL. 220--228. Siming Li Girish Kulkarni Tamara L Berg Alexander C Berg and Yejin Choi. 2011. Composing simple image descriptions using web-scale n-grams CoNLL. 220--228."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2984069"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967298"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Pingbo Pan Zhongwen Xu Yi Yang Fei Wu and Yueting Zhuang. 2016 b. Hierarchical recurrent neural encoder for video representation with application to captioning CVPR. 1029--1038. Pingbo Pan Zhongwen Xu Yi Yang Fei Wu and Yueting Zhuang. 2016 b. Hierarchical recurrent neural encoder for video representation with application to captioning CVPR. 1029--1038.","DOI":"10.1109\/CVPR.2016.117"},{"key":"e_1_3_2_1_22_1","unstructured":"Yingwei Pan Tao Mei Ting Yao Houqiang Li and Yong Rui. 2016 a. Jointly modeling embedding and translation to bridge video and language CVPR. 4594--4602. Yingwei Pan Tao Mei Ting Yao Houqiang Li and Yong Rui. 2016 a. Jointly modeling embedding and translation to bridge video and language CVPR. 4594--4602."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.61"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.5555\/2627435.2670313"},{"key":"e_1_3_2_1_26_1","unstructured":"Ilya Sutskever Oriol Vinyals and Quoc V Le. 2014. Sequence to sequence learning with neural networks NIPS. 3104--3112. Ilya Sutskever Oriol Vinyals and Quoc V Le. 2014. Sequence to sequence learning with neural networks NIPS. 3104--3112."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going deeper with convolutions. In CVPR. 1--9. Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2015. Going deeper with convolutions. In CVPR. 1--9.","DOI":"10.1109\/CVPR.2015.7298594"},{"volume-title":"Using descriptive video services to create a large data source for video annotation research. arXiv preprint arXiv:1503.01070","year":"2015","author":"Torabi Atousa","key":"e_1_3_2_1_28_1"},{"volume-title":"Cider: Consensus-based image description evaluation CVPR. 4566--4575.","year":"2015","author":"Vedantam Ramakrishna","key":"e_1_3_2_1_29_1"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Subhashini Venugopalan Huijuan Xu Jeff Donahue Marcus Rohrbach Raymond Mooney and Kate Saenko. 2015 b. Translating videos to natural language using deep recurrent neural networks. NAACL-HLT. Subhashini Venugopalan Huijuan Xu Jeff Donahue Marcus Rohrbach Raymond Mooney and Kate Saenko. 2015 b. Translating videos to natural language using deep recurrent neural networks. NAACL-HLT.","DOI":"10.3115\/v1\/N15-1173"},{"key":"e_1_3_2_1_32_1","volume-title":"Attend and Tell: Neural Image Caption Generation with Visual Attention ICML","volume":"14","author":"Xu Kelvin","year":"2015"},{"key":"e_1_3_2_1_33_1","unstructured":"Youjiang Xu Shichao Zhao Qinghua Hu Yahong Han and Fei Wu. 2017. Top Attention in Line with Time: A Light-Weight Strategy ICME. Youjiang Xu Shichao Zhao Qinghua Hu Yahong Han and Fei Wu. 2017. Top Attention in Line with Time: A Light-Weight Strategy ICME."},{"key":"e_1_3_2_1_34_1","unstructured":"Zhilin Yang Ye Yuan Yuexin Wu William W Cohen and Ruslan R Salakhutdinov. 2016. Review networks for caption generation. In NIPS. 2361--2369. Zhilin Yang Ye Yuan Yuexin Wu William W Cohen and Ruslan R Salakhutdinov. 2016. Review networks for caption generation. In NIPS. 2361--2369."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.512"},{"key":"e_1_3_2_1_36_1","unstructured":"Quanzeng You Hailin Jin Zhaowen Wang Chen Fang and Jiebo Luo. 2016. Image captioning with semantic attention. In CVPR. 4651--4659. Quanzeng You Hailin Jin Zhaowen Wang Chen Fang and Jiebo Luo. 2016. Image captioning with semantic attention. In CVPR. 4651--4659."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Haonan Yu Jiang Wang Zhiheng Huang Yi Yang and Wei Xu. 2016. Video paragraph captioning using hierarchical recurrent neural networks CVPR. 4584--4593. Haonan Yu Jiang Wang Zhiheng Huang Yi Yang and Wei Xu. 2016. Video paragraph captioning using hierarchical recurrent neural networks CVPR. 4584--4593.","DOI":"10.1109\/CVPR.2016.496"},{"volume-title":"Pooling the Convolutional Layers in Deep ConvNets for Video Action Recognition","year":"2017","author":"Zhao Shichao","key":"e_1_3_2_1_38_1"},{"key":"e_1_3_2_1_39_1","unstructured":"Linchao Zhu Zhongwen Xu and Yi Yang. 2017. Bidirectional Multirate Reconstruction for Temporal Modeling in Videos CVPR. Linchao Zhu Zhongwen Xu and Yi Yang. 2017. Bidirectional Multirate Reconstruction for Temporal Modeling in Videos CVPR."}],"event":{"name":"MM '17: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Mountain View California USA","acronym":"MM '17"},"container-title":["Proceedings of the 25th ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3123266.3123327","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3123266.3123327","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T16:33:37Z","timestamp":1750955617000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3123266.3123327"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,10,19]]},"references-count":39,"alternative-id":["10.1145\/3123266.3123327","10.1145\/3123266"],"URL":"https:\/\/doi.org\/10.1145\/3123266.3123327","relation":{},"subject":[],"published":{"date-parts":[[2017,10,19]]},"assertion":[{"value":"2017-10-19","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}