{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:58:45Z","timestamp":1772121525243,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,10,15]],"date-time":"2019-10-15T00:00:00Z","timestamp":1571097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,10,15]]},"DOI":"10.1145\/3343031.3350932","type":"proceedings-article","created":{"date-parts":[[2019,10,21]],"date-time":"2019-10-21T16:32:26Z","timestamp":1571675546000},"page":"802-810","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":30,"title":["Attention-based Densely Connected LSTM for Video Captioning"],"prefix":"10.1145","author":[{"given":"Yongqing","family":"Zhu","sequence":"first","affiliation":[{"name":"Institute of Computing Technology Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuqiang","family":"Jiang","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Advances in Neural Information Processing Systems (NIPS)","author":"Adam Santoro"},{"key":"e_1_3_2_1_2_1","volume-title":"Kaiser \u0141 ukasz, and Polosukhin Illia","author":"Ashish Vaswani","year":"2017"},{"key":"e_1_3_2_1_3_1","volume-title":"International Journal of Computer Vision (IJCV)","author":"Atsuhiro Kojima","year":"2002"},{"key":"e_1_3_2_1_4_1","volume-title":"Hierarchical Boundary-Aware Neural Encoder for Video Captioning. In Internaltional Conference on Computer Vision and Pattern Recogintion (CVPR) .","author":"Baraldi Lorenzo","year":"2017"},{"key":"e_1_3_2_1_5_1","volume-title":"Temporal Deformable Convolutional Encoder-Decoder Networks for Video Captioning","author":"Chen Jingwen"},{"key":"e_1_3_2_1_6_1","volume-title":"Less Is More: Picking Informative Frames for Video Captioning. In The European Conference on Computer Vision (ECCV) .","author":"Chen Yangyu","year":"2018"},{"key":"e_1_3_2_1_7_1","volume-title":"Attention-Based Multimodal Fusion for Video Description. In International Conference on Computer Vision (ICCV) .","author":"Chiori Hori","year":"2017"},{"key":"e_1_3_2_1_8_1","volume-title":"Long-term Recurrent Convolutional Networks for Visual Recognition and Description. In Internaltional Conference on Computer Vision and Pattern Recogintion (CVPR) .","author":"Donahue Jeff","year":"2015"},{"key":"e_1_3_2_1_9_1","volume-title":"Improving Interpretability of Deep Neural Networks with Semantic Information. In Internaltional Conference on Computer Vision and Pattern Recogintion (CVPR) .","author":"Dong Yinpeng","year":"2017"},{"key":"e_1_3_2_1_10_1","volume-title":"International Conference on Learning Representations (ICLR) .","author":"Dzmitry Bahdanau","year":"2015"},{"key":"e_1_3_2_1_11_1","volume-title":"Deep Networks with Stochastic Depth. In European Conference on Computer Vision (ECCV) .","author":"Gao Huang","year":"2016"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","author":"Gao Huang","year":"2017"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_14_1","volume-title":"Long Short-Term Memory. Neural Computation","author":"Hochreiter Sepp","year":"1997"},{"key":"e_1_3_2_1_15_1","unstructured":"Ba Jimmy Hinton Geoffrey E Mnih Volodymyr Leibo Joel Z and Ionescu Catalin. 2016. Using Fast Weights to Attend to the Recent Past. In Neural Information Processing Systems (NIPS) . 4331--4339.  Ba Jimmy Hinton Geoffrey E Mnih Volodymyr Leibo Joel Z and Ionescu Catalin. 2016. Using Fast Weights to Attend to the Recent Past. In Neural Information Processing Systems (NIPS) . 4331--4339."},{"key":"e_1_3_2_1_16_1","volume-title":"International Conference on Machine Learning (ICML) .","author":"Jonas Gehring","year":"2017"},{"key":"e_1_3_2_1_17_1","unstructured":"Rupesh Kumar Srivastava Klaus Greff and J\u00fcrgen Schmidhuber. 2015. Training Very Deep Networks. In Neural Information Processing Systems (NIPS) .  Rupesh Kumar Srivastava Klaus Greff and J\u00fcrgen Schmidhuber. 2015. Training Very Deep Networks. In Neural Information Processing Systems (NIPS) ."},{"key":"e_1_3_2_1_18_1","volume-title":"Multimodal Dual Attention Memory for Video Story Question Answering. In The European Conference on Computer Vision (ECCV) .","author":"Kyung-Min Kim","year":"2018"},{"key":"e_1_3_2_1_19_1","volume-title":"METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In Annual Meeting of the Association for Computational Linguistics (ACL) .","author":"Lavie Alon","year":"2005"},{"key":"e_1_3_2_1_20_1","volume-title":"Bundled Object Context for Referring Expressions","author":"Li Xiangyang"},{"key":"e_1_3_2_1_21_1","volume-title":"Know More Say Less: Image Captioning Based on Scene Graphs","author":"Li Xiangyang"},{"key":"e_1_3_2_1_22_1","volume-title":"Beyond RNNs: Positional Self-Attention with Co-Attention for Video Question Answering","author":"Li Xiangpeng"},{"key":"e_1_3_2_1_23_1","volume-title":"Annual Meeting of the Association for Computational Linguistics (ACL) .","author":"Lin Chin-Yew","year":"2004"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.117"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.497"},{"key":"e_1_3_2_1_26_1","volume-title":"Video Captioning with Transferred Semantic Attributes. In Internaltional Conference on Computer Vision and Pattern Recogintion (CVPR) .","author":"Pan Yingwei","year":"2017"},{"key":"e_1_3_2_1_27_1","volume-title":"Annual Meeting of the Association for Computational Linguistics (ACL) .","author":"Papineni Kishore","year":"2002"},{"key":"e_1_3_2_1_28_1","volume-title":"Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations (ICLR) .","author":"Diederik"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_30_1","volume-title":"International Conference on Computer Vision (ICCV) .","author":"Sergio Guadarrama","year":"2013"},{"key":"e_1_3_2_1_31_1","volume-title":"International Conference on Learning Representations (ICLR) .","author":"Simonyan Karen","year":"2015"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/381"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Subhashini Venugopalan Huijuan Xu Jeff Donahue Rohrbach Marcus Mooney Raymond and Saenko Kate. 2015b. Translating Videos to Natural Language Using Deep Recurrent Neural Networks. In The North American Chapter of the Association for Computational Linguistics (NAACL) .  Subhashini Venugopalan Huijuan Xu Jeff Donahue Rohrbach Marcus Mooney Raymond and Saenko Kate. 2015b. Translating Videos to Natural Language Using Deep Recurrent Neural Networks. In The North American Chapter of the Association for Computational Linguistics (NAACL) .","DOI":"10.3115\/v1\/N15-1173"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00443"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/143"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123448"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123327"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.512"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.496"},{"key":"e_1_3_2_1_44_1","volume-title":"Recurrent Neural Network Regularization. CoRR","author":"Zaremba Wojciech","year":"2014"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/164"}],"event":{"name":"MM '19: The 27th ACM International Conference on Multimedia","location":"Nice France","acronym":"MM '19","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 27th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3350932","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3343031.3350932","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:17Z","timestamp":1750201997000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3350932"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,15]]},"references-count":45,"alternative-id":["10.1145\/3343031.3350932","10.1145\/3343031"],"URL":"https:\/\/doi.org\/10.1145\/3343031.3350932","relation":{},"subject":[],"published":{"date-parts":[[2019,10,15]]},"assertion":[{"value":"2019-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}