{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T04:15:51Z","timestamp":1750306551215,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","license":[{"start":{"date-parts":[[2015,10,13]],"date-time":"2015-10-13T00:00:00Z","timestamp":1444694400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61202166, 61472276"],"award-info":[{"award-number":["61202166, 61472276"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2015,10,13]]},"DOI":"10.1145\/2733373.2806314","type":"proceedings-article","created":{"date-parts":[[2016,2,26]],"date-time":"2016-02-26T19:09:21Z","timestamp":1456513761000},"page":"1191-1194","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":31,"title":["Summarization-based Video Caption via Deep Neural Networks"],"prefix":"10.1145","author":[{"given":"Guang","family":"Li","sequence":"first","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shubo","family":"Ma","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yahong","family":"Han","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2015,10,13]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"65","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","author":"Banerjee S.","year":"2005"},{"key":"e_1_3_2_1_2_1","first-page":"190","volume-title":"Proc. of the 49th Annual Meeting of the Association for Computational Linguistics","author":"Chen D. L.","year":"2011"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.21236\/ADA623249"},{"journal-title":"Journal of Artificial Intelligence Research, pages 457--479","year":"2004","author":"Erkan G.","key":"e_1_3_2_1_5_1"},{"volume-title":"From captions to visual concepts and back. arXiv preprint arXiv:1411.4952","year":"2014","author":"Fang H.","key":"e_1_3_2_1_6_1"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/1888089.1888092"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.337"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"journal-title":"Journal of Artificial Intelligence Research, pages 853--899","year":"2013","author":"Hodosh M.","key":"e_1_3_2_1_10_1"},{"volume-title":"Caffe: Convolutional architecture for fast feature embedding. arXiv preprint arXiv:1408.5093","year":"2014","author":"Jia Y.","key":"e_1_3_2_1_11_1"},{"volume-title":"Deep visual-semantic alignments for generating image descriptions. arXiv preprint arXiv:1412.2306","year":"2014","author":"Karpathy A.","key":"e_1_3_2_1_12_1"},{"key":"e_1_3_2_1_13_1","first-page":"1143","volume-title":"NIPS","author":"Ordonez V.","year":"2011"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.61"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"volume-title":"COLING","year":"2014","author":"Thomason J.","key":"e_1_3_2_1_17_1"},{"volume-title":"Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729","year":"2014","author":"Venugopalan S.","key":"e_1_3_2_1_18_1"},{"volume-title":"Show and tell: A neural image caption generator. arXiv preprint arXiv:1411.4555","year":"2014","author":"Vinyals O.","key":"e_1_3_2_1_19_1"}],"event":{"name":"MM '15: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Brisbane Australia","acronym":"MM '15"},"container-title":["Proceedings of the 23rd ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2733373.2806314","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2733373.2806314","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T06:12:40Z","timestamp":1750227160000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2733373.2806314"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,10,13]]},"references-count":19,"alternative-id":["10.1145\/2733373.2806314","10.1145\/2733373"],"URL":"https:\/\/doi.org\/10.1145\/2733373.2806314","relation":{},"subject":[],"published":{"date-parts":[[2015,10,13]]},"assertion":[{"value":"2015-10-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}