{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T08:53:18Z","timestamp":1765356798297,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100004731","name":"Natural Science Foundation of Zhejiang Province","doi-asserted-by":"publisher","award":["LR19F020006"],"award-info":[{"award-number":["LR19F020006"]}],"id":[{"id":"10.13039\/501100004731","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Program of China","award":["2018AAA0101900, 2018AAA0100603"],"award-info":[{"award-number":["2018AAA0101900, 2018AAA0100603"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["2020QNA5024"],"award-info":[{"award-number":["2020QNA5024"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012659","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61625107, 61751209, 61836002"],"award-info":[{"award-number":["61625107, 61751209, 61836002"]}],"id":[{"id":"10.13039\/501100012659","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3413880","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T13:10:18Z","timestamp":1602508218000},"page":"1292-1301","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["Poet: Product-oriented Video Captioner for E-commerce"],"prefix":"10.1145","author":[{"given":"Shengyu","family":"Zhang","sequence":"first","affiliation":[{"name":"Zhejiang University, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ziqi","family":"Tan","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jin","family":"Yu","sequence":"additional","affiliation":[{"name":"Alibaba Group, Bei Jing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhou","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kun","family":"Kuang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Liu","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingren","family":"Zhou","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongxia","family":"Yang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fei","family":"Wu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hang Zhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"volume-title":"Proceedings of the 57th Conference of the Association for Computational Linguistics. 65--72","year":"2005","author":"Banerjee Satanjeev","key":"e_1_3_2_2_1_1"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351037"},{"volume-title":"Proceedings of the 57th Conference of the Association for Computational Linguistics.","author":"David","key":"e_1_3_2_2_3_1"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330725"},{"key":"e_1_3_2_2_5_1","unstructured":"Kyunghyun Cho Bart van Merrienboer Dzmitry Bahdanau and Yoshua Bengio. 2014. On the Properties of Neural Machine Translation: Encoder-Decoder Approaches. In SSST@EMNLP.  Kyunghyun Cho Bart van Merrienboer Dzmitry Bahdanau and Yoshua Bengio. 2014. On the Properties of Neural Machine Translation: Encoder-Decoder Approaches. In SSST@EMNLP."},{"volume":"201","journal-title":"Jason J. Corso.","author":"Das Pradipto","key":"e_1_3_2_2_6_1"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240566"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018303"},{"key":"e_1_3_2_2_9_1","unstructured":"Jonas Gehring Michael Auli David Grangier Denis Yarats and Yann N Dauphin. 2017. Convolutional sequence to sequence learning. In ICML.  Jonas Gehring Michael Auli David Grangier Denis Yarats and Yann N Dauphin. 2017. Convolutional sequence to sequence learning. In ICML."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"\u00c7aglar G\u00fcl\u00e7ehre Sarath Chandar Kyunghyun Cho and Yoshua Bengio. 2018. Dynamic Neural Turing Machine with Continuous and Discrete Addressing Schemes. Neural Computation (2018).  \u00c7aglar G\u00fcl\u00e7ehre Sarath Chandar Kyunghyun Cho and Yoshua Bengio. 2018. Dynamic Neural Turing Machine with Continuous and Discrete Addressing Schemes. Neural Computation (2018).","DOI":"10.1162\/neco_a_01060"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351072"},{"volume-title":"International Conference on Learning Representations.","year":"2017","author":"Kipf Thomas N","key":"e_1_3_2_2_12_1"},{"volume-title":"International Conference on Learning Representations.","year":"2017","author":"Kipf Thomas N","key":"e_1_3_2_2_13_1"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1230"},{"volume-title":"Proceedings of the 57th Conference of the Association for Computational Linguistics. 74--81","year":"2004","author":"Lin Chin-Yew","key":"e_1_3_2_2_15_1"},{"volume-title":"Deep Fashion Analysis with Feature Map Upsampling and Landmark-Driven Attention. In European Conference on Computer Vision. Springer.","year":"2018","author":"Liu Jingyuan","key":"e_1_3_2_2_16_1"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240667"},{"volume-title":"Fashion Landmark Detection in the Wild. In European Conference on Computer Vision.","year":"2016","author":"Liu Ziwei","key":"e_1_3_2_2_18_1"},{"volume-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In Advances in Neural Information Processing Systems.","year":"2019","author":"Lu Jiasen","key":"e_1_3_2_2_19_1"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016810"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.117"},{"volume-title":"Proceedings of the 57th Conference of the Association for Computational Linguistics. Association for Computational Linguistics, 311--318","year":"2002","author":"Papineni Kishore","key":"e_1_3_2_2_22_1"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Xufeng Qian Yueting Zhuang Yimeng Li Shaoning Xiao Shiliang Pu and Jun Xiao. 2019. Video Relation Detection with Spatio-Temporal Graph. In MM.  Xufeng Qian Yueting Zhuang Yimeng Li Shaoning Xiao Shiliang Pu and Jun Xiao. 2019. Video Relation Detection with Spatio-Temporal Graph. In MM.","DOI":"10.1145\/3343031.3351058"},{"volume-title":"Grounding Action Descriptions in Videos. TProceedings of the Conference of the Association for Computational Linguistics","year":"2013","author":"Regneri Michaela","key":"e_1_3_2_2_24_1"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Anna Rohrbach Marcus Rohrbach Wei Qiu Annemarie Friedrich Manfred Pinkal and Bernt Schiele. 2014. Coherent Multi-sentence Video Description with Variable Level of Detail. In GCPR.  Anna Rohrbach Marcus Rohrbach Wei Qiu Annemarie Friedrich Manfred Pinkal and Bernt Schiele. 2014. Coherent Multi-sentence Video Description with Variable Level of Detail. In GCPR.","DOI":"10.1007\/978-3-319-11752-2_15"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298940"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351060"},{"volume-title":"Hollywood in Homes: Crowdsourcing Data Collection for Activity Understanding. In European Conference on Computer Vision.","year":"2016","author":"Sigurdsson Gunnar A.","key":"e_1_3_2_2_28_1"},{"volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence.","year":"2017","author":"Speer Robyn","key":"e_1_3_2_2_29_1"},{"key":"e_1_3_2_2_30_1","unstructured":"Sainbayar Sukhbaatar Arthur Szlam Jason Weston and Rob Fergus. 2015. End-To-End Memory Networks. In Advances in Neural Information Processing Systems.  Sainbayar Sukhbaatar Arthur Szlam Jason Weston and Rob Fergus. 2015. End-To-End Memory Networks. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3127895"},{"key":"e_1_3_2_2_32_1","unstructured":"Atousa Torabi Christopher J. Pal Hugo Larochelle and Aaron C. Courville. 2015. Using Descriptive Video Services to Create a Large Data Source for Video Annotation Research. Arxiv (2015).  Atousa Torabi Christopher J. Pal Hugo Larochelle and Aaron C. Courville. 2015. Using Descriptive Video Services to Create a Large Data Source for Video Annotation Research. Arxiv (2015)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1204"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.515"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/N15-1173"},{"volume-title":"Reconstruction Network for Video Captioning. In IEEE Conference on Computer Vision and Pattern Recognition.","year":"2018","author":"Wang Bairui","key":"e_1_3_2_2_37_1"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240677"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240538"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_25"},{"volume-title":"Memory Networks. In International Conference on Learning Representations.","year":"2015","author":"Weston Jason","key":"e_1_3_2_2_41_1"},{"volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing.","author":"Whitehead Spencer","key":"e_1_3_2_2_42_1"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123448"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123327"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3127904"},{"volume-title":"IEEE\/CVF International Conference on Computer Vision.","author":"Yao Li","key":"e_1_3_2_2_46_1"},{"volume-title":"Title Generation for User Generated Videos. In European Conference on Computer Vision.","year":"2016","author":"Zeng Kuo-Hao","key":"e_1_3_2_2_47_1"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00719"},{"volume-title":"Object-Aware Aggregation With Bidirectional Temporal Graph for Video Captioning. In IEEE Conference on Computer Vision and Pattern Recognition.","year":"2019","author":"Zhang Junchao","key":"e_1_3_2_2_49_1"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350932"}],"event":{"name":"MM '20: The 28th ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Seattle WA USA","acronym":"MM '20"},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413880","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3413880","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:32:06Z","timestamp":1750195926000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413880"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":50,"alternative-id":["10.1145\/3394171.3413880","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3413880","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}