{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:56:38Z","timestamp":1783439798916,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,10,15]],"date-time":"2018-10-15T00:00:00Z","timestamp":1539561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Joint NSFC-ISF Research Program","award":["No.61561146397"],"award-info":[{"award-number":["No.61561146397"]}]},{"name":"the One Thousand Talents Plan of China"},{"name":"ARO grant","award":["W911NF-15-1-0290"],"award-info":[{"award-number":["W911NF-15-1-0290"]}]},{"name":"National Basic Research Program of China (973)","award":["No.2015CB352501, No.2015CB352502"],"award-info":[{"award-number":["No.2015CB352501, No.2015CB352502"]}]},{"name":"Tencent AI Lab Rhino-Bird Joint Research Program","award":["No.JR201805"],"award-info":[{"award-number":["No.JR201805"]}]},{"name":"Faculty Research Gift Awards by NEC Laboratories of America and Blippar"},{"DOI":"10.13039\/501100011002","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.61772310, No.61702300, No.61702302, No.61429201"],"award-info":[{"award-number":["No.61772310, No.61702300, No.61702302, No.61429201"]}],"id":[{"id":"10.13039\/501100011002","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,10,15]]},"DOI":"10.1145\/3240508.3240549","type":"proceedings-article","created":{"date-parts":[[2018,10,18]],"date-time":"2018-10-18T17:52:08Z","timestamp":1539885128000},"page":"843-851","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":212,"title":["Cross-modal Moment Localization in Videos"],"prefix":"10.1145","author":[{"given":"Meng","family":"Liu","sequence":"first","affiliation":[{"name":"Shandong University, Qingdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Wang","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liqiang","family":"Nie","sequence":"additional","affiliation":[{"name":"Shandong University, Qingdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qi","family":"Tian","sequence":"additional","affiliation":[{"name":"Huawei Noah's Ark Lab &amp; University of Texas at San Antonio, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baoquan","family":"Chen","sequence":"additional","affiliation":[{"name":"Peking University &amp; Shandong University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2018,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.49"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/69.755615"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.507"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3131288"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46487-9_47"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.392"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806349"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.470"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.493"},{"key":"e_1_3_2_1_12_1","volume-title":"Skip-thought Vectors. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 3294--3302","author":"Kiros Ryan","year":"2015","unstructured":"Ryan Kiros , Yukun Zhu , Ruslan R Salakhutdinov , Richard Zemel , Raquel Urtasun , Antonio Torralba , and Sanja Fidler . 2015 . Skip-thought Vectors. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 3294--3302 . Ryan Kiros, Yukun Zhu, Ruslan R Salakhutdinov, Richard Zemel, Raquel Urtasun, Antonio Torralba, and Sanja Fidler. 2015. Skip-thought Vectors. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 3294--3302."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_47"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.340"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123343"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123341"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.333"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210035"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.214"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.9"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_48"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46604-0_46"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00207"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/1027527.1027693"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2807417"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.119"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems. NIPS, 568--576","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014 a. Two-stream convolutional networks for action recognition in videos . In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 568--576 . Karen Simonyan and Andrew Zisserman. 2014a. Two-stream convolutional networks for action recognition in videos. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 568--576."},{"key":"e_1_3_2_1_29_1","volume-title":"Very Deep Convolutional Networks for Large-scale Image Recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014b. Very Deep Convolutional Networks for Large-scale Image Recognition. arXiv preprint arXiv:1409.1556 ( 2014 ), 1--14. Karen Simonyan and Andrew Zisserman. 2014b. Very Deep Convolutional Networks for Large-scale Image Recognition. arXiv preprint arXiv:1409.1556 (2014), 1--14."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.216"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/2072298.2072354"},{"key":"e_1_3_2_1_32_1","volume-title":"International Joint Conference on Artificial Intelligence. Morgan Kaufmann","author":"Song Young Chol","year":"2016","unstructured":"Young Chol Song , Iftekhar Naim , Abdullah Al Mamun , Kaustubh Kulkarni , Parag Singla , Jiebo Luo , Daniel Gildea , and Henry A Kautz . 2016 . Unsupervised Alignment of Actions in Video with Text Descriptions .. In International Joint Conference on Artificial Intelligence. Morgan Kaufmann , 2025--2031. Young Chol Song, Iftekhar Naim, Abdullah Al Mamun, Kaustubh Kulkarni, Parag Singla, Jiebo Luo, Daniel Gildea, and Henry A Kautz. 2016. Unsupervised Alignment of Actions in Video with Text Descriptions.. In International Joint Conference on Artificial Intelligence. Morgan Kaufmann, 2025--2031."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806226"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/1646396.1646442"},{"key":"e_1_3_2_1_35_1","volume-title":"Learning Language-Visual Embedding for Movie Understanding with Natural Language. arXiv preprint arXiv:1609.08124","author":"Torabi Atousa","year":"2016","unstructured":"Atousa Torabi , Niket Tandon , and Leonid Sigal . 2016. Learning Language-Visual Embedding for Movie Understanding with Natural Language. arXiv preprint arXiv:1609.08124 ( 2016 ), 1--13. Atousa Torabi, Niket Tandon, and Leonid Sigal. 2016. Learning Language-Visual Embedding for Movie Understanding with Natural Language. arXiv preprint arXiv:1609.08124 (2016), 1--13."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-013-0620-5"},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence. AAAI, 2346--2352","author":"Xu Ran","year":"2015","unstructured":"Ran Xu , Caiming Xiong , Wei Chen , and Jason J Corso . 2015 . Jointly Modeling Deep Video and Compositional Text to Bridge Vision and Language in a Unified Framework . In Proceedings of the AAAI Conference on Artificial Intelligence. AAAI, 2346--2352 . Ran Xu, Caiming Xiong, Wei Chen, and Jason J Corso. 2015. Jointly Modeling Deep Video and Compositional Text to Bridge Vision and Language in a Unified Framework. In Proceedings of the AAAI Conference on Artificial Intelligence. AAAI, 2346--2352."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/957013.957087"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/1027527.1027661"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics. ACL, 53--63","author":"Yu Haonan","year":"2013","unstructured":"Haonan Yu and Jeffrey Mark Siskind . 2013 . Grounded Language Learning from Video Described with Sentences . In Proceedings of the Annual Meeting of the Association for Computational Linguistics. ACL, 53--63 . Haonan Yu and Jeffrey Mark Siskind. 2013. Grounded Language Learning from Video Described with Sentences. In Proceedings of the Annual Meeting of the Association for Computational Linguistics. ACL, 53--63."},{"key":"e_1_3_2_1_41_1","volume-title":"MAttNet: Modular Attention Network for Referring Expression Comprehension. arXiv preprint arXiv:1801.08186","author":"Yu Licheng","year":"2018","unstructured":"Licheng Yu , Zhe Lin , Xiaohui Shen , Jimei Yang , Xin Lu , Mohit Bansal , and Tamara L Berg . 2018. MAttNet: Modular Attention Network for Referring Expression Comprehension. arXiv preprint arXiv:1801.08186 ( 2018 ), 1--14. Licheng Yu, Zhe Lin, Xiaohui Shen, Jimei Yang, Xin Lu, Mohit Bansal, and Tamara L Berg. 2018. MAttNet: Modular Attention Network for Referring Expression Comprehension. arXiv preprint arXiv:1801.08186 (2018), 1--14."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.375"},{"key":"e_1_3_2_1_44_1","volume-title":"Parallel Attention: A Unified Framework for Visual Object Discovery through Dialogs and Queries. arXiv preprint arXiv:1711.06370","author":"Zhuang Bohan","year":"2017","unstructured":"Bohan Zhuang , Qi Wu , Chunhua Shen , Ian Reid , and Anton van den Hengel . 2017 . Parallel Attention: A Unified Framework for Visual Object Discovery through Dialogs and Queries. arXiv preprint arXiv:1711.06370 (2017), 1--11. Bohan Zhuang, Qi Wu, Chunhua Shen, Ian Reid, and Anton van den Hengel. 2017. Parallel Attention: A Unified Framework for Visual Object Discovery through Dialogs and Queries. arXiv preprint arXiv:1711.06370 (2017), 1--11."},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the European Conference on Computer Vision. Springer, 391--405","author":"Lawrence Zitnick C","year":"2014","unstructured":"C Lawrence Zitnick and Piotr Doll\u00e1r . 2014 . Edge boxes: Locating Object Proposals from Edges . In Proceedings of the European Conference on Computer Vision. Springer, 391--405 . C Lawrence Zitnick and Piotr Doll\u00e1r. 2014. Edge boxes: Locating Object Proposals from Edges. In Proceedings of the European Conference on Computer Vision. Springer, 391--405."}],"event":{"name":"MM '18: ACM Multimedia Conference","location":"Seoul Republic of Korea","acronym":"MM '18","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 26th ACM international conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240549","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3240508.3240549","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T00:44:01Z","timestamp":1750207441000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3240508.3240549"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,10,15]]},"references-count":45,"alternative-id":["10.1145\/3240508.3240549","10.1145\/3240508"],"URL":"https:\/\/doi.org\/10.1145\/3240508.3240549","relation":{},"subject":[],"published":{"date-parts":[[2018,10,15]]},"assertion":[{"value":"2018-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}