{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T02:39:32Z","timestamp":1784860772515,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,6,27]],"date-time":"2018-06-27T00:00:00Z","timestamp":1530057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Basic Research Program of China (973)","award":["No.2015CB352501, No.2015CB352502"],"award-info":[{"award-number":["No.2015CB352501, No.2015CB352502"]}]},{"name":"One Thousand Talents Plan of China","award":["No.11150087963001"],"award-info":[{"award-number":["No.11150087963001"]}]},{"name":"National Natural Science Foundation of China","award":["No.61772310"],"award-info":[{"award-number":["No.61772310"]}]},{"name":"National Research Foundation, Prime Ministers Office, Singapore"},{"name":"Joint NSFC-ISF Research Program","award":["No.61561146397"],"award-info":[{"award-number":["No.61561146397"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,6,27]]},"DOI":"10.1145\/3209978.3210003","type":"proceedings-article","created":{"date-parts":[[2018,7,2]],"date-time":"2018-07-02T12:12:40Z","timestamp":1530533560000},"page":"15-24","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":247,"title":["Attentive Moment Retrieval in Videos"],"prefix":"10.1145","author":[{"given":"Meng","family":"Liu","sequence":"first","affiliation":[{"name":"Shandong University, Qingdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Wang","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liqiang","family":"Nie","sequence":"additional","affiliation":[{"name":"Shandong University, Qingdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiangnan","family":"He","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baoquan","family":"Chen","sequence":"additional","affiliation":[{"name":"Shandong University, Qingdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2018,6,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Unsupervised Learning from Narrated Instruction Videos Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Alayrac Jean-Baptiste","unstructured":"Jean-Baptiste Alayrac , Piotr Bojanowski , Nishant Agrawal , Josef Sivic , Ivan Laptev , and Simon Lacoste-Julien . 2016. Unsupervised Learning from Narrated Instruction Videos Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition . IEEE , 4575--4583. Jean-Baptiste Alayrac, Piotr Bojanowski, Nishant Agrawal, Josef Sivic, Ivan Laptev, and Simon Lacoste-Julien . 2016. Unsupervised Learning from Narrated Instruction Videos Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 4575--4583."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3017429"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080779"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911451.2914765"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3186064"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080773"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems. NIPS, 2121--2129","author":"Frome Andrea","year":"2013","unstructured":"Andrea Frome , Greg S Corrado , Jon Shlens , Samy Bengio , Jeff Dean , Tomas Mikolov , 2013 . Devise: A Deep Visual-semantic Embedding Model . In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 2121--2129 . Andrea Frome, Greg S Corrado, Jon Shlens, Samy Bengio, Jeff Dean, Tomas Mikolov, et almbox. . 2013. Devise: A Deep Visual-semantic Embedding Model. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 2121--2129."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995646"},{"key":"e_1_3_2_1_10_1","volume-title":"2017 a. TALL: Temporal Activity Localization via Language Query Proceedings of the IEEE International Conference on Computer Vision","author":"Gao Jiyang","unstructured":"Jiyang Gao , Chen Sun , Zhenheng Yang , and Ram Nevatia . 2017 a. TALL: Temporal Activity Localization via Language Query Proceedings of the IEEE International Conference on Computer Vision . IEEE , 5267--5275. Jiyang Gao, Chen Sun, Zhenheng Yang, and Ram Nevatia . 2017 a. TALL: Temporal Activity Localization via Language Query Proceedings of the IEEE International Conference on Computer Vision. IEEE, 5267--5275."},{"key":"e_1_3_2_1_11_1","volume-title":"TURN TAP: Temporal Unit Regression Network for Temporal Action Proposals Proceedings of the IEEE International Conference on Computer Vision. IEEE, 3628--3636","author":"Gao Jiyang","year":"2017","unstructured":"Jiyang Gao , Zhenheng Yang , Chen Sun , Kan Chen , and Ram Nevatia . 2017 b . TURN TAP: Temporal Unit Regression Network for Temporal Action Proposals Proceedings of the IEEE International Conference on Computer Vision. IEEE, 3628--3636 . Jiyang Gao, Zhenheng Yang, Chen Sun, Kan Chen, and Ram Nevatia . 2017 b. TURN TAP: Temporal Unit Regression Network for Temporal Action Proposals Proceedings of the IEEE International Conference on Computer Vision. IEEE, 3628--3636."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080777"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052569"},{"key":"e_1_3_2_1_14_1","volume-title":"Attention-based Multimodal Fusion for Video Description Proceedings of the IEEE International Conference on Computer Vision. IEEE, 4203--4212","author":"Hori Chiori","year":"2017","unstructured":"Chiori Hori , Takaaki Hori , Teng-Yok Lee , Ziming Zhang , Bret Harsham , John R Hershey , Tim K Marks , and Kazuhiko Sumi . 2017 . Attention-based Multimodal Fusion for Video Description Proceedings of the IEEE International Conference on Computer Vision. IEEE, 4203--4212 . Chiori Hori, Takaaki Hori, Teng-Yok Lee, Ziming Zhang, Bret Harsham, John R Hershey, Tim K Marks, and Kazuhiko Sumi . 2017. Attention-based Multimodal Fusion for Video Description Proceedings of the IEEE International Conference on Computer Vision. IEEE, 4203--4212."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.493"},{"key":"e_1_3_2_1_16_1","volume-title":"Deep Visual-semantic Alignments for Generating Image Descriptions Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 3128--3137","author":"Karpathy Andrej","year":"2015","unstructured":"Andrej Karpathy and Li Fei-Fei . 2015 . Deep Visual-semantic Alignments for Generating Image Descriptions Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 3128--3137 . Andrej Karpathy and Li Fei-Fei . 2015. Deep Visual-semantic Alignments for Generating Image Descriptions Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 3128--3137."},{"key":"e_1_3_2_1_17_1","volume-title":"Skip-thought Vectors. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 3294--3302","author":"Kiros Ryan","year":"2015","unstructured":"Ryan Kiros , Yukun Zhu , Ruslan R Salakhutdinov , Richard Zemel , Raquel Urtasun , Antonio Torralba , and Sanja Fidler . 2015 . Skip-thought Vectors. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 3294--3302 . Ryan Kiros, Yukun Zhu, Ruslan R Salakhutdinov, Richard Zemel, Raquel Urtasun, Antonio Torralba, and Sanja Fidler . 2015. Skip-thought Vectors. In Proceedings of the Advances in Neural Information Processing Systems. NIPS, 3294--3302."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.340"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.340"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123341"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Minh-Thang Luong Hieu Pham and Christopher D Manning . 2015. Effective Approaches to Attention-based Neural Machine Translation Proceedings of the Conference on Empirical Methods in Natural Language Processing. ACL 1412--1421.  Minh-Thang Luong Hieu Pham and Christopher D Manning . 2015. Effective Approaches to Attention-based Neural Machine Translation Proceedings of the Conference on Empirical Methods in Natural Language Processing. ACL 1412--1421.","DOI":"10.18653\/v1\/D15-1166"},{"key":"e_1_3_2_1_22_1","volume-title":"Learning Joint Representations of Videos and Sentences with Web Image Search Proceedings of the European Conference on Computer Vision. Springer, 651--667","author":"Otani Mayu","year":"2016","unstructured":"Mayu Otani , Yuta Nakashima , Esa Rahtu , Janne Heikkil\"a, and Naokazu Yokoya . 2016 . Learning Joint Representations of Videos and Sentences with Web Image Search Proceedings of the European Conference on Computer Vision. Springer, 651--667 . Mayu Otani, Yuta Nakashima, Esa Rahtu, Janne Heikkil\"a, and Naokazu Yokoya . 2016. Learning Joint Representations of Videos and Sentences with Web Image Search Proceedings of the European Conference on Computer Vision. Springer, 651--667."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3084144"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00207"},{"key":"e_1_3_2_1_26_1","volume-title":"Faster R-CNN: Towards Real-time Object Detection with Region Proposal Networks Proceedings of the Advances in Neural Information Processing Systems. NIPS, 91--99","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren , Kaiming He , Ross Girshick , and Jian Sun . 2015 . Faster R-CNN: Towards Real-time Object Detection with Region Proposal Networks Proceedings of the Advances in Neural Information Processing Systems. NIPS, 91--99 . Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun . 2015. Faster R-CNN: Towards Real-time Object Detection with Region Proposal Networks Proceedings of the Advances in Neural Information Processing Systems. NIPS, 91--99."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33718-5_11"},{"key":"e_1_3_2_1_28_1","volume-title":"Temporal Action Localization in Untrimmed Videos via Multi-stage CNNs Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","author":"Shou Zheng","unstructured":"Zheng Shou , Dongang Wang , and Shih-Fu Chang . 2016. Temporal Action Localization in Untrimmed Videos via Multi-stage CNNs Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition . IEEE , 1049--1058. Zheng Shou, Dongang Wang, and Shih-Fu Chang . 2016. Temporal Action Localization in Untrimmed Videos via Multi-stage CNNs Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 1049--1058."},{"key":"e_1_3_2_1_29_1","volume-title":"Very Deep Convolutional Networks for Large-scale Image Recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman . 2014. Very Deep Convolutional Networks for Large-scale Image Recognition. arXiv preprint arXiv:1409.1556 ( 2014 ). Karen Simonyan and Andrew Zisserman . 2014. Very Deep Convolutional Networks for Large-scale Image Recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.216"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00177"},{"key":"e_1_3_2_1_32_1","volume-title":"2017 b. Hierarchical LSTM with Adjusted Temporal Attention for Video Captioning. arXiv preprint arXiv:1706.01231","author":"Song Jingkuan","year":"2017","unstructured":"Jingkuan Song , Zhao Guo , Lianli Gao , Wu Liu , Dongxiang Zhang , and Heng Tao Shen . 2017 b. Hierarchical LSTM with Adjusted Temporal Attention for Video Captioning. arXiv preprint arXiv:1706.01231 ( 2017 ). Jingkuan Song, Zhao Guo, Lianli Gao, Wu Liu, Dongxiang Zhang, and Heng Tao Shen . 2017 b. Hierarchical LSTM with Adjusted Temporal Attention for Video Captioning. arXiv preprint arXiv:1706.01231 (2017)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123314"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/1646396.1646442"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/1961209.1961214"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3186066"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080771"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3052774"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"e_1_3_2_1_41_1","volume-title":"MSR-VTT: A Large Video Description Dataset for Bridging Video and Language Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 5288--5296","author":"Xu Jun","year":"2016","unstructured":"Jun Xu , Tao Mei , Ting Yao , and Yong Rui . 2016 . MSR-VTT: A Large Video Description Dataset for Bridging Video and Language Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 5288--5296 . Jun Xu, Tao Mei, Ting Yao, and Yong Rui . 2016. MSR-VTT: A Large Video Description Dataset for Bridging Video and Language Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 5288--5296."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123448"},{"key":"e_1_3_2_1_43_1","volume-title":"Attend and Tell: Neural Image Caption Generation with Visual Attention International Conference on Machine Learning. ACM","author":"Xu Kelvin","year":"2015","unstructured":"Kelvin Xu , Jimmy Ba , Ryan Kiros , Kyunghyun Cho , Aaron Courville , Ruslan Salakhudinov , Rich Zemel , and Yoshua Bengio . 2015 a. Show , Attend and Tell: Neural Image Caption Generation with Visual Attention International Conference on Machine Learning. ACM , 2048--2057. Kelvin Xu, Jimmy Ba, Ryan Kiros, Kyunghyun Cho, Aaron Courville, Ruslan Salakhudinov, Rich Zemel, and Yoshua Bengio . 2015 a. Show, Attend and Tell: Neural Image Caption Generation with Visual Attention International Conference on Machine Learning. ACM, 2048--2057."},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the American Association for Artificial Intelligence","volume":"5","author":"Xu Ran","year":"2015","unstructured":"Ran Xu , Caiming Xiong , Wei Chen , and Jason J Corso . 2015 b. Jointly Modeling Deep Video and Compositional Text to Bridge Vision and Language in a Unified Framework .. In Proceedings of the American Association for Artificial Intelligence , Vol. Vol. 5 . AAAI, 6. Ran Xu, Caiming Xiong, Wei Chen, and Jason J Corso . 2015 b. Jointly Modeling Deep Video and Compositional Text to Bridge Vision and Language in a Unified Framework.. In Proceedings of the American Association for Artificial Intelligence, Vol. Vol. 5. AAAI, 6."},{"key":"e_1_3_2_1_45_1","volume-title":"Stacked Attention Networks for Image Question Answering Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 21--29","author":"Yang Zichao","year":"2016","unstructured":"Zichao Yang , Xiaodong He , Jianfeng Gao , Li Deng , and Alex Smola . 2016 . Stacked Attention Networks for Image Question Answering Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 21--29 . Zichao Yang, Xiaodong He, Jianfeng Gao, Li Deng, and Alex Smola . 2016. Stacked Attention Networks for Image Question Answering Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 21--29."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080655"},{"key":"e_1_3_2_1_47_1","volume-title":"Visual Translation Embedding Network for Visual Relation Detection Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 5532--5540","author":"Zhang Hanwang","year":"2017","unstructured":"Hanwang Zhang , Zawlin Kyaw , Shih-Fu Chang , and Tat-Seng Chua . 2017 . Visual Translation Embedding Network for Visual Relation Detection Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 5532--5540 . Hanwang Zhang, Zawlin Kyaw, Shih-Fu Chang, and Tat-Seng Chua . 2017. Visual Translation Embedding Network for Visual Relation Detection Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 5532--5540."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123364"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Zhou Zhao Qifan Yang Deng Cai Xiaofei He and Yueting Zhuang . 2017 b. Video Question Answering via Hierarchical Spatio-Temporal Attention Networks Proceedings of the International Joint Conference on Artificial Intelligence. Morgan Kaufmann 3518--3524.   Zhou Zhao Qifan Yang Deng Cai Xiaofei He and Yueting Zhuang . 2017 b. Video Question Answering via Hierarchical Spatio-Temporal Attention Networks Proceedings of the International Joint Conference on Artificial Intelligence. Morgan Kaufmann 3518--3524.","DOI":"10.24963\/ijcai.2017\/492"}],"event":{"name":"SIGIR '18: The 41st International ACM SIGIR conference on research and development in Information Retrieval","location":"Ann Arbor MI USA","acronym":"SIGIR '18","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["The 41st International ACM SIGIR Conference on Research &amp; Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3209978.3210003","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3209978.3210003","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:07:48Z","timestamp":1750212468000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3209978.3210003"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,6,27]]},"references-count":49,"alternative-id":["10.1145\/3209978.3210003","10.1145\/3209978"],"URL":"https:\/\/doi.org\/10.1145\/3209978.3210003","relation":{},"subject":[],"published":{"date-parts":[[2018,6,27]]},"assertion":[{"value":"2018-06-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}