{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:17:09Z","timestamp":1750220229181,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,6,27]],"date-time":"2022-06-27T00:00:00Z","timestamp":1656288000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61906210"],"award-info":[{"award-number":["61906210"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Grand R&D Plan","award":["2020AAA0103501"],"award-info":[{"award-number":["2020AAA0103501"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,6,27]]},"DOI":"10.1145\/3512527.3531396","type":"proceedings-article","created":{"date-parts":[[2022,6,23]],"date-time":"2022-06-23T22:23:32Z","timestamp":1656023012000},"page":"369-379","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Joint Modality Synergy and Spatio-temporal Cue Purification for Moment Localization"],"prefix":"10.1145","author":[{"given":"Xingyu","family":"Shen","sequence":"first","affiliation":[{"name":"Science and Technology on Parallel and Distributed Processing, National University of Defense Technology,College of Computer, National University of Defense Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Long","family":"Lan","sequence":"additional","affiliation":[{"name":"Institute for Quantum Information and State Key Laboratory of High Performance Computing, National University of Defense Technology, College of Computer, National University of Defense Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huibin","family":"Tan","sequence":"additional","affiliation":[{"name":"Institute for Quantum Information and State Key Laboratory of High Performance Computing, National University of Defense Technology, College of Computer, National University of Defense Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute for Quantum Information and State Key Laboratory of High Performance Computing, National University of Defense Technology, College of Computer, National University of Defense Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xurui","family":"Ma","sequence":"additional","affiliation":[{"name":"Science and Technology on Parallel and Distributed Processing, National University of Defense Technology, College of Computer, National University of Defense Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhigang","family":"Luo","sequence":"additional","affiliation":[{"name":"Science and Technology on Parallel and Distributed Processing, National University of Defense Technology, College of Computer, National University of Defense Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,6,27]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Can Zhang, and Yuexian Zou.","author":"Cao Meng","year":"2021","unstructured":"Meng Cao , Long Chen , Mike Zheng Shou , Can Zhang, and Yuexian Zou. 2021 . On Pursuit of Designing Multi-modal Transformer for Video Grounding. In EMNLP. 9810--9823. Meng Cao, Long Chen, Mike Zheng Shou, Can Zhang, and Yuexian Zou. 2021. On Pursuit of Designing Multi-modal Transformer for Video Grounding. In EMNLP. 9810--9823."},{"volume-title":"Action Recognition? A New Model and the Kinetics Dataset","author":"Carreira Jo","key":"e_1_3_2_2_2_1","unstructured":"Jo a o Carreira and Andrew Zisserman . 2017. Quo Vadis , Action Recognition? A New Model and the Kinetics Dataset . In CVPR. IEEE Computer Society , 4724--4733. Jo a o Carreira and Andrew Zisserman. 2017. Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset. In CVPR. IEEE Computer Society, 4724--4733."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Jingyuan Chen Xinpeng Chen Lin Ma Zequn Jie and Tat-Seng Chua. 2018. Temporally Grounding Natural Sentence in Video. In EMNLP. 162--171. Jingyuan Chen Xinpeng Chen Lin Ma Zequn Jie and Tat-Seng Chua. 2018. Temporally Grounding Natural Sentence in Video. In EMNLP. 162--171.","DOI":"10.18653\/v1\/D18-1015"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Long Chen Chujie Lu Siliang Tang Jun Xiao Dong Zhang Chilie Tan and Xiaolin Li. 2020. Rethinking the Bottom-Up Framework for Query-Based Video Localization. In AAAI. 10551--10558. Long Chen Chujie Lu Siliang Tang Jun Xiao Dong Zhang Chilie Tan and Xiaolin Li. 2020. Rethinking the Bottom-Up Framework for Query-Based Video Localization. In AAAI. 10551--10558.","DOI":"10.1609\/aaai.v34i07.6627"},{"key":"e_1_3_2_2_5_1","first-page":"601","article-title":"Hierarchical Visual-Textual Graph for Temporal Activity Localization via Language","volume":"12365","author":"Chen Shaoxiang","year":"2020","unstructured":"Shaoxiang Chen and Yu-Gang Jiang . 2020 . Hierarchical Visual-Textual Graph for Temporal Activity Localization via Language . In ECCV , Vol. 12365. 601 -- 618 . Shaoxiang Chen and Yu-Gang Jiang. 2020. Hierarchical Visual-Textual Graph for Temporal Activity Localization via Language. In ECCV, Vol. 12365. 601--618.","journal-title":"ECCV"},{"key":"e_1_3_2_2_6_1","volume-title":"TALL: Temporal Activity Localization via Language Query. In ICCV Italy, October 22--29. 5277--5285.","author":"Gao Jiyang","year":"2017","unstructured":"Jiyang Gao , Chen Sun , Zhenheng Yang , and Ram Nevatia . 2017 . TALL: Temporal Activity Localization via Language Query. In ICCV Italy, October 22--29. 5277--5285. Jiyang Gao, Chen Sun, Zhenheng Yang, and Ram Nevatia. 2017. TALL: Temporal Activity Localization via Language Query. In ICCV Italy, October 22--29. 5277--5285."},{"key":"e_1_3_2_2_7_1","volume-title":"Zhaohan Guo, Mohammad Gheshlaghi Azar, Bilal Piot, Koray Kavukcuoglu, R\u00e9 mi Munos, and Michal Valko.","author":"Grill Jean-Bastien","year":"2020","unstructured":"Jean-Bastien Grill , Florian Strub , Florent Altch\u00e9 , Corentin Tallec , Pierre H. Richemond , Elena Buchatskaya , Carl Doersch , Bernardo \u00c1 vila Pires , Zhaohan Guo, Mohammad Gheshlaghi Azar, Bilal Piot, Koray Kavukcuoglu, R\u00e9 mi Munos, and Michal Valko. 2020 . Bootstrap Your Own Latent - A New Approach to Self-Supervised Learning. In NIPS . Jean-Bastien Grill, Florian Strub, Florent Altch\u00e9, Corentin Tallec, Pierre H. Richemond, Elena Buchatskaya, Carl Doersch, Bernardo \u00c1 vila Pires, Zhaohan Guo, Mohammad Gheshlaghi Azar, Bilal Piot, Koray Kavukcuoglu, R\u00e9 mi Munos, and Michal Valko. 2020. Bootstrap Your Own Latent - A New Approach to Self-Supervised Learning. In NIPS ."},{"key":"e_1_3_2_2_8_1","volume-title":"Russell","author":"Hendricks Lisa Anne","year":"2017","unstructured":"Lisa Anne Hendricks , Oliver Wang , Eli Shechtman , Josef Sivic , Trevor Darrell , and Bryan C . Russell . 2017 . Localizing Moments in Video with Natural Language. In ICCV. 5804--5813. Lisa Anne Hendricks, Oliver Wang, Eli Shechtman, Josef Sivic, Trevor Darrell, and Bryan C. Russell. 2017. Localizing Moments in Video with Natural Language. In ICCV. 5804--5813."},{"key":"e_1_3_2_2_9_1","unstructured":"R. Devon Hjelm Alex Fedorov Samuel Lavoie-Marchildon Karan Grewal Philip Bachman Adam Trischler and Yoshua Bengio. 2019. Learning deep representations by mutual information estimation and maximization. In ICLR . R. Devon Hjelm Alex Fedorov Samuel Lavoie-Marchildon Karan Grewal Philip Bachman Adam Trischler and Yoshua Bengio. 2019. Learning deep representations by mutual information estimation and maximization. In ICLR ."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Jingtao Hu En Zhu Siqi Wang Siwei Wang Xinwang Liu and Jianping Yin. 2019. Two-stage Unsupervised Video Anomaly Detection using Low-rank based Unsupervised One-class Learning with Ridge Regression. In IJCNN. 1--8. Jingtao Hu En Zhu Siqi Wang Siwei Wang Xinwang Liu and Jianping Yin. 2019. Two-stage Unsupervised Video Anomaly Detection using Low-rank based Unsupervised One-class Learning with Ridge Regression. In IJCNN. 1--8.","DOI":"10.1109\/IJCNN.2019.8852022"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Bin Jiang Xin Huang Chao Yang and Junsong Yuan. 2019. Cross-Modal Video Moment Retrieval with Spatial and Language-Temporal Attention. In ICMR. 217--225. Bin Jiang Xin Huang Chao Yang and Junsong Yuan. 2019. Cross-Modal Video Moment Retrieval with Spatial and Language-Temporal Attention. In ICMR. 217--225.","DOI":"10.1145\/3323873.3325019"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Ranjay Krishna Kenji Hata Frederic Ren Li Fei-Fei and Juan Carlos Niebles. 2017. Dense-Captioning Events in Videos. In ICCV. 706--715. Ranjay Krishna Kenji Hata Frederic Ren Li Fei-Fei and Juan Carlos Niebles. 2017. Dense-Captioning Events in Videos. In ICCV. 706--715.","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_2_13_1","volume-title":"A Survey on Temporal Sentence Grounding in Videos. CoRR","author":"Lan Xiaohan","year":"2021","unstructured":"Xiaohan Lan , Yitian Yuan , Xin Wang , Zhi Wang , and Wenwu Zhu . 2021. A Survey on Temporal Sentence Grounding in Videos. CoRR , Vol. abs\/ 2109 .08039 ( 2021 ). Xiaohan Lan, Yitian Yuan, Xin Wang, Zhi Wang, and Wenwu Zhu. 2021. A Survey on Temporal Sentence Grounding in Videos. CoRR, Vol. abs\/2109.08039 (2021)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2965987"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"crossref","unstructured":"Meng Liu Xiang Wang Liqiang Nie Xiangnan He Baoquan Chen and Tat-Seng Chua. 2018a. Attentive Moment Retrieval in Videos. In SIGIR. 15--24. Meng Liu Xiang Wang Liqiang Nie Xiangnan He Baoquan Chen and Tat-Seng Chua. 2018a. Attentive Moment Retrieval in Videos. In SIGIR. 15--24.","DOI":"10.1145\/3209978.3210003"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Meng Liu Xiang Wang Liqiang Nie Qi Tian Baoquan Chen and Tat-Seng Chua. 2018b. Cross-modal Moment Localization in Videos. In ACM MM. ACM 843--851. Meng Liu Xiang Wang Liqiang Nie Qi Tian Baoquan Chen and Tat-Seng Chua. 2018b. Cross-modal Moment Localization in Videos. In ACM MM. ACM 843--851.","DOI":"10.1145\/3240508.3240549"},{"key":"e_1_3_2_2_17_1","volume-title":"Local-Global Video-Text Interactions for Temporal Grounding. In CVPR","author":"Mun Jonghwan","year":"2020","unstructured":"Jonghwan Mun , Minsu Cho , and Bohyung Han . 2020 . Local-Global Video-Text Interactions for Temporal Grounding. In CVPR 2020. 10807--10816. Jonghwan Mun, Minsu Cho, and Bohyung Han. 2020. Local-Global Video-Text Interactions for Temporal Grounding. In CVPR 2020. 10807--10816."},{"key":"e_1_3_2_2_18_1","unstructured":"Guoshun Nan Rui Qiao Yao Xiao Jun Liu Sicong Leng Hao Zhang and Wei Lu. 2021. Interventional Video Grounding With Dual Contrastive Learning. In CVPR. 2765--2775. Guoshun Nan Rui Qiao Yao Xiao Jun Liu Sicong Leng Hao Zhang and Wei Lu. 2021. Interventional Video Grounding With Dual Contrastive Learning. In CVPR. 2765--2775."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Cristian Rodriguez Opazo Edison Marrese-Taylor Basura Fernando Hongdong Li and Stephen Gould. 2021. DORi: Discovering Object Relationships for Moment Localization of a Natural Language Query in a Video. In WACV. 1078--1087. Cristian Rodriguez Opazo Edison Marrese-Taylor Basura Fernando Hongdong Li and Stephen Gould. 2021. DORi: Discovering Object Relationships for Moment Localization of a Natural Language Query in a Video. In WACV. 1078--1087.","DOI":"10.1109\/WACV48630.2021.00112"},{"key":"e_1_3_2_2_20_1","volume-title":"Hongdong Li, and Stephen Gould.","author":"Opazo Cristian Rodriguez","year":"2020","unstructured":"Cristian Rodriguez Opazo , Edison Marrese-Taylor , Fatemeh Sadat Saleh , Hongdong Li, and Stephen Gould. 2020 . Proposal-free Temporal Moment Localization of a Natural-Language Query in Video using Guided Attention. In WACV. 2453--2462. Cristian Rodriguez Opazo, Edison Marrese-Taylor, Fatemeh Sadat Saleh, Hongdong Li, and Stephen Gould. 2020. Proposal-free Temporal Moment Localization of a Natural-Language Query in Video using Guided Attention. In WACV. 2453--2462."},{"key":"e_1_3_2_2_21_1","first-page":"144","article-title":"Script Data for Attribute-Based Recognition of Composite Activities","volume":"7572","author":"Rohrbach Marcus","year":"2012","unstructured":"Marcus Rohrbach , Michaela Regneri , Mykhaylo Andriluka , Sikandar Amin , Manfred Pinkal , and Bernt Schiele . 2012 . Script Data for Attribute-Based Recognition of Composite Activities . In ECCV , Vol. 7572. 144 -- 157 . Marcus Rohrbach, Michaela Regneri, Mykhaylo Andriluka, Sikandar Amin, Manfred Pinkal, and Bernt Schiele. 2012. Script Data for Attribute-Based Recognition of Composite Activities. In ECCV, Vol. 7572. 144--157.","journal-title":"ECCV"},{"key":"e_1_3_2_2_22_1","volume-title":"Xiaolong Wang, Ali Farhadi, Ivan Laptev, and Abhinav Gupta.","author":"Sigurdsson Gunnar A.","year":"2016","unstructured":"Gunnar A. Sigurdsson , G\u00fc l Varol , Xiaolong Wang, Ali Farhadi, Ivan Laptev, and Abhinav Gupta. 2016 . Hollywood in Homes : Crowdsourcing Data Collection for Activity Understanding. In ECCV (Lecture Notes in Computer Science , Vol. 9905). 510-- 526 . Gunnar A. Sigurdsson, G\u00fc l Varol, Xiaolong Wang, Ali Farhadi, Ivan Laptev, and Abhinav Gupta. 2016. Hollywood in Homes: Crowdsourcing Data Collection for Activity Understanding. In ECCV (Lecture Notes in Computer Science, Vol. 9905). 510--526."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Zhang Songyang Peng Houwen Fu Jianlong and Luo Jiebo. 2020. Learning 2D Temporal Adjacent Networks for Moment Localization with Natural Language. In AAAI. 12870--12877. Zhang Songyang Peng Houwen Fu Jianlong and Luo Jiebo. 2020. Learning 2D Temporal Adjacent Networks for Moment Localization with Natural Language. In AAAI. 12870--12877.","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"e_1_3_2_2_24_1","volume-title":"LXMERT: Learning Cross-Modality Encoder Representations from Transformers","author":"Tan Hao","year":"2019","unstructured":"Hao Tan and Mohit Bansal . 2019 . LXMERT: Learning Cross-Modality Encoder Representations from Transformers . In EMNLP-IJCNLP. Association for Computational Linguistics , 5099--5110. Hao Tan and Mohit Bansal. 2019. LXMERT: Learning Cross-Modality Encoder Representations from Transformers. In EMNLP-IJCNLP. Association for Computational Linguistics, 5099--5110."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Du Tran Lubomir D. Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning Spatiotemporal Features with 3D Convolutional Networks. In ICCV. 4489--4497. Du Tran Lubomir D. Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning Spatiotemporal Features with 3D Convolutional Networks. In ICCV. 4489--4497.","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"crossref","unstructured":"Jingwen Wang Lin Ma and Wenhao Jiang. 2020. Temporally Grounding Language Queries in Videos by Contextual Boundary-Aware Prediction. In AAAI. 12168--12175. Jingwen Wang Lin Ma and Wenhao Jiang. 2020. Temporally Grounding Language Queries in Videos by Contextual Boundary-Aware Prediction. In AAAI. 12168--12175.","DOI":"10.1609\/aaai.v34i07.6897"},{"key":"e_1_3_2_2_27_1","unstructured":"Guang Yu Siqi Wang Zhiping Cai En Zhu Chuanfu Xu Jianping Yin and Marius Kloft. 2020. Cloze Test Helps: Effective Video Anomaly Detection via Learning to Complete Video Events. In ACM MM. 583--591. Guang Yu Siqi Wang Zhiping Cai En Zhu Chuanfu Xu Jianping Yin and Marius Kloft. 2020. Cloze Test Helps: Effective Video Anomaly Detection via Learning to Complete Video Events. In ACM MM. 583--591."},{"key":"e_1_3_2_2_28_1","unstructured":"Yitian Yuan Lin Ma Jingwen Wang Wei Liu and Wenwu Zhu. 2019 a. Semantic Conditioned Dynamic Modulation for Temporal Sentence Grounding in Videos. In NIPS. 534--544. Yitian Yuan Lin Ma Jingwen Wang Wei Liu and Wenwu Zhu. 2019 a. Semantic Conditioned Dynamic Modulation for Temporal Sentence Grounding in Videos. In NIPS. 534--544."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Hao Zhang Aixin Sun Wei Jing Guoshun Nan Liangli Zhen Joey Tianyi Zhou and Rick Siow Mong Goh. 2021 a. Video Corpus Moment Retrieval with Contrastive Learning. In SIGIR. 685--695. Hao Zhang Aixin Sun Wei Jing Guoshun Nan Liangli Zhen Joey Tianyi Zhou and Rick Siow Mong Goh. 2021 a. Video Corpus Moment Retrieval with Contrastive Learning. In SIGIR. 685--695.","DOI":"10.1145\/3404835.3462874"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Hao Zhang Aixin Sun Wei Jing Liangli Zhen Joey Tianyi Zhou and Rick Siow Mong Goh. 2021 b. Parallel Attention Network with Sequence Matching for Video Grounding. In Findings of ACL\/IJCNLP. 776--790. Hao Zhang Aixin Sun Wei Jing Liangli Zhen Joey Tianyi Zhou and Rick Siow Mong Goh. 2021 b. Parallel Attention Network with Sequence Matching for Video Grounding. In Findings of ACL\/IJCNLP. 776--790.","DOI":"10.18653\/v1\/2021.findings-acl.69"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Hao Zhang Aixin Sun Wei Jing and Joey Tianyi Zhou. 2020. Span-based Localizing Network for Natural Language Video Localization. In ACL Dan Jurafsky Joyce Chai Natalie Schluter and Joel R. Tetreault (Eds.). 6543--6554. Hao Zhang Aixin Sun Wei Jing and Joey Tianyi Zhou. 2020. Span-based Localizing Network for Natural Language Video Localization. In ACL Dan Jurafsky Joyce Chai Natalie Schluter and Joel R. Tetreault (Eds.). 6543--6554.","DOI":"10.18653\/v1\/2020.acl-main.585"}],"event":{"name":"ICMR '22: International Conference on Multimedia Retrieval","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Newark NJ USA","acronym":"ICMR '22"},"container-title":["Proceedings of the 2022 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3512527.3531396","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3512527.3531396","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:30:12Z","timestamp":1750188612000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3512527.3531396"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,6,27]]},"references-count":32,"alternative-id":["10.1145\/3512527.3531396","10.1145\/3512527"],"URL":"https:\/\/doi.org\/10.1145\/3512527.3531396","relation":{},"subject":[],"published":{"date-parts":[[2022,6,27]]},"assertion":[{"value":"2022-06-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}