{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T07:46:29Z","timestamp":1767339989973,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key Research and Development Program of China under Grant","award":["2020YFB1406604"],"award-info":[{"award-number":["2020YFB1406604"]}]},{"name":"Youth Innovation Promotion Association of Chinese Academy of Sciences under Grant","award":["2020108"],"award-info":[{"award-number":["2020108"]}]},{"name":"National Nature Science Foundation of China","award":["62322211,U21B2024,61931008,62071415,12273076,12133001,11873066"],"award-info":[{"award-number":["62322211,U21B2024,61931008,62071415,12273076,12133001,11873066"]}]},{"name":"Pioneer, Zhejiang Provincial Natural Science Foundation of China","award":["LDT23F01011F01,LDT23F01015F01,LDT23F01014F01"],"award-info":[{"award-number":["LDT23F01011F01,LDT23F01015F01,LDT23F01014F01"]}]},{"name":"Leading Goose R&D Program of Zhejiang Province","award":["2022C01068"],"award-info":[{"award-number":["2022C01068"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612384","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:26:54Z","timestamp":1698391614000},"page":"538-546","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Dynamic Contrastive Learning with Pseudo-samples Intervention for Weakly Supervised Joint Video MR and HD"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7649-9827","authenticated-orcid":false,"given":"Shuhan","family":"Kong","sequence":"first","affiliation":[{"name":"Shandong University, Weihai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1943-8219","authenticated-orcid":false,"given":"Liang","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5030-0632","authenticated-orcid":false,"given":"Beichen","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of the Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2243-489X","authenticated-orcid":false,"given":"Wenyu","family":"Wang","sequence":"additional","affiliation":[{"name":"Shandong University, Weihai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2897-5745","authenticated-orcid":false,"given":"Bin","family":"Jiang","sequence":"additional","affiliation":[{"name":"Shandong University, Weihai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1204-0512","authenticated-orcid":false,"given":"Chenggang","family":"Yan","sequence":"additional","affiliation":[{"name":"Hangzhou Dianzi University &amp; Lishui Institute of Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3205-7354","authenticated-orcid":false,"given":"Changhao","family":"Xu","sequence":"additional","affiliation":[{"name":"Shandong University, Weihai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT. 4171--4186.","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT. 4171--4186."},{"key":"e_1_3_2_2_2_1","volume-title":"Russell","author":"Escorcia Victor","year":"2019","unstructured":"Victor Escorcia, Mattia Soldan, Josef Sivic, Bernard Ghanem, and Bryan C. Russell. 2019. Temporal Localization of Moments in Video Collections with Natural Language. CoRR abs\/1907.12763 (2019)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Christoph Feichtenhofer Haoqi Fan Jitendra Malik and Kaiming He. 2019. Slow- Fast Networks for Video Recognition. In ICCV. 6201--6210.","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_2_4_1","volume-title":"TALL: Temporal Activity Localization via Language Query. In ICCV. 5277--5285.","author":"Gao Jiyang","year":"2017","unstructured":"Jiyang Gao, Chen Sun, Zhenheng Yang, and Ram Nevatia. 2017. TALL: Temporal Activity Localization via Language Query. In ICCV. 5277--5285."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Junyu Gao and Changsheng Xu. 2021. Fast Video Moment Retrieval. In ICCV. 1503--1512.","DOI":"10.1109\/ICCV48922.2021.00155"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Jort F. Gemmeke Daniel P. W. Ellis Dylan Freedman Aren Jansen Wade Lawrence R. Channing Moore Manoj Plakal and Marvin Ritter. 2017. Au- dio Set: An ontology and human-labeled dataset for audio events. In ICASSP. 776--780.","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"Michael Gygli Yale Song and Liangliang Cao. 2016. Video2GIF: Automatic Generation of Animated GIFs from Video. In CVPR. 1001--1009.","DOI":"10.1109\/CVPR.2016.114"},{"key":"e_1_3_2_2_8_1","volume-title":"Russell","author":"Hendricks Lisa Anne","year":"2017","unstructured":"Lisa Anne Hendricks, Oliver Wang, Eli Shechtman, Josef Sivic, Trevor Darrell, and Bryan C. Russell. 2017. Localizing Moments in Video with Natural Language. In ICCV. 5804--5813."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Jiabo Huang Yang Liu Shaogang Gong and Hailin Jin. 2021. Cross-Sentence Temporal and Semantic Relations in Video Activity Localisation. In ICCV. 7179--7188.","DOI":"10.1109\/ICCV48922.2021.00711"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2815998"},{"key":"e_1_3_2_2_11_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2015","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Opti- mization. In ICLR."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"crossref","unstructured":"Haofei Kuang Yi Zhu Zhi Zhang Xinyu Li Joseph Tighe S\u00f6ren Schwertfeger Cyrill Stachniss and Mu Li. 2021. Video Contrastive Learning with Global Context. In ICCVW. 3188.","DOI":"10.1109\/ICCVW54120.2021.00358"},{"key":"e_1_3_2_2_14_1","volume-title":"QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries. CoRR abs\/2107.09609","author":"Lei Jie","year":"2021","unstructured":"Jie Lei, Tamara L. Berg, and Mohit Bansal. 2021. QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries. CoRR abs\/2107.09609 (2021)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3158546"},{"key":"e_1_3_2_2_16_1","first-page":"413","article-title":"EclipSE: Efficient Long-Range Video Retrieval Using Sight and Sound","volume":"13694","author":"Lin Yan-Bo","year":"2022","unstructured":"Yan-Bo Lin, Jie Lei, Mohit Bansal, and Gedas Bertasius. 2022. EclipSE: Efficient Long-Range Video Retrieval Using Sight and Sound. In ECCV, Vol. 13694. 413--430.","journal-title":"ECCV"},{"key":"e_1_3_2_2_17_1","volume-title":"Mo Yu, Bing Xiang, Bowen Zhou, and Yoshua Bengio.","author":"Lin Zhouhan","year":"2017","unstructured":"Zhouhan Lin, Minwei Feng, C\u00edcero Nogueira dos Santos, Mo Yu, Bing Xiang, Bowen Zhou, and Yoshua Bengio. 2017. A Structured Self-Attentive Sentence Embedding. In ICLR."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Zhijie Lin Zhou Zhao Zhu Zhang Qi Wang and Huasheng Liu. 2020. Weakly- Supervised Video Moment Retrieval via Semantic Completion Network. In AAAI. 11539--11546.","DOI":"10.1609\/aaai.v34i07.6820"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Wu Liu Tao Mei Yongdong Zhang Cherry Che and Jiebo Luo. 2015. Multi- task deep visual-semantic embedding for video thumbnail selection. In CVPR. 3707--3715.","DOI":"10.1109\/CVPR.2015.7298994"},{"key":"e_1_3_2_2_20_1","volume-title":"Entity-enhanced adaptive reconstruction network for weakly supervised referring expression grounding. TPAMI","author":"Liu Xuejing","year":"2022","unstructured":"Xuejing Liu, Liang Li, Shuhui Wang, Zheng-Jun Zha, Zechao Li, Qi Tian, and Qingming Huang. 2022. Entity-enhanced adaptive reconstruction network for weakly supervised referring expression grounding. TPAMI (2022), 3003--3018."},{"key":"e_1_3_2_2_21_1","volume-title":"Ying Shan, and Xiaohu Qie.","author":"Liu Ye","year":"2022","unstructured":"Ye Liu, Siyuan Li, Yang Wu, Chang Wen Chen, Ying Shan, and Xiaohu Qie. 2022. UMT: Unified Multi-modal Transformers for Joint Video Moment Retrieval and Highlight Detection. In CVPR. 3032--3041."},{"key":"e_1_3_2_2_22_1","volume-title":"Video Moment Retrieval with Text Query Considering Many-to-Many Correspondence Using Potentially Relevant Pair. CoRR abs\/2106.13566","author":"Maeoki Sho","year":"2021","unstructured":"Sho Maeoki, Yusuke Mukuta, and Tatsuya Harada. 2021. Video Moment Retrieval with Text Query Considering Many-to-Many Correspondence Using Potentially Relevant Pair. CoRR abs\/2106.13566 (2021)."},{"key":"e_1_3_2_2_23_1","volume-title":"Roy-Chowdhury","author":"Mithun Niluthpol Chowdhury","year":"2019","unstructured":"Niluthpol Chowdhury Mithun, Sujoy Paul, and Amit K. Roy-Chowdhury. 2019. Weakly Supervised Video Moment Retrieval From Text Queries. In CVPR. 11592-- 11601."},{"key":"e_1_3_2_2_24_1","volume-title":"Multi-Scale Self-Contrastive Learning with Hard Negative Mining for Weakly-Supervised Query-based Video Grounding. CoRR abs\/2203.03838","author":"Mo Shentong","year":"2022","unstructured":"Shentong Mo, Daizong Liu, and Wei Hu. 2022. Multi-Scale Self-Contrastive Learning with Hard Negative Mining for Weakly-Supervised Query-based Video Grounding. CoRR abs\/2203.03838 (2022)."},{"key":"e_1_3_2_2_25_1","volume-title":"Query-Dependent Video Representation for Moment Retrieval and High- light Detection. CoRR abs\/2303.13874","author":"Moon WonJun","year":"2023","unstructured":"WonJun Moon, Sangeek Hyun, Sanguk Park, Dongchan Park, and Jae-Pil Heo. 2023. Query-Dependent Video Representation for Moment Retrieval and High- light Detection. CoRR abs\/2303.13874 (2023)."},{"key":"e_1_3_2_2_26_1","unstructured":"Arsha Nagrani Shan Yang Anurag Arnab Aren Jansen Cordelia Schmid and Chen Sun. 2021. Attention Bottlenecks for Multimodal Fusion. In NeurIPS. 14200--14213."},{"key":"e_1_3_2_2_27_1","volume-title":"Manning","author":"Pennington Jeffrey","year":"2014","unstructured":"Jeffrey Pennington, Richard Socher, and Christopher D. Manning. 2014. Glove: Global Vectors for Word Representation. In EMNLP. 1532--1543."},{"key":"e_1_3_2_2_28_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. CoRR abs\/2103.00020 (2021)."},{"key":"e_1_3_2_2_29_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In ICLR Yoshua Bengio and Yann LeCun (Eds.)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Yale Song Miriam Redi Jordi Vallmitjana and Alejandro Jaimes. 2016. To Click or Not To Click: Automatic Selection of Beautiful Thumbnails from Videos. In CIKM. 659--668.","DOI":"10.1145\/2983323.2983349"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Xin Sun Xuan Wang Jialin Gao Qiong Liu and Xi Zhou. 2022. You Need to Read Again: Multi-granularity Perception Network for Moment Retrieval in Videos. In SIGIR. 1022--1032.","DOI":"10.1145\/3477495.3532083"},{"key":"e_1_3_2_2_32_1","volume-title":"Plummer","author":"Tan Reuben","year":"2021","unstructured":"Reuben Tan, Huijuan Xu, Kate Saenko, and Bryan A. Plummer. 2021. LoGAN: Latent Graph Co-Attention Network for Weakly-Supervised Video Moment Retrieval. In WACV. 2082--2091."},{"key":"e_1_3_2_2_33_1","first-page":"63","article-title":"Semantic relation-aware difference representation learning for change captioning. In Findings of the association for computational linguistics","volume":"2021","author":"Tu Yunbin","year":"2021","unstructured":"Yunbin Tu, Tingting Yao, Liang Li, Jiedong Lou, Shengxiang Gao, Zhengtao Yu, and Chenggang Yan. 2021. Semantic relation-aware difference representation learning for change captioning. In Findings of the association for computational linguistics: ACL-IJCNLP 2021. 63--73.","journal-title":"ACL-IJCNLP"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109204"},{"key":"e_1_3_2_2_35_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In NIPS. 5998--6008."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","unstructured":"Hao Wang Zheng-Jun Zha Xuejin Chen Zhiwei Xiong and Jiebo Luo. 2020. Dual Path Interaction Network for Video Moment Localization. In ACM MM. 4116--4124.","DOI":"10.1145\/3394171.3413975"},{"key":"e_1_3_2_2_37_1","volume-title":"Semantic and relation modulation for audio-visual event localization. TPAMI","author":"Wang Hao","year":"2022","unstructured":"Hao Wang, Zheng-Jun Zha, Liang Li, Xuejin Chen, and Jiebo Luo. 2022. Semantic and relation modulation for audio-visual event localization. TPAMI (2022)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3290083"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00695"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Zheng Wang Jingjing Chen and Yu-Gang Jiang. 2021. Visual Co-Occurrence Alignment Learning for Weakly-Supervised Video Moment Retrieval. In ACM MM. 1459--1468.","DOI":"10.1145\/3474085.3475278"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"crossref","unstructured":"Zheng Wang Jingjing Chen and Yu-Gang Jiang. 2021. Visual co-occurrence alignment learning for weakly-supervised video moment retrieval. In ACM MM. 1459--1468.","DOI":"10.1145\/3474085.3475278"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Zeyu Xiong Daizong Liu and Pan Zhou. 2022. Gaussian Kernel-Based Cross Modal Network for Spatio-Temporal Video Grounding. In ICIP. 2481--2485.","DOI":"10.1109\/ICIP46576.2022.9897707"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00265"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3058614"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3058614"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Ting Yao Tao Mei and Yong Rui. 2016. Highlight Detection with Pairwise Deep Ranking for First-Person Video Summarization. In CVPR. 982--990.","DOI":"10.1109\/CVPR.2016.112"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Yitian Yuan Lin Ma and Wenwu Zhu. 2019. Sentence Specified Dynamic Video Thumbnail Generation. In ACM MM. 2332--2340.","DOI":"10.1145\/3343031.3350985"},{"key":"e_1_3_2_2_48_1","volume-title":"Davis","author":"Zhang Da","year":"2019","unstructured":"Da Zhang, Xiyang Dai, Xin Wang, Yuan-Fang Wang, and Larry S. Davis. 2019. MAN: Moment Alignment Network for Natural Language Moment Retrieval via Iterative Graph Adjustment. In CVPR. 1247--1257."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"crossref","unstructured":"Songyang Zhang Houwen Peng Jianlong Fu and Jiebo Luo. 2020. Learning 2D Temporal Adjacent Networks for Moment Localization with Natural Language. In AAAI. 12870--12877.","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"crossref","unstructured":"Zhu Zhang Zhijie Lin Zhou Zhao Jieming Zhu and Xiuqiang He. 2020. Regular- ized Two-Branch Proposal Networks for Weakly-Supervised Moment Retrieval in Videos. In ACM MM. 4098--4106.","DOI":"10.1145\/3394171.3413967"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"crossref","unstructured":"Minghang Zheng Yanjie Huang Qingchao Chen and Yang Liu. 2022. Weakly Supervised Video Moment Localization with Contrastive Negative Sample Mining. In AAAI. 3517--3525.","DOI":"10.1609\/aaai.v36i3.20263"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"crossref","unstructured":"Minghang Zheng Yanjie Huang Qingchao Chen Yuxin Peng and Yang Liu. 2022. Weakly Supervised Temporal Sentence Grounding with Gaussian-based Contrastive Proposal Learning. In CVPR. 15534--15543.","DOI":"10.1109\/CVPR52688.2022.01511"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612384","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612384","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:55:56Z","timestamp":1755820556000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612384"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":52,"alternative-id":["10.1145\/3581783.3612384","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612384","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}