{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T18:38:09Z","timestamp":1777487889447,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Zhejiang Natural Science Foundation","award":["LR19F020006"],"award-info":[{"award-number":["LR19F020006"]}]},{"name":"the National Key R&D Program of China","award":["2020QNA5024"],"award-info":[{"award-number":["2020QNA5024"]}]},{"name":"National Natural Science Foundation of China","award":["61836002, U1611461, 61751209"],"award-info":[{"award-number":["61836002, U1611461, 61751209"]}]},{"name":"Fundamental Research Funds for the Central Universities","award":["2020QNA5024"],"award-info":[{"award-number":["2020QNA5024"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3413967","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T13:12:00Z","timestamp":1602508320000},"page":"4098-4106","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":67,"title":["Regularized Two-Branch Proposal Networks for Weakly-Supervised Moment Retrieval in Videos"],"prefix":"10.1145","author":[{"given":"Zhu","family":"Zhang","sequence":"first","affiliation":[{"name":"Zhejiang University, HangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhijie","family":"Lin","sequence":"additional","affiliation":[{"name":"Zhejiang University, HangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhou","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang University, HangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jieming","family":"Zhu","sequence":"additional","affiliation":[{"name":"Huawei Noah's Ark Lab, HangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiuqiang","family":"He","sequence":"additional","affiliation":[{"name":"Huawei Noah's Ark Lab, HangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.572"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00124"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1015"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018175"},{"key":"e_1_3_2_2_6_1","volume-title":"Proceedings of the American Association for Artificial Intelligence.","author":"Chen Long","year":"2020","unstructured":"Long Chen , Chujie Lu , Siliang Tang , Jun Xiao , Dong Zhang , Chilie Tan , and Xiaolin Li . 2020 a. Rethinking the Bottom-Up Framework for Query-based Video Localization . In Proceedings of the American Association for Artificial Intelligence. Long Chen, Chujie Lu, Siliang Tang, Jun Xiao, Dong Zhang, Chilie Tan, and Xiaolin Li. 2020 a. Rethinking the Bottom-Up Framework for Query-based Video Localization. In Proceedings of the American Association for Artificial Intelligence."},{"key":"e_1_3_2_2_7_1","volume-title":"2020 b. Look Closer to Ground Better: Weakly-Supervised Temporal Grounding of Sentence in Video. arXiv preprint arXiv:2001.09308","author":"Chen Zhenfang","year":"2020","unstructured":"Zhenfang Chen , Lin Ma , Wenhan Luo , Peng Tang , and Kwan-Yee K Wong . 2020 b. Look Closer to Ground Better: Weakly-Supervised Temporal Grounding of Sentence in Video. arXiv preprint arXiv:2001.09308 ( 2020 ). Zhenfang Chen, Lin Ma, Wenhan Luo, Peng Tang, and Kwan-Yee K Wong. 2020 b. Look Closer to Ground Better: Weakly-Supervised Temporal Grounding of Sentence in Video. arXiv preprint arXiv:2001.09308 (2020)."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1183"},{"key":"e_1_3_2_2_9_1","unstructured":"Junyoung Chung Caglar Gulcehre KyungHyun Cho and Yoshua Bengio. 2014. Empirical evaluation of gated recurrent neural networks on sequence modeling. In Advances in Neural Information Processing Systems.  Junyoung Chung Caglar Gulcehre KyungHyun Cho and Yoshua Bengio. 2014. Empirical evaluation of gated recurrent neural networks on sequence modeling. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_10_1","unstructured":"Xuguang Duan Wenbing Huang Chuang Gan Jingdong Wang Wenwu Zhu and Junzhou Huang. 2018. Weakly supervised dense event captioning in videos. In Advances in Neural Information Processing Systems. 3059--3069.  Xuguang Duan Wenbing Huang Chuang Gan Jingdong Wang Wenwu Zhu and Junzhou Huang. 2018. Weakly supervised dense event captioning in videos. In Advances in Neural Information Processing Systems. 3059--3069."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.5555\/1953048.2021068"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"e_1_3_2_2_13_1","volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing","author":"Gao Mingfei","year":"2019","unstructured":"Mingfei Gao , Larry S Davis , Richard Socher , and Caiming Xiong . 2019 . WSLLN: Weakly Supervised Natural Language Localization Networks . Proceedings of the Conference on Empirical Methods in Natural Language Processing (2019). Mingfei Gao, Larry S Davis, Richard Socher, and Caiming Xiong. 2019. WSLLN: Weakly Supervised Natural Language Localization Networks. Proceedings of the Conference on Empirical Methods in Natural Language Processing (2019)."},{"key":"e_1_3_2_2_14_1","first-page":"8393","article-title":"Read, watch, and move: Reinforcement learning for temporally grounding natural language descriptions in videos","volume":"33","author":"He Dongliang","year":"2019","unstructured":"Dongliang He , Xiang Zhao , Jizhou Huang , Fu Li , Xiao Liu , and Shilei Wen . 2019 . Read, watch, and move: Reinforcement learning for temporally grounding natural language descriptions in videos . In Proceedings of the American Association for Artificial Intelligence , Vol. 33. 8393 -- 8400 . Dongliang He, Xiang Zhao, Jizhou Huang, Fu Li, Xiao Liu, and Shilei Wen. 2019. Read, watch, and move: Reinforcement learning for temporally grounding natural language descriptions in videos. In Proceedings of the American Association for Artificial Intelligence, Vol. 33. 8393--8400.","journal-title":"Proceedings of the American Association for Artificial Intelligence"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1168"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6820"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2965987"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00139"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210003"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240549"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01186"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00706"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_2_26_1","unstructured":"Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. In Advances in Neural Information Processing Systems. 91--99.  Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. In Advances in Neural Information Processing Systems. 91--99."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.155"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01270-0_10"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.119"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_31"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.678"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00042"},{"key":"e_1_3_2_2_34_1","first-page":"7","article-title":"Multilevel Language and Vision Integration for Text-to-Clip Retrieval","volume":"2","author":"Xu Huijuan","year":"2019","unstructured":"Huijuan Xu , Kun He , L Sigal , S Sclaroff , and K Saenko . 2019 . Multilevel Language and Vision Integration for Text-to-Clip Retrieval . In Proceedings of the American Association for Artificial Intelligence , Vol. 2. 7 . Huijuan Xu, Kun He, L Sigal, S Sclaroff, and K Saenko. 2019. Multilevel Language and Vision Integration for Text-to-Clip Retrieval. In Proceedings of the American Association for Artificial Intelligence, Vol. 2. 7.","journal-title":"Proceedings of the American Association for Artificial Intelligence"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00562"},{"key":"e_1_3_2_2_36_1","unstructured":"Yitian Yuan Lin Ma Jingwen Wang Wei Liu and Wenwu Zhu. 2019. Semantic Conditioned Dynamic Modulation for Temporal Sentence Grounding in Videos. In Advances in Neural Information Processing Systems. 534--544.  Yitian Yuan Lin Ma Jingwen Wang Wei Liu and Wenwu Zhu. 2019. Semantic Conditioned Dynamic Modulation for Temporal Sentence Grounding in Videos. In Advances in Neural Information Processing Systems. 534--544."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00719"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00134"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3331184.3331235"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/149"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/610"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01068"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.317"}],"event":{"name":"MM '20: The 28th ACM International Conference on Multimedia","location":"Seattle WA USA","acronym":"MM '20","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413967","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3413967","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T21:32:07Z","timestamp":1750195927000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413967"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":44,"alternative-id":["10.1145\/3394171.3413967","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3413967","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}