{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:44:57Z","timestamp":1772325897188,"version":"3.50.1"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"7-8","license":[{"start":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T00:00:00Z","timestamp":1737331200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T00:00:00Z","timestamp":1737331200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100004543","name":"China Scholarship Council","doi-asserted-by":"publisher","award":["202206130025"],"award-info":[{"award-number":["202206130025"]}],"id":[{"id":"10.13039\/501100004543","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Postgraduate Scientific Research Innovation Project of Hunan Province","award":["QL20220096"],"award-info":[{"award-number":["QL20220096"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072169"],"award-info":[{"award-number":["62072169"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004735","name":"Natural Science Foundation of Hunan Province","doi-asserted-by":"publisher","award":["2021JJ30152"],"award-info":[{"award-number":["2021JJ30152"]}],"id":[{"id":"10.13039\/501100004735","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s13042-024-02520-w","type":"journal-article","created":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T14:30:03Z","timestamp":1737383403000},"page":"4509-4524","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Multi-proposal collaboration and multi-task training for weakly-supervised video moment retrieval"],"prefix":"10.1007","volume":"16","author":[{"given":"Bolin","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chao","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takahiro","family":"Komamizu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ichiro","family":"Ide","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,20]]},"reference":[{"key":"2520_CR1","first-page":"1","volume":"2000","author":"RT Collins","year":"2000","unstructured":"Collins RT, Lipton AJ, Kanade T, Fujiyoshi H, Duggins D, Tsin Y, Tolliver D, Enomoto N, Hasegawa O, Burt P, Wixson L (2000) A system for video surveillance and monitoring. VSAM Final Rep 2000:1\u201368","journal-title":"VSAM Final Rep"},{"key":"2520_CR2","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s13042-023-02011-4","volume":"15","author":"Q He","year":"2024","unstructured":"He Q, Shi R, Chen L, Huo L (2024) Video anomaly detection based on multi-scale optical flow spatio-temporal enhancement and normality mining. Int J Mach Learn Cybern 15:1\u201316","journal-title":"Int J Mach Learn Cybern"},{"key":"2520_CR3","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1109\/MRA.2007.339604","volume":"14","author":"CC Kemp","year":"2007","unstructured":"Kemp CC, Edsinger A, Torres-Jara E (2007) Challenges for robot manipulation in human environments [grand challenges of robotics]. IEEE Robot Autom Mag 14:20\u201329","journal-title":"IEEE Robot Autom Mag"},{"key":"2520_CR4","doi-asserted-by":"crossref","unstructured":"Anne Hendricks L, Wang O, Shechtman E, Sivic J, Darrell T, Russell B (2017) Localizing moments in video with natural language. In: Proceedings of the 16th IEEE international conference on computer vision. IEEE, Venice, Italy, pp 5803\u20135812","DOI":"10.1109\/ICCV.2017.618"},{"key":"2520_CR5","doi-asserted-by":"crossref","unstructured":"Gao J, Sun C, Yang Z, Nevatia R (2017) Tall: temporal activity localization via language query. In: Proceedings of the 16th IEEE international conference on computer vision. IEEE, Venice, Italy, pp 5267\u20135275","DOI":"10.1109\/ICCV.2017.563"},{"key":"2520_CR6","doi-asserted-by":"crossref","unstructured":"Zhang B, Jiang B, Yang C, Pang L ( 2022) Dual-channel localization networks for moment retrieval with natural language. In: Proceedings of the 2022 international conference on multimedia retrieval. ACM, Newark, NJ, USA, pp 351\u2013359","DOI":"10.1145\/3512527.3531394"},{"key":"2520_CR7","doi-asserted-by":"crossref","unstructured":"Zhang B, Yang C, Jiang B, Zhou X (2022) Video moment retrieval with hierarchical contrastive learning. In: Proceedings of the 30th ACM international conference on multimedia. ACM, Lisbon, Portugal, pp 346\u2013355","DOI":"10.1145\/3503161.3547963"},{"key":"2520_CR8","doi-asserted-by":"crossref","unstructured":"Liu M, Wang X, Nie L, He X, Chen B, Chua T-S (2018) Attentive moment retrieval in videos. In: The 41st international ACM SIGIR conference on research & development in information retrieval. ACM, Ann Arbor, MI, USA, pp 15\u201324","DOI":"10.1145\/3209978.3210003"},{"key":"2520_CR9","doi-asserted-by":"publisher","first-page":"3921","DOI":"10.1109\/TMM.2022.3168424","volume":"25","author":"Y Wang","year":"2022","unstructured":"Wang Y, Liu M, Wei Y, Cheng Z, Wang Y, Nie L (2022) Siamese alignment network for weakly supervised video moment retrieval. IEEE Trans Multimed 25:3921\u20133933","journal-title":"IEEE Trans Multimed"},{"key":"2520_CR10","doi-asserted-by":"crossref","unstructured":"Yoon S, Koo G, Kim D, Yoo CD (2023) Scanet: scene complexity aware network for weakly-supervised video moment retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision. IEEE\/CVF, Paris, France, pp 13576\u201313586","DOI":"10.1109\/ICCV51070.2023.01249"},{"key":"2520_CR11","doi-asserted-by":"crossref","unstructured":"Huang Y, Yang L, Sato Y (2023) Weakly supervised temporal sentence grounding with uncertainty-guided self-training. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. IEEE\/CVF, Vancouver, Canada, pp 18908\u201318918","DOI":"10.1109\/CVPR52729.2023.01813"},{"key":"2520_CR12","doi-asserted-by":"crossref","unstructured":"Lv Z, Su B, Wen J-R (2023) Counterfactual cross-modality reasoning for weakly supervised video moment localization. In: Proceedings of the 31st ACM international conference on multimedia. ACM, Ottawa, Canada, pp 6539\u20136547","DOI":"10.1145\/3581783.3612495"},{"key":"2520_CR13","doi-asserted-by":"crossref","unstructured":"Mithun NC, Paul S, Roy-Chowdhury AK (2019) Weakly supervised video moment retrieval from text queries. In: Proceedings of the 2019 IEEE\/CVF conference on computer vision and pattern recognition. IEEE\/CVF, Long Beach, CA, USA, pp 11592\u201311601","DOI":"10.1109\/CVPR.2019.01186"},{"key":"2520_CR14","doi-asserted-by":"crossref","unstructured":"Tan R, Xu H, Saenko K, Plummer BA (2021) Logan: latent graph co-attention network for weakly-supervised video moment retrieval. In: Proceedings of the 2021 IEEE\/CVF winter conference on applications of computer vision. IEEE\/CVF, Waikoloa, HI, USA, pp 2083\u20132092","DOI":"10.1109\/WACV48630.2021.00213"},{"key":"2520_CR15","doi-asserted-by":"crossref","unstructured":"Huang J, Liu Y, Gong S, Jin H (2021) Cross-sentence temporal and semantic relations in video activity localisation. In: Proceedings of the 18th IEEE\/CVF international conference on computer vision. IEEE\/CVF, Montreal, Canada, pp 7199\u20137208","DOI":"10.1109\/ICCV48922.2021.00711"},{"key":"2520_CR16","doi-asserted-by":"publisher","first-page":"3252","DOI":"10.1109\/TIP.2021.3058614","volume":"30","author":"W Yang","year":"2021","unstructured":"Yang W, Zhang T, Zhang Y, Wu F (2021) Local correspondence network for weakly supervised temporal sentence grounding. IEEE Trans Image Process 30:3252\u20133262","journal-title":"IEEE Trans Image Process"},{"key":"2520_CR17","first-page":"1","volume":"31","author":"X Duan","year":"2018","unstructured":"Duan X, Huang W, Gan C, Wang J, Zhu W, Huang J (2018) Weakly supervised dense event captioning in videos. Adv Neural Inf Process Syst 31:1\u201311","journal-title":"Adv Neural Inf Process Syst"},{"key":"2520_CR18","doi-asserted-by":"crossref","unstructured":"Lin Z, Zhao Z, Zhang Z, Wang Q, Liu H (2020) Weakly-supervised video moment retrieval via semantic completion network. In: Proceedings of the AAAI conference on artificial intelligence. AAAI Press, New York, NY, USA, pp 11539\u201311546","DOI":"10.1609\/aaai.v34i07.6820"},{"key":"2520_CR19","doi-asserted-by":"crossref","unstructured":"Chen S, Jiang Y-G (2021) Towards bridging event captioner and sentence localizer for weakly supervised dense event captioning. In: Proceedings of the 2021 IEEE\/CVF conference on computer vision and pattern recognition. IEEE\/CVF, Nashville, TN, USA, pp 8425\u20138435","DOI":"10.1109\/CVPR46437.2021.00832"},{"key":"2520_CR20","doi-asserted-by":"crossref","unstructured":"Zheng M, Huang Y, Chen Q, Liu Y (2022) Weakly supervised video moment localization with contrastive negative sample mining. In: Proceedings of the AAAI conference on artificial intelligence. AAAI Press, Palo Alto, CA, USA, pp 3517\u20133525","DOI":"10.1609\/aaai.v36i3.20263"},{"key":"2520_CR21","doi-asserted-by":"publisher","first-page":"10443","DOI":"10.1109\/TPAMI.2023.3258628","volume":"45","author":"H Zhang","year":"2023","unstructured":"Zhang H, Sun A, Jing W, Zhou JT (2023) Temporal sentence grounding in videos: a survey and future directions. IEEE Trans Pattern Anal Mach Intell 45:10443\u201310465","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"2520_CR22","first-page":"1","volume":"55","author":"M Liu","year":"2023","unstructured":"Liu M, Nie L, Wang Y, Wang M, Rui Y (2023) A survey on video moment localization. ACM Comput Surv 55:1\u201337","journal-title":"ACM Comput Surv"},{"key":"2520_CR23","doi-asserted-by":"crossref","unstructured":"Gao M, Davis LS, Socher R, Xiong C (2019) Wslln: weakly supervised natural language localization networks. Computing Research Repository arXiv Preprint, arXiv:1909.00239","DOI":"10.18653\/v1\/D19-1157"},{"key":"2520_CR24","doi-asserted-by":"crossref","unstructured":"Ma M, Yoon S, Kim J, Lee Y, Kang S, Yoo CD (2020) VLANet: Video-language alignment network for weakly-supervised video moment retrieval. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXVIII. Springer, Glasgow, UK, pp 156\u2013171","DOI":"10.1007\/978-3-030-58604-1_10"},{"key":"2520_CR25","doi-asserted-by":"crossref","unstructured":"Chen Z, Ma L, Luo W, Tang P, Wong K-YK (2020) Look closer to ground better: weakly-supervised temporal grounding of sentence in video. Computing Research Repository arXiv Preprint, arXiv:2001.09308","DOI":"10.18653\/v1\/P19-1183"},{"key":"2520_CR26","doi-asserted-by":"publisher","first-page":"3276","DOI":"10.1109\/TMM.2021.3096087","volume":"24","author":"Y Wang","year":"2022","unstructured":"Wang Y, Deng J, Zhou W, Li H (2022) Weakly supervised temporal adjacent network for language grounding. IEEE Trans Multimed 24:3276\u20133286","journal-title":"IEEE Trans Multimed"},{"key":"2520_CR27","unstructured":"Song Y, Wang J, Ma L, Yu Z, Yu J (2020) Weakly-supervised multi-level attentional reconstruction network for grounding textual queries in videos. Computing Research Repository arXiv Preprint, arXiv:2003.07048"},{"key":"2520_CR28","doi-asserted-by":"crossref","unstructured":"Zhang Z, Lin Z, Zhao Z, Zhu J, He X (2020) Regularized two-branch proposal networks for weakly-supervised moment retrieval in videos. In: Proceedings of the 28th ACM international conference on multimedia. ACM, Seattle, WA, USA, pp 4098\u20134106","DOI":"10.1145\/3394171.3413967"},{"key":"2520_CR29","doi-asserted-by":"crossref","unstructured":"Nam J, Ahn D, Kang D, Ha SJ, Choi J (2021) Zero-shot natural language video localization. In: Proceedings of the 18th IEEE\/CVF international conference on computer vision. IEEE\/CVF, Montreal, Canada, pp 1470\u20131479","DOI":"10.1109\/ICCV48922.2021.00150"},{"key":"2520_CR30","doi-asserted-by":"publisher","first-page":"1646","DOI":"10.1109\/TCSVT.2021.3075470","volume":"32","author":"J Gao","year":"2021","unstructured":"Gao J, Xu C (2021) Learning video moment retrieval without a single annotated video. IEEE Trans Circuits Syst Video Technol 32:1646\u20131657","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"2520_CR31","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, Manning CD (2014) Glove: global vectors for word representation. In: Proceedings of the 2014 conference on empirical methods in natural language processing. Association for Computational Linguistics, Doha, Qatar, pp 1532\u20131543","DOI":"10.3115\/v1\/D14-1162"},{"key":"2520_CR32","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, Krueger G, Sutskever I (2021) Learning transferable visual models from natural language supervision. In: Proceedings of the 38th international conference on machine learning. pp 8748\u20138763"},{"key":"2520_CR33","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A ( 2017) Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the 2017 IEEE conference on computer vision and pattern recognition. IEEE, Honolulu, HI, USA, pp 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"2520_CR34","first-page":"1","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30:1\u201311","journal-title":"Adv Neural Inf Process Syst"},{"key":"2520_CR35","unstructured":"Bahdanau D, Cho KH, Bengio Y (2015) Neural machine translation by jointly learning to align and translate. In: Proceedings of the 3rd international conference on learning representations. ICLR, San Diego, CA, USA, pp 1\u201310"},{"key":"2520_CR36","doi-asserted-by":"publisher","unstructured":"Zhou Z-H (2009) Ensemble learning. Encycl Biom 270\u2013273. https:\/\/doi.org\/10.1007\/978-0-387-73003-5_293","DOI":"10.1007\/978-0-387-73003-5_293"},{"key":"2520_CR37","doi-asserted-by":"crossref","unstructured":"Zheng M, Huang Y, Chen Q, Peng Y, Liu Y (2022) Weakly supervised temporal sentence grounding with gaussian-based contrastive proposal learning. In: Proceedings of the 2022 IEEE\/CVF conference on computer vision and pattern recognition. pp 15555\u201315564","DOI":"10.1109\/CVPR52688.2022.01511"},{"key":"2520_CR38","doi-asserted-by":"crossref","unstructured":"Krishna R, Hata K, Ren F, Fei-Fei L, Carlos Niebles J (2017) Dense-captioning events in videos. In: Proceedings of the 16th IEEE international conference on computer vision. IEEE, Venice, Italy, pp 706\u2013715","DOI":"10.1109\/ICCV.2017.83"},{"key":"2520_CR39","unstructured":"Kingma DP, Ba J (2014) Adam: a method for stochastic optimization. Computing Research Repository arXiv Preprint, arXiv:1412.6980"},{"key":"2520_CR40","doi-asserted-by":"crossref","unstructured":"Wu H, Lyu Y, Shen X, Zhao X, Wang M, Zhang X, Luo Z (2023) Atomic-action-based contrastive network for weakly supervised temporal language grounding. In: 2023 IEEE international conference on multimedia and expo (ICME). pp 1523\u20131528","DOI":"10.1109\/ICME55011.2023.00263"},{"key":"2520_CR41","doi-asserted-by":"publisher","first-page":"126625","DOI":"10.1016\/j.neucom.2023.126625","volume":"554","author":"Y Song","year":"2023","unstructured":"Song Y, Wang J, Ma L, Yu J, Liang J, Yuan L, Yu Z (2023) MARN: multi-level attentional reconstruction networks for weakly supervised video temporal grounding. Neurocomputing 554:126625","journal-title":"Neurocomputing"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02520-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-024-02520-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-024-02520-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T04:20:10Z","timestamp":1757132410000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-024-02520-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,20]]},"references-count":41,"journal-issue":{"issue":"7-8","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["2520"],"URL":"https:\/\/doi.org\/10.1007\/s13042-024-02520-w","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,20]]},"assertion":[{"value":"14 December 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 December 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and informed consent for data used"}}]}}