{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T23:52:18Z","timestamp":1781826738929,"version":"3.54.5"},"reference-count":67,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T00:00:00Z","timestamp":1775520000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T00:00:00Z","timestamp":1775520000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100004344","name":"Adobe Systems","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100004344","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1007\/s11263-026-02777-4","type":"journal-article","created":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T02:26:43Z","timestamp":1775528803000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Collaborative Temporal Consistency Learning for Point-supervised Natural Language Video Localization"],"prefix":"10.1007","volume":"134","author":[{"given":"Zhuo","family":"Tao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunbin","family":"Tu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng-Jun","family":"Zha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Amin","family":"Beheshti","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingming","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuankai","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4848-2304","authenticated-orcid":false,"given":"Ming-Hsuan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,7]]},"reference":[{"key":"2777_CR1","doi-asserted-by":"crossref","unstructured":"Cao M, Wei F, Xu C, Geng X, Chen L, Zhang C, Zou Y, Shen T, & Jiang D (2023). Iterative proposal refinement for weakly-supervised video grounding. In: CVPR, pp. 6524\u20136534","DOI":"10.1109\/CVPR52729.2023.00631"},{"key":"2777_CR2","doi-asserted-by":"crossref","unstructured":"Carreira J, & Zisserman A (2017). Quo vadis, action recognition? A new model and the kinetics dataset. In: CVPR, pp. 4724\u20134733","DOI":"10.1109\/CVPR.2017.502"},{"key":"2777_CR3","doi-asserted-by":"crossref","unstructured":"Chen H, Wang X, Lan X, Chen H, Duan X, Jia J, & Zhu W (2023). Curriculum-listener: Consistency- and complementarity-aware audio-enhanced temporal sentence grounding. In: MM, pp. 3117\u20133128","DOI":"10.1145\/3581783.3612504"},{"key":"2777_CR4","doi-asserted-by":"crossref","unstructured":"Chen J, Chen X, Ma L, Jie Z, & Chua T (2018). Temporally grounding natural sentence in video. In: EMNLP, pp. 162\u2013171","DOI":"10.18653\/v1\/D18-1015"},{"key":"2777_CR5","doi-asserted-by":"crossref","unstructured":"Chen J, Luo W, Zhang W, & Ma L (2022). Explore inter-contrast between videos via composition for weakly supervised temporal sentence grounding. In: AAAI, pp. 267\u2013275","DOI":"10.1609\/aaai.v36i1.19902"},{"key":"2777_CR6","doi-asserted-by":"crossref","unstructured":"Chen S, & Jiang Y (2019). Semantic proposal for activity localization in videos via sentence query. In: AAAI, pp. 8199\u20138206","DOI":"10.1609\/aaai.v33i01.33018199"},{"key":"2777_CR7","doi-asserted-by":"crossref","unstructured":"Cui R, Qian T, Peng P, Daskalaki E, Chen J, Guo X, Sun H, & Jiang Y (2022). Video moment retrieval from text queries via single frame annotation. In: SIGIR, pp. 1033\u20131043","DOI":"10.1145\/3477495.3532078"},{"key":"2777_CR8","doi-asserted-by":"crossref","unstructured":"Ding X, Wang N, Zhang S, Cheng D, Li X, Huang Z, Tang M, & Gao X (2021). Supp.ort-set based cross-supervision for video grounding. In: ICCV, pp. 11553\u201311562","DOI":"10.1109\/ICCV48922.2021.01137"},{"key":"2777_CR9","doi-asserted-by":"crossref","unstructured":"Gao J, & Xu C (2021). Fast video moment retrieval. In: ICCV, pp. 1503\u20131512","DOI":"10.1109\/ICCV48922.2021.00155"},{"key":"2777_CR10","doi-asserted-by":"crossref","unstructured":"Gao J, Sun C, Yang Z, & Nevatia R (2017). TALL: temporal activity localization via language query. In: ICCV, pp. 5277\u20135285","DOI":"10.1109\/ICCV.2017.563"},{"key":"2777_CR11","doi-asserted-by":"crossref","unstructured":"Ghosh S, Agarwal A, Parekh Z, & Hauptmann AG (2019). Excl: Extractive clip localization using natural language descriptions. In: NAACL, pp. 1984\u20131990","DOI":"10.18653\/v1\/N19-1198"},{"key":"2777_CR12","doi-asserted-by":"crossref","unstructured":"Guo Y, Siddiqui F, Zhao Y, Chellappa R, & Lo S (2024). Stimuvar: Spatiotemporal stimuli-aware video affective reasoning with multimodal large language models. CoRR arXiv:2409.00304","DOI":"10.1007\/s11263-025-02495-3"},{"key":"2777_CR13","doi-asserted-by":"crossref","unstructured":"He T, Liu H, Ni Z, Li Y, Ma X, Zhong C, Zhang Y, Wang Y, & Lin W (2025). Achieving procedure-aware instructional video correlation learning under weak supervision from a collaborative perspective. IJCV 133(4):2070\u20132095","DOI":"10.1007\/s11263-024-02272-8"},{"issue":"4","key":"2777_CR14","doi-asserted-by":"publisher","first-page":"2070","DOI":"10.1007\/s11263-024-02272-8","volume":"133","author":"T He","year":"2025","unstructured":"He, T., Liu, H., Ni, Z., Li, Y., Ma, X., Zhong, C., Zhang, Y., Wang, Y., & Lin, W. (2025). Achieving procedure-aware instructional video correlation learning under weak supervision from a collaborative perspective. International Journal of Computer Vision, 133(4), 2070\u20132095.","journal-title":"International Journal of Computer Vision"},{"key":"2777_CR15","doi-asserted-by":"crossref","unstructured":"Hendricks LA, Wang O, Shechtman E, Sivic J, Darrell T, & Russell BC (2017). Localizing moments in video with natural language. In: ICCV, pp. 5804\u20135813","DOI":"10.1109\/ICCV.2017.618"},{"key":"2777_CR16","doi-asserted-by":"crossref","unstructured":"Huang B, Wang X, Chen H, Song Z, & Zhu W (2024). Vtimellm: Empower llm to grasp video moments. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 14271\u201314280","DOI":"10.1109\/CVPR52733.2024.01353"},{"key":"2777_CR17","doi-asserted-by":"crossref","unstructured":"Huang J, Liu Y, Gong S, & Jin H (2021). Cross-sentence temporal and semantic relations in video activity localisation. In: ICCV, pp. 7179\u20137188","DOI":"10.1109\/ICCV48922.2021.00711"},{"key":"2777_CR18","doi-asserted-by":"crossref","unstructured":"Jiang X, Zhou Z, Xu X, Yang Y, Wang G, & Shen HT (2023). Faster video moment retrieval with point-level supervision. In: MM, pp. 1334\u20131342","DOI":"10.1145\/3581783.3612394"},{"key":"2777_CR19","doi-asserted-by":"crossref","unstructured":"Kim S, Cho J, Yu J, Yoo Y, & Choi JY (2024). Gaussian mixture proposals with pull-push learning scheme to capture diverse events for weakly supervised temporal video grounding. In: AAAI, pp. 2795\u20132803","DOI":"10.1609\/aaai.v38i3.28059"},{"key":"2777_CR20","doi-asserted-by":"crossref","unstructured":"Krishna R, Hata K, Ren F, Fei-Fei L, & Niebles JC (2017). Dense-captioning events in videos. In: ICCV, pp. 706\u2013715","DOI":"10.1109\/ICCV.2017.83"},{"issue":"2","key":"2777_CR21","first-page":"51:1","volume":"19","author":"X Lan","year":"2023","unstructured":"Lan, X., Yuan, Y., Wang, X., Wang, Z., & Zhu, W. (2023). A survey on temporal sentence grounding in videos. Test of Memory Malingering, 19(2), 51:1-51:33.","journal-title":"Test of Memory Malingering"},{"key":"2777_CR22","doi-asserted-by":"crossref","unstructured":"Lee P, & Byun H (2021). Learning action completeness from points for weakly-supervised temporal action localization. In: ICCV, pp. 13628\u201313637","DOI":"10.1109\/ICCV48922.2021.01339"},{"key":"2777_CR23","unstructured":"Lei J, Berg TL, & Bansal M (2021). Detecting moments and highlights in videos via natural language queries. In: NIPS, pp. 11846\u201311858"},{"issue":"6","key":"2777_CR24","doi-asserted-by":"publisher","first-page":"3667","DOI":"10.1007\/s11263-025-02350-5","volume":"133","author":"J Leng","year":"2025","unstructured":"Leng, J., Kuang, C., Li, S., Gan, J., Chen, H., & Gao, X. (2025). Dual-space video person re-identification. International Journal of Computer Vision, 133(6), 3667\u20133688.","journal-title":"International Journal of Computer Vision"},{"key":"2777_CR25","doi-asserted-by":"crossref","unstructured":"Li H, Shu X, He S, Qiao R, Wen W, Guo T, Gan B, & Sun X (2023a). D3G: exploring gaussian prior for temporal sentence grounding with glance annotation. In: ICCV, pp. 13688\u201313700","DOI":"10.1109\/ICCV51070.2023.01263"},{"key":"2777_CR26","doi-asserted-by":"crossref","unstructured":"Li P, Xie C, Xie H, Zhao L, Zhang L, Zheng Y, Zhao D, & Zhang Y (2023b). Momentdiff: Generative video moment retrieval from random to real. In: NIPS","DOI":"10.52202\/075280-2880"},{"key":"2777_CR27","doi-asserted-by":"crossref","unstructured":"Lin J, Liu Z, Wang W, Wu W, & Wang L (2024). VLG: general video recognition with web textual knowledge. IJCV 132(10):4792\u20134817","DOI":"10.1007\/s11263-024-02081-z"},{"issue":"10","key":"2777_CR28","doi-asserted-by":"publisher","first-page":"4792","DOI":"10.1007\/s11263-024-02081-z","volume":"132","author":"J Lin","year":"2024","unstructured":"Lin, J., Liu, Z., Wang, W., Wu, W., & Wang, L. (2024). VLG: General video recognition with web textual knowledge. International Journal of Computer Vision, 132(10), 4792\u20134817.","journal-title":"International Journal of Computer Vision"},{"key":"2777_CR29","doi-asserted-by":"crossref","unstructured":"Liu M, Wang X, Nie L, He X, Chen B, Chua T (2018). Attentive moment retrieval in videos. In: SIGIR, pp. 15\u201324","DOI":"10.1145\/3209978.3210003"},{"issue":"4","key":"2777_CR30","doi-asserted-by":"publisher","first-page":"1940","DOI":"10.1007\/s11263-024-02279-1","volume":"133","author":"T Liu","year":"2025","unstructured":"Liu, T., Lam, K., & Bao, B. (2025). A memory-assisted knowledge transferring framework with curriculum anticipation for weakly supervised online activity detection. International Journal of Computer Vision, 133(4), 1940\u20131963.","journal-title":"International Journal of Computer Vision"},{"key":"2777_CR31","unstructured":"Loshchilov I, & Hutter F (2019). Decoupled weight decay regularization. In: ICLR"},{"issue":"5","key":"2777_CR32","doi-asserted-by":"publisher","first-page":"1244","DOI":"10.1007\/s11263-022-01600-0","volume":"130","author":"F Ma","year":"2022","unstructured":"Ma, F., Zhu, L., & Yang, Y. (2022). Weakly supervised moment localization with decoupled consistent concept prediction. International Journal of Computer Vision, 130(5), 1244\u20131258.","journal-title":"International Journal of Computer Vision"},{"key":"2777_CR33","doi-asserted-by":"crossref","unstructured":"Mithun NC, Paul S, & Roy-Chowdhury AK (2019). Weakly supervised video moment retrieval from text queries. In: CVPR, pp. 11592\u201311601","DOI":"10.1109\/CVPR.2019.01186"},{"key":"2777_CR34","doi-asserted-by":"crossref","unstructured":"Moon W, Hyun S, Park S, Park D, & Heo J (2023). Query - dependent video representation for moment retrieval and highlight detection. In: CVPR, pp. 23023\u201323033","DOI":"10.1109\/CVPR52729.2023.02205"},{"key":"2777_CR35","doi-asserted-by":"crossref","unstructured":"Mun J, Cho M,& Han B (2020). Local-global video-text interactions for temporal grounding. In: CVPR, pp. 10807\u201310816","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"2777_CR36","first-page":"2538","volume":"30","author":"K Ning","year":"2021","unstructured":"Ning, K., Xie, L., Liu, J., Wu, F., & Tian, Q. (2021). Interaction-integrated network for natural language moment localization. To Insure Promptness, 30, 2538\u20132548.","journal-title":"To Insure Promptness"},{"key":"2777_CR37","unstructured":"van\u00a0den Oord A, Li Y, & Vinyals O (2018). Representation learning with contrastive predictive coding. CoRR arXiv:1807.03748"},{"key":"2777_CR38","doi-asserted-by":"crossref","unstructured":"Otani M, Nakashima Y, Rahtu E, & Heikkil\u00e4 J (2020). Uncovering hidden challenges in query-based video moment retrieval. In: BMVC","DOI":"10.5244\/C.34.84"},{"key":"2777_CR39","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, & Manning CD (2014). Glove: Global vectors for word representation. In: EMNLP, pp. 1532\u20131543","DOI":"10.3115\/v1\/D14-1162"},{"key":"2777_CR40","doi-asserted-by":"crossref","unstructured":"Qu M, Chen X, Liu W, Li A, & Zhao Y (2024). Chatvtg: Video temporal grounding via chat with video dialogue large language models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 1847\u20131856","DOI":"10.1109\/CVPRW63382.2024.00191"},{"key":"2777_CR41","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1162\/tacl_a_00207","volume":"1","author":"M Regneri","year":"2013","unstructured":"Regneri, M., Rohrbach, M., Wetzel, D., Thater, S., Schiele, B., & Pinkal, M. (2013). Grounding action descriptions in videos. Association for Computational Linguistics, 1, 25\u201336.","journal-title":"Association for Computational Linguistics"},{"key":"2777_CR42","doi-asserted-by":"crossref","unstructured":"Regneri M, Rohrbach M, Wetzel D, Thater S, Schiele B, & Pinkal M (2013). Grounding action descriptions in videos. TACL 1:25\u201336","DOI":"10.1162\/tacl_a_00207"},{"key":"2777_CR43","doi-asserted-by":"crossref","unstructured":"Rohrbach M, Regneri M, Andriluka M, Amin S, Pinkal M, & Schiele B (2012). Script data for attribute-based recognition of composite activities. In: ECCV, pp. 144\u2013157","DOI":"10.1007\/978-3-642-33718-5_11"},{"key":"2777_CR44","doi-asserted-by":"crossref","unstructured":"Sigurdsson GA, Varol G, Wang X, Farhadi A, Laptev I, & Gupta A (2016). Hollywood in homes: Crowdsourcing data collection for activity understanding. In: ECCV, pp. 510\u2013526","DOI":"10.1007\/978-3-319-46448-0_31"},{"key":"2777_CR45","first-page":"95","volume":"27","author":"K Tang","year":"2025","unstructured":"Tang, K., He, L., Wang, N., & Gao, X. (2025). Dual semantic reconstruction network for weakly supervised temporal sentence grounding. Test Maturity Model, 27, 95\u2013107.","journal-title":"Test Maturity Model"},{"key":"2777_CR46","doi-asserted-by":"crossref","unstructured":"Tang K, He L, Wang N, & Gao X (2025). Dual semantic reconstruction network for weakly supervised temporal sentence grounding. TMM 27:95\u2013107","DOI":"10.1109\/TMM.2024.3521676"},{"key":"2777_CR47","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev LD, Fergus R, Torresani L, & Paluri M (2015). Learning spatiotemporal features with 3d convolutional networks. In: ICCV, pp. 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"2777_CR48","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, & Polosukhin I (2017). Attention is all you need. In: NIPS, pp. 5998\u20136008"},{"key":"2777_CR49","doi-asserted-by":"crossref","unstructured":"Wang J, Sun A, Zhang H, & Li X (2023). MS-DETR: natural language video localization with sampling moment-moment interaction. In: ACL, pp. 1387\u20131400","DOI":"10.18653\/v1\/2023.acl-long.77"},{"key":"2777_CR50","doi-asserted-by":"crossref","unstructured":"Wang K, Liu H, Jie L, Li Z, Hu Y, & Nie L (2024). Explicit granularity and implicit scale correspondence learning for point-supervised video moment localization. In: MM, pp. 9214\u20139223","DOI":"10.1145\/3664647.3680774"},{"key":"2777_CR51","doi-asserted-by":"crossref","unstructured":"Wang Z, Wang L, Wu T, Li T, & Wu G (2022). Negative sample matters: A renaissance of metric learning for temporal grounding. In: AAAI, pp. 2613\u20132623","DOI":"10.1609\/aaai.v36i3.20163"},{"issue":"7","key":"2777_CR52","doi-asserted-by":"publisher","first-page":"3970","DOI":"10.1007\/s11263-025-02385-8","volume":"133","author":"J Xiao","year":"2025","unstructured":"Xiao, J., Huang, N., Qin, H., Li, D., Li, Y., Zhu, F., Tao, Z., Yu, J., Lin, L., Chua, T., & Yao, A. (2025). Videoqa in the era of llms: An empirical study. International Journal of Computer Vision, 133(7), 3970\u20133993.","journal-title":"International Journal of Computer Vision"},{"key":"2777_CR53","unstructured":"Xu H, He K, Sigal L, Sclaroff S, & Saenko K (2018). Text-to-clip video retrieval with early fusion and re-captioning. CoRR arXiv:1804.05113"},{"key":"2777_CR54","doi-asserted-by":"crossref","unstructured":"Xu H, He K, Plummer BA, Sigal L, Sclaroff S, & Saenko K (2019). Multilevel language and vision integration for text-to-clip retrieval. In: AAAI, pp. 9062\u20139069","DOI":"10.1609\/aaai.v33i01.33019062"},{"key":"2777_CR55","doi-asserted-by":"crossref","unstructured":"Yang J, Wei P, Li H, & Ren Z (2024). Task-driven exploration: Decoupling and inter-task feedback for joint moment retrieval and highlight detection. In: CVPR, pp. 18308\u201318318","DOI":"10.1109\/CVPR52733.2024.01733"},{"key":"2777_CR56","first-page":"3252","volume":"30","author":"W Yang","year":"2021","unstructured":"Yang, W., Zhang, T., Zhang, Y., & Wu, F. (2021). Local correspondence network for weakly supervised temporal sentence grounding. TIP, 30, 3252\u20133262.","journal-title":"TIP"},{"key":"2777_CR57","doi-asserted-by":"crossref","unstructured":"Yuan Y, Mei T, & Zhu W (2019). To find where you talk: Temporal sentence localization in video with attention based location regression. In: AAAI, pp. 9159\u20139166","DOI":"10.1609\/aaai.v33i01.33019159"},{"issue":"5","key":"2777_CR58","first-page":"2725","volume":"44","author":"Y Yuan","year":"2022","unstructured":"Yuan, Y., Ma, L., Wang, J., Liu, W., & Zhu, W. (2022). Semantic conditioned dynamic modulation for temporal sentence grounding in videos. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(5), 2725\u20132741.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2777_CR59","doi-asserted-by":"crossref","unstructured":"Zhang D, Dai X, Wang X, Wang Y, & Davis LS (2019). MAN: moment alignment network for natural language moment retrieval via iterative graph adjustment. In: CVPR, pp. 1247\u20131257","DOI":"10.1109\/CVPR.2019.00134"},{"key":"2777_CR60","doi-asserted-by":"crossref","unstructured":"Zhang H, Sun A, Jing W, & Zhou JT (2020a). Span-based localizing network for natural language video localization. In: ACL, pp. 6543\u20136554","DOI":"10.18653\/v1\/2020.acl-main.585"},{"key":"2777_CR61","doi-asserted-by":"crossref","unstructured":"Zhang H, Sun A, Jing W, & Zhou JT (2023). Temporal sentence grounding in videos: A survey and future directions. TPAMI 45(8):10443\u201310465","DOI":"10.1109\/TPAMI.2023.3258628"},{"key":"2777_CR62","doi-asserted-by":"crossref","unstructured":"Zhang M, Yang Y, Chen X, Ji Y, Xu X, Li J, & Shen HT (2021). Multi-stage aggregated transformer network for temporal language localization in videos. In: CVPR, pp. 12669\u201312678","DOI":"10.1109\/CVPR46437.2021.01248"},{"issue":"8","key":"2777_CR63","doi-asserted-by":"publisher","first-page":"10443","DOI":"10.1109\/TPAMI.2023.3258628","volume":"45","author":"H Zhang","year":"2023","unstructured":"Zhang, H., Sun, A., Jing, W., & Zhou, J. T. (2023). Temporal sentence grounding in videos: A survey and future directions. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(8), 10443\u201310465.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"8","key":"2777_CR64","doi-asserted-by":"publisher","first-page":"3232","DOI":"10.1007\/s11263-024-02024-8","volume":"132","author":"X Zhao","year":"2024","unstructured":"Zhao, X., Chang, S., Pang, Y., Yang, J., Zhang, L., & Lu, H. (2024). Adaptive multi-source predictor for zero-shot video object segmentation. International Journal of Computer Vision, 132(8), 3232\u20133250.","journal-title":"International Journal of Computer Vision"},{"key":"2777_CR65","doi-asserted-by":"crossref","unstructured":"Zheng M, Huang Y, Chen Q, & Liu Y (2022a). Weakly supervised video moment localization with contrastive negative sample mining. In: AAAI, pp. 3517\u20133525","DOI":"10.1609\/aaai.v36i3.20263"},{"key":"2777_CR66","doi-asserted-by":"crossref","unstructured":"Zheng M, Huang Y, Chen Q, Peng Y, & Liu Y (2022b). Weakly supervised temporal sentence grounding with gaussian-based contrastive proposal learning. In: CVPR, pp. 15534\u201315543","DOI":"10.1109\/CVPR52688.2022.01511"},{"issue":"11","key":"2777_CR67","doi-asserted-by":"publisher","first-page":"5308","DOI":"10.1007\/s11263-024-02142-3","volume":"132","author":"J Zhou","year":"2024","unstructured":"Zhou, J., Guo, D., Zhong, Y., & Wang, M. (2024). Advancing weakly-supervised audio-visual video parsing via segment-wise pseudo labeling. International Journal of Computer Vision, 132(11), 5308\u20135329.","journal-title":"International Journal of Computer Vision"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02777-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02777-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02777-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T23:07:01Z","timestamp":1781824021000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02777-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,7]]},"references-count":67,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,5]]}},"alternative-id":["2777"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02777-4","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,7]]},"assertion":[{"value":"30 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"198"}}