{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T05:47:52Z","timestamp":1785908872424,"version":"3.56.0"},"reference-count":39,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272438"],"award-info":[{"award-number":["62272438"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004826","name":"Beijing Natural Science Foundation","doi-asserted-by":"publisher","award":["L25700"],"award-info":[{"award-number":["L25700"]}],"id":[{"id":"10.13039\/501100004826","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.cviu.2026.104830","type":"journal-article","created":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T16:39:48Z","timestamp":1780936788000},"page":"104830","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Synergistic Dual-Graph Co-Evolutionary Network for point-supervised temporal action localization"],"prefix":"10.1016","volume":"270","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5875-5017","authenticated-orcid":false,"given":"Pengfei","family":"Yan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2083-2188","authenticated-orcid":false,"given":"Xu","family":"Cui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9923-5034","authenticated-orcid":false,"given":"Laiyun","family":"Qing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guorong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingming","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104830_b1","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXVIII 16","first-page":"121","article-title":"Boundary content graph neural network for temporal action proposal generation","author":"Bai","year":"2020"},{"key":"10.1016\/j.cviu.2026.104830_b2","doi-asserted-by":"crossref","unstructured":"Caba Heilbron, F., Escorcia, V., Ghanem, B., Carlos Niebles, J., 2015. Activitynet: A large-scale video benchmark for human activity understanding. In: Proceedings of the Ieee Conference on Computer Vision and Pattern Recognition. pp. 961\u2013970.","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"10.1016\/j.cviu.2026.104830_b3","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A., 2017. Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 6299\u20136308.","DOI":"10.1109\/CVPR.2017.502"},{"key":"10.1016\/j.cviu.2026.104830_b4","series-title":"European Conference on Computer Vision","first-page":"192","article-title":"Dual-evidential learning for weakly-supervised temporal action localization","author":"Chen","year":"2022"},{"issue":"000","key":"10.1016\/j.cviu.2026.104830_b5","article-title":"MKP-net: Memory knowledge propagation network for point-supervised temporal action localization in livestreaming","volume":"248","author":"Chen","year":"2024","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104830_b6","first-page":"3","article-title":"You-do, I-Learn: Discovering task relevant objects and their modes of interaction from multi-user egocentric video","volume":"vol. 2","author":"Damen","year":"2014"},{"key":"10.1016\/j.cviu.2026.104830_b7","series-title":"CVPR 2011","first-page":"3281","article-title":"Learning to recognize objects in egocentric activities","author":"Fathi","year":"2011"},{"key":"10.1016\/j.cviu.2026.104830_b8","doi-asserted-by":"crossref","first-page":"7363","DOI":"10.1109\/TIP.2022.3222623","article-title":"Compact representation and reliable classification learning for point-level weakly-supervised action localization","volume":"31","author":"Fu","year":"2022","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104830_b9","doi-asserted-by":"crossref","unstructured":"He, B., Yang, X., Kang, L., Cheng, Z., Zhou, X., Shrivastava, A., 2022. Asm-loc: Action-aware segment modeling for weakly-supervised temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 13925\u201313935.","DOI":"10.1109\/CVPR52688.2022.01355"},{"key":"10.1016\/j.cviu.2026.104830_b10","doi-asserted-by":"crossref","unstructured":"Hu, X., Li, K., Patel, D., Kruus, E., Min, M.R., Ding, Z., 2024. Weakly-Supervised Temporal Action Localization with Multi-Modal Plateau Transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2704\u20132713.","DOI":"10.1109\/CVPRW63382.2024.00276"},{"key":"10.1016\/j.cviu.2026.104830_b11","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.cviu.2016.10.018","article-title":"The thumos challenge on action recognition for videos \u201cin the wild\u201d","volume":"155","author":"Idrees","year":"2017","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104830_b12","doi-asserted-by":"crossref","unstructured":"Ju, C., Zhao, P., Chen, S., Zhang, Y., Wang, Y., Tian, Q., 2021. Divide and conquer for single-frame temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 13455\u201313464.","DOI":"10.1109\/ICCV48922.2021.01320"},{"key":"10.1016\/j.cviu.2026.104830_b13","series-title":"The kinetics human action video dataset","author":"Kay","year":"2017"},{"key":"10.1016\/j.cviu.2026.104830_b14","doi-asserted-by":"crossref","unstructured":"Kim, H.-J., Hong, J.-H., Kong, H., Lee, S.-W., 2024. TE-TAD: Towards Full End-to-End Temporal Action Detection via Time-Aligned Coordinate Expression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 18837\u201318846.","DOI":"10.1109\/CVPR52733.2024.01782"},{"key":"10.1016\/j.cviu.2026.104830_b15","doi-asserted-by":"crossref","unstructured":"Kim, Y.H., Kang, H., Kim, S., 2022. A Sliding Window Scheme for Online Temporal Action Localization. In: European Conference on Computer Vision.","DOI":"10.1007\/978-3-031-19830-4_37"},{"key":"10.1016\/j.cviu.2026.104830_b16","doi-asserted-by":"crossref","unstructured":"Lee, P., Byun, H., 2021. Learning action completeness from points for weakly-supervised temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 13648\u201313657.","DOI":"10.1109\/ICCV48922.2021.01339"},{"key":"10.1016\/j.cviu.2026.104830_b17","first-page":"11320","article-title":"Background suppression network for weakly-supervised temporal action localization","volume":"vol. 34, no. 07","author":"Lee","year":"2020"},{"key":"10.1016\/j.cviu.2026.104830_b18","article-title":"Neighbor-guided pseudo-label generation and refinement for single-frame supervised temporal action localization","author":"Li","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104830_b19","doi-asserted-by":"crossref","unstructured":"Lin, C., Xu, C., Luo, D., Wang, Y., Tai, Y., Wang, C., Li, J., Huang, F., Fu, Y., 2021. Learning salient boundary feature for anchor-free temporal action localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3320\u20133329.","DOI":"10.1109\/CVPR46437.2021.00333"},{"key":"10.1016\/j.cviu.2026.104830_b20","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IV 16","first-page":"420","article-title":"Sf-net: Single-frame supervision for temporal action localization","author":"Ma","year":"2020"},{"key":"10.1016\/j.cviu.2026.104830_b21","doi-asserted-by":"crossref","unstructured":"Nag, S., Zhu, X., Deng, J., Song, Y.-Z., Xiang, T., 2023. Difftad: Temporal action detection with proposal denoising diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 10362\u201310374.","DOI":"10.1109\/ICCV51070.2023.00951"},{"key":"10.1016\/j.cviu.2026.104830_b22","series-title":"International Conference on Machine Learning","first-page":"2014","article-title":"Learning convolutional neural networks for graphs","author":"Niepert","year":"2016"},{"key":"10.1016\/j.cviu.2026.104830_b23","unstructured":"Peng, K., Huang, J., Huang, X., Wen, D., Zheng, J., Chen, Y., Yang, K., Wu, J., Hao, C., Stiefelhagen, R., 2025. HopaDIFF: Holistic-Partial Aware Fourier Conditioned Diffusion for Referring Human Action Segmentation in Multi-Person Scenarios. In: The Thirty-Ninth Annual Conference on Neural Information Processing Systems."},{"key":"10.1016\/j.cviu.2026.104830_b24","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1016\/j.patrec.2025.02.027","article-title":"Summarized knowledge guidance for single-frame temporal action localization","volume":"191","author":"Sheng","year":"2025","journal-title":"Pattern Recognit. Lett."},{"key":"10.1016\/j.cviu.2026.104830_b25","doi-asserted-by":"crossref","unstructured":"Shi, D., Zhong, Y., Cao, Q., Ma, L., Li, J., Tao, D., 2023. Tridet: Temporal action detection with relative boundary modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 18857\u201318866.","DOI":"10.1109\/CVPR52729.2023.01808"},{"key":"10.1016\/j.cviu.2026.104830_b26","doi-asserted-by":"crossref","unstructured":"Tang, X., Fan, J., Luo, C., Zhang, Z., Zhang, M., Yang, Z., 2023. DDG-Net: Discriminability-Driven Graph Network for Weakly-supervised Temporal Action Localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6622\u20136632.","DOI":"10.1109\/ICCV51070.2023.00609"},{"key":"10.1016\/j.cviu.2026.104830_b27","doi-asserted-by":"crossref","unstructured":"Wang, L., Koniusz, P., 2023. 3mformer: Multi-order multi-mode transformer for skeletal action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 5620\u20135631.","DOI":"10.1109\/CVPR52729.2023.00544"},{"key":"10.1016\/j.cviu.2026.104830_b28","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Lin, D., Van Gool, L., 2017. Untrimmednets for weakly supervised action recognition and detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 4325\u20134334.","DOI":"10.1109\/CVPR.2017.678"},{"key":"10.1016\/j.cviu.2026.104830_b29","doi-asserted-by":"crossref","unstructured":"Xia, Z., Cheng, J., Liu, S., Hu, Y., Wang, S., Zhang, Y., Dang, L., 2024. Realigning Confidence with Temporal Saliency Information for Point-Level Weakly-Supervised Temporal Action Localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 18440\u201318450.","DOI":"10.1109\/CVPR52733.2024.01745"},{"key":"10.1016\/j.cviu.2026.104830_b30","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhao, C., Rojas, D.S., Thabet, A., Ghanem, B., 2020. G-tad: Sub-graph localization for temporal action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10156\u201310165.","DOI":"10.1109\/CVPR42600.2020.01017"},{"issue":"12","key":"10.1016\/j.cviu.2026.104830_b31","doi-asserted-by":"crossref","first-page":"9814","DOI":"10.1109\/TPAMI.2021.3132058","article-title":"Background-click supervision for temporal action localization","volume":"44","author":"Yang","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104830_b32","first-page":"3090","article-title":"ACGNet: Action complement graph network for weakly-supervised temporal action localization","volume":"vol. 36, no. 3","author":"Yang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104830_b33","doi-asserted-by":"crossref","unstructured":"Zach, C., Pock, T., Bischof, H., 2007. A Duality Based Approach for Realtime TV-L1 Optical Flow. In: Proceedings of the DAGM Symposium on Pattern Recognition. pp. 214\u2013223.","DOI":"10.1007\/978-3-540-74936-3_22"},{"key":"10.1016\/j.cviu.2026.104830_b34","doi-asserted-by":"crossref","unstructured":"Zeng, R., Huang, W., Tan, M., Rong, Y., Zhao, P., Huang, J., Gan, C., 2019. Graph convolutional networks for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 7094\u20137103.","DOI":"10.1109\/ICCV.2019.00719"},{"key":"10.1016\/j.cviu.2026.104830_b35","doi-asserted-by":"crossref","unstructured":"Zhang, C., Cao, M., Yang, D., Chen, J., Zou, Y., 2021. Cola: Weakly-supervised temporal action localization with snippet contrastive learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16010\u201316019.","DOI":"10.1109\/CVPR46437.2021.01575"},{"key":"10.1016\/j.cviu.2026.104830_b36","first-page":"7115","article-title":"Hr-pro: Point-supervised temporal action localization via hierarchical reliability propagation","volume":"vol. 38, no. 7","author":"Zhang","year":"2024"},{"key":"10.1016\/j.cviu.2026.104830_b37","series-title":"European Conference on Computer Vision","first-page":"492","article-title":"Actionformer: Localizing moments of actions with transformers","author":"Zhang","year":"2022"},{"key":"10.1016\/j.cviu.2026.104830_b38","doi-asserted-by":"crossref","unstructured":"Zhao, C., Thabet, A.K., Ghanem, B., 2021. Video self-stitching graph network for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 13658\u201313667.","DOI":"10.1109\/ICCV48922.2021.01340"},{"key":"10.1016\/j.cviu.2026.104830_b39","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111207","article-title":"Diffusion-based framework for weakly-supervised temporal action localization","volume":"160","author":"Zou","year":"2025","journal-title":"Pattern Recognit."}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001979?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001979?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T05:08:01Z","timestamp":1785906481000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001979"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":39,"alternative-id":["S1077314226001979"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104830","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Synergistic Dual-Graph Co-Evolutionary Network for point-supervised temporal action localization","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104830","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104830"}}