{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T15:17:54Z","timestamp":1759331874394,"version":"3.44.0"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Science and Technology Development Fund(FDCT) of Macau","award":["0071\/2022\/A","0071\/2022\/A"],"award-info":[{"award-number":["0071\/2022\/A","0071\/2022\/A"]}]},{"name":"Natural Science Research Project of Anhui Educational Committee- Key Project","award":["2024AH051339"],"award-info":[{"award-number":["2024AH051339"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s00530-025-01825-2","type":"journal-article","created":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T05:45:42Z","timestamp":1746078342000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Causalseg: investigating causality modeling for semi-supervised video object segmentation"],"prefix":"10.1007","volume":"31","author":[{"given":"Zhengjin","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nannan","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenmin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huiwen","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,1]]},"reference":[{"key":"1825_CR1","unstructured":"Ping, Hu, Fabian, Caba, Oliver, Wang, Zhe, Lin, Stan, Sclaroff, Federico, Perazzi: \u201cTemporally distributed networks for fast video semantic segmentation.\u201d Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 8818-8827 (2020)"},{"key":"1825_CR2","doi-asserted-by":"crossref","unstructured":"Ruoxi, Deng, Liu, Shengjun: \u201cDeep structural contour detection.\u201d Proceedings of the 28th ACM international conference on multimedia. 304-312 (2020)","DOI":"10.1145\/3394171.3413750"},{"issue":"4","key":"1825_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3391743","volume":"11","author":"Y Rui","year":"2020","unstructured":"Rui, Y., Guosheng, Yo Lin, Shixiong, X., Jiaqi, Z., Yong, Z.: Video object segmentation and tracking: a survey. ACM Trans. Intell. Syst. Technol. (TIST) 11(4), 1\u201347 (2020)","journal-title":"ACM Trans. Intell. Syst. Technol. (TIST)"},{"key":"1825_CR4","unstructured":"Xiankai, Lu, Wenguan, Wang, Jianbing, Shen, Yu-Wing, Tai, Crandall, David J., Hoi Steven, C.H.: \u201cLearning video object segmentation from unlabeled videos.\u201d Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition 8960-8970 (2020)"},{"key":"1825_CR5","unstructured":"Wug, Oh Seoung, Joon-Young, Lee, Ning, Xu, Joo, Kim Seon: \u201cVideo object segmentation using space-time memory networks.\u201d In Proceedings of the IEEE\/CVF international conference on computer vision 9226\u20139235 (2019)"},{"key":"1825_CR6","first-page":"3430","volume":"33","author":"L Yongqing","year":"2020","unstructured":"Yongqing, L., Xin, L., Navid, J., Jim, C.: Video object segmentation with adaptive feature bank and uncertain-region refinement. Adv. Neural Inform. Proc. Syst. 33, 3430\u20133441 (2020)","journal-title":"Adv. Neural Inform. Proc. Syst."},{"key":"1825_CR7","first-page":"11781","volume":"34","author":"CH Kei","year":"2021","unstructured":"Kei, C.H., Tai, Y.W., Tang, C.K.: Rethinking space-time networks with improved memory coverage for efficient video object segmentation. Adv. Neural Inform. Proc. Syst. 34, 11781\u201311794 (2021)","journal-title":"Adv. Neural Inform. Proc. Syst."},{"key":"1825_CR8","doi-asserted-by":"crossref","unstructured":"Xiaohao, Xu., Jinglu, Wang, Xiao, Li., Yan, Lu.: Reliable propagation-correction modulation for video object segmentation. Proceedings of the AAAI Conference on Artificial Intelligence 36(3), 2946\u20132954 (2022)","DOI":"10.1609\/aaai.v36i3.20200"},{"key":"1825_CR9","doi-asserted-by":"publisher","first-page":"109399","DOI":"10.1016\/j.patcog.2023.109399","volume":"138","author":"S Jiadai","year":"2023","unstructured":"Jiadai, S., Sun, M.Y., Yuchao, D., Yiran, Z., Jianyuan, W.: Munet: motion uncertainty-aware semi-supervised video object segmentation. Pattern Recogn. 138, 109399 (2023)","journal-title":"Pattern Recogn."},{"key":"1825_CR10","doi-asserted-by":"publisher","first-page":"103843","DOI":"10.1016\/j.cviu.2023.103843","volume":"237","author":"C Yadang","year":"2023","unstructured":"Yadang, C., Chuanjun, J., Zhi-Xin, Y., Enhua, W.: Spatial constraint for efficient semi-supervised video object segmentation. Comput. Vision Image Understanding 237, 103843 (2023)","journal-title":"Comput. Vision Image Understanding"},{"key":"1825_CR11","doi-asserted-by":"crossref","unstructured":"Suhwan, Cho, Heansung, Lee, Minhyeok, Lee, Chaewon, Park, Sungjun, Jang, Minjung, Kim, Sangyoun, Lee: \u201cTackling background distraction in video object segmentation.\u201d European Conference on Computer Vision 446-462 (2022)","DOI":"10.1007\/978-3-031-20047-2_26"},{"key":"1825_CR12","doi-asserted-by":"crossref","unstructured":"Tan, Wang, Chang, Zhou, Qianru, Sun, Hanwang, Zhang: \u201cCausal attention for unbiased visual recognition.\u201d Proceedings of the IEEE\/CVF International Conference on Computer Vision 3091-3100 (2021)","DOI":"10.1109\/ICCV48922.2021.00308"},{"key":"1825_CR13","doi-asserted-by":"crossref","unstructured":"Yuan, Liu, Jingyuan, Chen, Zhenfang, Chen, Bing, Deng, Jianqiang, Huang, Hanwang, Zhang: \u201cThe blessings of unlabeled background in untrimmed videos.\u201d Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 6176-6185 (2021)","DOI":"10.1109\/CVPR46437.2021.00611"},{"key":"1825_CR14","doi-asserted-by":"crossref","unstructured":"Wangbo, Zhao, Jing, Zhang, Long, Li, Nick, Barnes, Nian, Liu, Junwei, Han: \u201cWeakly supervised video salient object detection.\u201d Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition 16826-16835 (2021)","DOI":"10.1109\/CVPR46437.2021.01655"},{"key":"1825_CR15","doi-asserted-by":"crossref","unstructured":"Junsuk, Choe, Shim, Hyunjung: \u201cAttention-based dropout layer for weakly supervised object localization.\u201d Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 2219-2228 (2019)","DOI":"10.1109\/CVPR.2019.00232"},{"issue":"3","key":"1825_CR16","first-page":"3933","volume":"45","author":"W Wei","year":"2022","unstructured":"Wei, W., Junyu, G., Changsheng, X.: Weakly-supervised video object grounding via causal intervention. IEEE Trans. Pattern Anal. Mach. Intell. 45(3), 3933\u20133948 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1825_CR17","first-page":"3","volume-title":"Models, reasoning and inference","author":"Judea Pearl","year":"2000","unstructured":"Pearl, Judea: Models, reasoning and inference, vol. 19, p. 3. Cambridge University Press, Cambridge, UK (2000)"},{"key":"1825_CR18","first-page":"655","volume":"33","author":"Z Dong","year":"2020","unstructured":"Dong, Z., Hanwang, Z., Jinhui, T., XianSheng, H., Qianru, S.: Causal intervention for weakly-supervised semantic segmentation. Adv. Neural Inform. Proc. Syst. 33, 655\u2013666 (2020)","journal-title":"Adv. Neural Inform. Proc. Syst."},{"key":"1825_CR19","first-page":"2734","volume":"33","author":"Yue Zhongqi","year":"2020","unstructured":"Zhongqi, Yue, Hanwang, Zhang, Qianru, Sun, Xian-Sheng, Hua: Interventional few-shot learning. Advances in neural information processing systems 33, 2734\u20132746 (2020)","journal-title":"Advances in neural information processing systems"},{"key":"1825_CR20","first-page":"1513","volume":"33","author":"T Kaihua","year":"2020","unstructured":"Kaihua, T., Jianqiang, H., Hanwang, Z.: Long-tailed classification by keeping the good and removing the bad momentum causal effect. Adv. Neural Inform. Proc. Syst. 33, 1513\u20131524 (2020)","journal-title":"Adv. Neural Inform. Proc. Syst."},{"key":"1825_CR21","unstructured":"Kaihua, Tang, Mingyuan, Tao, Hanwang, Zhang: \u201cAdversarial visual robustness by causal intervention.\u201d (2021). arXiv preprint arXiv:2106.09534"},{"key":"1825_CR22","doi-asserted-by":"publisher","first-page":"1033","DOI":"10.1109\/TMM.2021.3136717","volume":"25","author":"W Qin","year":"2021","unstructured":"Qin, W., Zhang, H., Hong, R., Lim, E.P., Sun, Q.: Causal interventional training for image recognition. IEEE Trans. Multimedia 25, 1033\u20131044 (2021)","journal-title":"IEEE Trans. Multimedia"},{"key":"1825_CR23","doi-asserted-by":"crossref","unstructured":"Kaihua, Tang, Yulei, Niu, Jianqiang, Huang, Jiaxin, Shi, Hanwang, Zhang: \u201cUnbiased scene graph generation from biased training.\u201d in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition 3716-3725 (2020)","DOI":"10.1109\/CVPR42600.2020.00377"},{"key":"1825_CR24","doi-asserted-by":"crossref","unstructured":"Bing, Tian, Yixin, Cao, Yong, Zhang, Chunxiao, Xing: Debiasing NLU models via causal intervention and counterfactual reasoning. Proceedings of the AAAI Conference on Artificial Intelligence. 36(10), 11376\u201311384 (2022)","DOI":"10.1609\/aaai.v36i10.21389"},{"key":"1825_CR25","doi-asserted-by":"crossref","unstructured":"Chen, Yuedong, Xu, Yang, Tat-Jen, Cham, Jianfei, Cai: \u201cTowards unbiased visual emotion recognition via causal intervention.\u201d In Proceedings of the 30th ACM International Conference on Multimedia 60-69 (2022)","DOI":"10.1145\/3503161.3547936"},{"key":"1825_CR26","doi-asserted-by":"crossref","unstructured":"Xu, Yang, Hanwang, Zhang, Guojun, Qi, Jianfei, Cai: \u201cCausal attention for vision-language tasks.\u201d In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition 9847-9857 (2021)","DOI":"10.1109\/CVPR46437.2021.00972"},{"issue":"1","key":"1825_CR27","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1007\/s44267-023-00018-7","volume":"1","author":"M Siwei","year":"2023","unstructured":"Siwei, M., Gao, J., Wang, R., Chang, J., Mao, Q., Huang, Zhimeng, Jia, Chuanmin: Overview of intelligent video coding: from model-based to learning-based approaches. Visual Intell. 1(1), 15 (2023)","journal-title":"Visual Intell."},{"issue":"1","key":"1825_CR28","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s44267-023-00034-7","volume":"2","author":"Q Rui","year":"2024","unstructured":"Rui, Q., Lin, W., See, J., Li, D.: Controllable augmentations for video representation learning. Visual Intell. 2(1), 1 (2024)","journal-title":"Visual Intell."},{"key":"1825_CR29","doi-asserted-by":"crossref","unstructured":"Junyu, Gao, Chen, Mengyuan, Xu, Changsheng: \u201cCollecting cross-modal presence-absence evidence for weakly-supervised audio-visual event perception.\u201d In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 18827-18836 (2023)","DOI":"10.1109\/CVPR52729.2023.01805"},{"issue":"6","key":"1825_CR30","doi-asserted-by":"publisher","first-page":"1515","DOI":"10.1109\/TPAMI.2018.2838670","volume":"41","author":"K-K Maninis","year":"2018","unstructured":"Maninis, K.-K., Caelles, S., et al.: Video object segmentation without temporal information. IEEE Trans. Pattern Anal. Mach. Intell. 41(6), 1515\u20131530 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1825_CR31","unstructured":"Kaiming, He, Georgia, Gkioxari, Piotr, Doll\u00e1r, Ross, Girshick: \u201cMask r-cnn.\u201d Proceedings of the IEEE international conference on computer vision 2961-2969 (2017)"},{"key":"1825_CR32","doi-asserted-by":"crossref","unstructured":"Sergi, Caelles, Kevis-Kokitsi, Maninis, Jordi, Pont-Tuset, Laura, Leal-Taix\u00e9, Daniel, Cremers, Luc, Van Gool: \u201cOne-shot video object segmentation.\u201d Proceedings of the IEEE conference on computer vision and pattern recognition 221-230 (2017)","DOI":"10.1109\/CVPR.2017.565"},{"issue":"12","key":"1825_CR33","doi-asserted-by":"publisher","first-page":"3035","DOI":"10.1007\/s11263-022-01678-6","volume":"130","author":"G Bras\u00f3","year":"2022","unstructured":"Bras\u00f3, G., Cetintas, O., Leal-Taix\u00e9, L.: Multi-object tracking and segmentation via neural message passing. Int. J. Comput. Vision 130(12), 3035\u20133053 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"1825_CR34","doi-asserted-by":"crossref","unstructured":"Zhong, Shan, Li, Guoqiang, Ying, Wenhao, Zhao, Fuzhou, Xie, Gengsheng, Gong, Shengrong: \u201cEfficient semi-Sspervised object segmentation for long-term videos using adaptive memory network.\u201d IEEE Transactions on Cognitive and Developmental Systems (2024)","DOI":"10.1109\/TCDS.2024.3385849"},{"key":"1825_CR35","doi-asserted-by":"crossref","unstructured":"Miao, Bo, Bennamoun, Mohammed, Gao, Yongsheng, Mian, Ajmal: \u201cRegion aware video object segmentation with deep motion modeling.\u201d IEEE Transactions on Image Processing (2024)","DOI":"10.1109\/TIP.2024.3381445"},{"issue":"20","key":"1825_CR36","doi-asserted-by":"publisher","first-page":"23426","DOI":"10.1007\/s10489-023-04617-1","volume":"53","author":"G Songbo","year":"2023","unstructured":"Songbo, G., Ma, J., Hui, G., Xiao, Q., Shi, W.: STMT: spatio-temporal memory transformer for multi-object tracking. Appl. Intell. 53(20), 23426\u201323441 (2023)","journal-title":"Appl. Intell."},{"key":"1825_CR37","doi-asserted-by":"publisher","first-page":"112075","DOI":"10.1016\/j.knosys.2024.112075","volume":"299","author":"G Songbo","year":"2024","unstructured":"Songbo, G., Miaohui, Z., Qiyang, X., Wentao, S.: Cascaded matching based on detection box area for multi-object tracking. Knowledge-Based Syst. 299, 112075 (2024)","journal-title":"Knowledge-Based Syst."},{"key":"1825_CR38","doi-asserted-by":"crossref","unstructured":"Xun, Yang, Fuli, Feng, Wei, Ji, Meng, Wang, Tat-Seng, Chua: \u201cDeconfounded video moment retrieval with causal intervention.\u201d In Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval 1-10 (2021)","DOI":"10.1145\/3404835.3462823"},{"key":"1825_CR39","doi-asserted-by":"crossref","unstructured":"Bolei, Zhou, Aditya, Khosla, Agata, Lapedriza, Aude, Oliva, Torralba, Antonio: \u201cLearning deep features for discriminative localization.\u201d In Proceedings of the IEEE conference on computer vision and pattern recognition. (2016)","DOI":"10.1109\/CVPR.2016.319"},{"key":"1825_CR40","unstructured":"Philipp, Kr\u00e4henb\u00fchl, Koltun, Vladlen: \u201cEfficient inference in fully connected crfs with gaussian edge potentials.\u201d Advances in Neural Information Processing Systems 24 (2011)"},{"key":"1825_CR41","doi-asserted-by":"crossref","unstructured":"Jun, Wei, Qin, Wang, Zhen, Li, Sheng, Wang, Kevin, Zhou S., Shuguang, Cui: \u201cShallow feature matters for weakly supervised object localization.\u201d In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition 5993-6001 (2021)","DOI":"10.1109\/CVPR46437.2021.00593"},{"key":"1825_CR42","doi-asserted-by":"crossref","unstructured":"Kei, Cheng Ho, Schwing, Alexander G.: \u201cXmem: Long-term video object segmentation with an atkinson-shiffrin memory model.\u201d In European Conference on Computer Vision, Cham: Springer Nature Switzerland 640-658 (2022)","DOI":"10.1007\/978-3-031-19815-1_37"},{"key":"1825_CR43","unstructured":"Pont-Tuset, Jordi, Perazzi, Federico, Caelles, Sergi, Arbel\u00e1ez, Pablo, Sorkine-Hornung, Alexander, Van Gool, Luc: \u201cThe 2017 davis challenge on video object segmentation\u201d. In arXiv:1704.00675, (2017)"},{"key":"1825_CR44","doi-asserted-by":"crossref","unstructured":"Wang, Ziqin, Xu, Jun, Liu, Li, Zhu, Fan, Shao, Ling: \u201cRANet: ranking attention network for fast video object segmentation.\u201d 3978-3987, (2019)","DOI":"10.1109\/ICCV.2019.00408"},{"key":"1825_CR45","doi-asserted-by":"crossref","unstructured":"Oh, Seoung Wug, Lee, Joon-Young, Sunkavalli, Kalyan, Kim, Seon Joo: \u201cFast video object segmentation by reference-guided mask propagation\u201d. In Proceedings of the IEEE conference on computer vision and pattern recognition, pages 7376\u20137385, (2018)","DOI":"10.1109\/CVPR.2018.00770"},{"key":"1825_CR46","unstructured":"Ning, Xu, Linjie, Yang, Yuchen, Fan, Jianchao, Yang, Dingcheng, Yue Yuchen, Liang, Brian, Price, Scott, Cohen, Thomas, Huang: \u201cYoutube-vos: Sequence-to-sequence video object segmentation.\u201d In Proceedings of the European conference on computer vision (ECCV). 585-601 (2018)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01825-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01825-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01825-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,4]],"date-time":"2025-09-04T15:03:51Z","timestamp":1756998231000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01825-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,1]]},"references-count":46,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["1825"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01825-2","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2025,5,1]]},"assertion":[{"value":"20 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"220"}}