{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T14:48:22Z","timestamp":1786114102310,"version":"build-2736575974"},"reference-count":46,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100015401","name":"Key Research and Development Projects of Shaanxi Province","doi-asserted-by":"publisher","award":["2024PT-ZCK-91"],"award-info":[{"award-number":["2024PT-ZCK-91"]}],"id":[{"id":"10.13039\/501100015401","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.knosys.2026.116581","type":"journal-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T15:06:26Z","timestamp":1784214386000},"page":"116581","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["A cognitive dual-process framework for Referring Video Object Segmentation"],"prefix":"10.1016","volume":"351","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4274-6190","authenticated-orcid":false,"given":"Yong","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7478-9074","authenticated-orcid":false,"given":"Xiaotong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaobo","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianyao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoming","family":"Xi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116581_b1","doi-asserted-by":"crossref","first-page":"7099","DOI":"10.1109\/TPAMI.2022.3225573","article-title":"A survey on deep learning technique for video segmentation","volume":"45","author":"Zhou","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.knosys.2026.116581_b2","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113006","article-title":"Expression prompt collaboration transformer for universal referring video object segmentation","volume":"311","author":"Chen","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116581_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113786","article-title":"Co-saliency guided multi-modal learning for referring video object segmentation","volume":"324","author":"Tong","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116581_b4","doi-asserted-by":"crossref","unstructured":"S. Seo, J.Y. Lee, B. Han, Urvos: Unified referring video object segmentation network with a large-scale benchmark, in: Proceedings of the European Conference on Computer Vision, 2020, pp. 208\u2013223.","DOI":"10.1007\/978-3-030-58555-6_13"},{"key":"10.1016\/j.knosys.2026.116581_b5","doi-asserted-by":"crossref","unstructured":"A. Botach, E. Zheltonozhskii, C. Baskin, End-to-end referring video object segmentation with multimodal transformers, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 4985\u20134995.","DOI":"10.1109\/CVPR52688.2022.00493"},{"key":"10.1016\/j.knosys.2026.116581_b6","doi-asserted-by":"crossref","unstructured":"K. Gavrilyuk, A. Ghodrati, Z. Li, C.G. Snoek, Actor and action video segmentation from a sentence, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 5958\u20135966.","DOI":"10.1109\/CVPR.2018.00624"},{"key":"10.1016\/j.knosys.2026.116581_b7","series-title":"Referdino: Referring video object segmentation with visual grounding foundations","author":"Liang","year":"2025"},{"key":"10.1016\/j.knosys.2026.116581_b8","doi-asserted-by":"crossref","unstructured":"J. Wu, Y. Jiang, P. Sun, Z. Yuan, P. Luo, Language as queries for referring video object segmentation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 4974\u20134984.","DOI":"10.1109\/CVPR52688.2022.00492"},{"key":"10.1016\/j.knosys.2026.116581_b9","series-title":"Grounded sam: Assembling open-world models for diverse visual tasks","author":"Ren","year":"2024"},{"key":"10.1016\/j.knosys.2026.116581_b10","doi-asserted-by":"crossref","unstructured":"S. Liu, Z. Zeng, T. Ren, F. Li, H. Zhang, J. Yang, Q. Jiang, C. Li, J. Yang, H. Su, et al., Grounding dino: Marrying dino with grounded pre-training for open-set object detection, in: Proceedings of the European Conference on Computer Vision, 2024, pp. 38\u201355.","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"10.1016\/j.knosys.2026.116581_b11","doi-asserted-by":"crossref","unstructured":"M. Wang, J. Xing, B. Jiang, J. Chen, J. Mei, X. Zuo, G. Dai, J. Wang, Y. Liu, A multimodal, multi-task adapting framework for video action recognition, in: Proceedings of the AAAI Conference on Artificial Intelligence, 2024, pp. 5517\u20135525.","DOI":"10.1609\/aaai.v38i6.28361"},{"key":"10.1016\/j.knosys.2026.116581_b12","series-title":"Harnessing vision-language pretrained models with temporal-aware adaptation for referring video object segmentation","author":"Zhou","year":"2024"},{"key":"10.1016\/j.knosys.2026.116581_b13","doi-asserted-by":"crossref","unstructured":"X. Lai, Z. Tian, Y. Chen, Y. Li, Y. Yuan, S. Liu, J. Jia, Lisa: Reasoning segmentation via large language model, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 9579\u20139589.","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"10.1016\/j.knosys.2026.116581_b14","doi-asserted-by":"crossref","unstructured":"R. Zheng, L. Qi, X. Chen, Y. Wang, K. Wang, H. Zhao, Villa: Video reasoning segmentation with large language model, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 23667\u201323677.","DOI":"10.1109\/ICCV51701.2025.02197"},{"key":"10.1016\/j.knosys.2026.116581_b15","doi-asserted-by":"crossref","unstructured":"J. Tian, J. Zhang, S. Liu, L. Xu, Z. Huang, G. Huang, Dtos: Dynamic time object sensing with large multimodal model, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 13810\u201313820.","DOI":"10.1109\/CVPR52734.2025.01289"},{"key":"10.1016\/j.knosys.2026.116581_b16","doi-asserted-by":"crossref","unstructured":"Y.C. Chen, W.H. Li, C. Sun, Y.C.F. Wang, C.S. Chen, Sam4mllm: Enhance multi-modal large language model for referring expression segmentation, in: Proceedings of the European Conference on Computer Vision, 2024, pp. 323\u2013340.","DOI":"10.1007\/978-3-031-73004-7_19"},{"key":"10.1016\/j.knosys.2026.116581_b17","doi-asserted-by":"crossref","first-page":"6543","DOI":"10.1109\/TIP.2023.3328485","article-title":"Reformulating graph kernels for self-supervised space-time correspondence learning","volume":"32","author":"Qin","year":"2023","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.knosys.2026.116581_b18","doi-asserted-by":"crossref","first-page":"223","DOI":"10.1177\/1745691612460685","article-title":"Dual-process theories of higher cognition: Advancing the debate","volume":"8","author":"Evans","year":"2013","journal-title":"Perspect. Psychol. Sci."},{"key":"10.1016\/j.knosys.2026.116581_b19","doi-asserted-by":"crossref","first-page":"1192","DOI":"10.1109\/JAS.2023.123456","article-title":"Coarse-to-fine video instance segmentation with factorized conditional appearance flows","volume":"10","author":"Qin","year":"2023","journal-title":"IEEE\/CAA J. Autom. Sin."},{"key":"10.1016\/j.knosys.2026.116581_b20","doi-asserted-by":"crossref","unstructured":"L. Ye, M. Rochan, Z. Liu, Y. Wang, Cross-modal self-attention network for referring image segmentation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 10502\u201310511.","DOI":"10.1109\/CVPR.2019.01075"},{"key":"10.1016\/j.knosys.2026.116581_b21","doi-asserted-by":"crossref","first-page":"11373","DOI":"10.1109\/TCSVT.2024.3419119","article-title":"Temporally consistent referring video object segmentation with hybrid memory","volume":"34","author":"Miao","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.knosys.2026.116581_b22","doi-asserted-by":"crossref","unstructured":"Z. Qin, D. Yu, C. Luo, Z. Chen, Sliced wasserstein bridge for open-vocabulary video instance segmentation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025a, pp. 12470\u201312478.","DOI":"10.1109\/ICCV51701.2025.01159"},{"key":"10.1016\/j.knosys.2026.116581_b23","first-page":"143","article-title":"Two-and three-dimensional electron microscopy techniques: powerful tools for studying the brain under physiological and pathological conditions","volume":"1","author":"Luj\u00e1n","year":"2024","journal-title":"Adv. Technol. Neurosci."},{"key":"10.1016\/j.knosys.2026.116581_b24","first-page":"105","article-title":"Applications of fractal analysis techniques in magnetic resonance imaging and computed tomography for stroke diagnosis and stroke-related brain damage: a narrative review","volume":"1","author":"Maryenko","year":"2024","journal-title":"Adv. Technol. Neurosci."},{"key":"10.1016\/j.knosys.2026.116581_b25","doi-asserted-by":"crossref","unstructured":"R. Wang, X. Wang, T. Feng, X. Gong, G. Li, Y.W. Zhan, Q. Li, W. Zhu, Improving compositional generalization in cross-embodiment learning via mixture of disentangled prototypes, in: Proceedings of the 33rd ACM International Conference on Multimedia, 2025b, pp. 7162\u20137171.","DOI":"10.1145\/3746027.3754499"},{"key":"10.1016\/j.knosys.2026.116581_b26","doi-asserted-by":"crossref","unstructured":"J. Tang, G. Zheng, S. Yang, Temporal collection and distribution for referring video object segmentation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 15466\u201315476.","DOI":"10.1109\/ICCV51070.2023.01418"},{"key":"10.1016\/j.knosys.2026.116581_b27","doi-asserted-by":"crossref","unstructured":"Z. Zhu, X. Feng, D. Chen, J. Yuan, C. Qiao, G. Hua, Exploring pre-trained text-to-video diffusion models for referring video object segmentation, in: Proceedings of the European Conference on Computer Vision, 2024, pp. 452\u2013469.","DOI":"10.1007\/978-3-031-73254-6_26"},{"key":"10.1016\/j.knosys.2026.116581_b28","doi-asserted-by":"crossref","unstructured":"H. Ding, C. Liu, S. He, X. Jiang, C.C. Loy, Mevis: A large-scale benchmark for video segmentation with motion expressions, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 2694\u20132703.","DOI":"10.1109\/ICCV51070.2023.00254"},{"key":"10.1016\/j.knosys.2026.116581_b29","doi-asserted-by":"crossref","unstructured":"Z. Qin, D. Yu, Y. Shi, Q. Wang, Z. Chen, Video instance segmentation by weighted structure inference, in: Proceedings of the 33rd ACM International Conference on Multimedia, 2025b, pp. 7616\u20137624.","DOI":"10.1145\/3746027.3755007"},{"key":"10.1016\/j.knosys.2026.116581_b30","doi-asserted-by":"crossref","unstructured":"R. Wang, H. Sun, Y. Lin, C. Zuo, Y. Gong, Y. Yin, W. Meng, Seqmvrl: A sequential fusion framework for multi-view representation learning, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025a, pp. 25822\u201325831.","DOI":"10.1109\/CVPR52734.2025.02405"},{"key":"10.1016\/j.knosys.2026.116581_b31","doi-asserted-by":"crossref","first-page":"26425","DOI":"10.52202\/075280-1149","article-title":"Soc: Semantic-assisted object cluster for referring video object segmentation","author":"Luo","year":"2023","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116581_b32","doi-asserted-by":"crossref","unstructured":"S. He, H. Ding, Decoupling static and hierarchical motion perception for referring video segmentation, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13332\u201313341.","DOI":"10.1109\/CVPR52733.2024.01266"},{"key":"10.1016\/j.knosys.2026.116581_b33","doi-asserted-by":"crossref","unstructured":"D. Wu, T. Wang, Y. Zhang, X. Zhang, J. Shen, Onlinerefer: A simple online baseline for referring video object segmentation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 2761\u20132770.","DOI":"10.1109\/ICCV51070.2023.00259"},{"key":"10.1016\/j.knosys.2026.116581_b34","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition","article-title":"Spot: Spatiotemporal prompt optimization for motion-stabilized mllm-guided video segmentation","author":"Fan","year":"2026"},{"key":"10.1016\/j.knosys.2026.116581_b35","doi-asserted-by":"crossref","unstructured":"B. Xu, R. Hou, T. Ren, G. Wu, Rgb-d video object segmentation via enhanced multi-store feature memory, in: Proceedings of the 2024 International Conference on Multimedia Retrieval, 2024a, pp. 1016\u20131024.","DOI":"10.1145\/3652583.3658036"},{"key":"10.1016\/j.knosys.2026.116581_b36","article-title":"Hypsam: Hybrid prompt-driven segment anything model for rgb-thermal salient object detection","author":"Hou","year":"2025","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.knosys.2026.116581_b37","doi-asserted-by":"crossref","unstructured":"S. Ding, R. Qian, X. Dong, P. Zhang, Y. Zang, Y. Cao, Y. Guo, D. Lin, J. Wang, Sam2long: Enhancing sam 2 for long video segmentation with a training-free memory tree, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2025, pp. 13614\u201313624.","DOI":"10.1109\/ICCV51701.2025.01264"},{"key":"10.1016\/j.knosys.2026.116581_b38","doi-asserted-by":"crossref","unstructured":"H. Rasheed, M. Maaz, S. Shaji, A. Shaker, S. Khan, H. Cholakkal, R.M. Anwer, E. Xing, M.H. Yang, F.S. Khan, Glamm: Pixel grounding large multimodal model, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13009\u201313018.","DOI":"10.1109\/CVPR52733.2024.01236"},{"key":"10.1016\/j.knosys.2026.116581_b39","series-title":"U-llava: Unifying multi-modal tasks via large language model","author":"Xu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116581_b40","series-title":"Evf-sam: Early vision-language fusion for text-prompted segment anything model","author":"Zhang","year":"2024"},{"key":"10.1016\/j.knosys.2026.116581_b41","series-title":"Mpg-sam 2: Adapting sam 2 with mask priors and global context for referring video object segmentation","author":"Rong","year":"2025"},{"key":"10.1016\/j.knosys.2026.116581_b42","doi-asserted-by":"crossref","unstructured":"C. Cuttano, G. Trivigno, G. Rosi, C. Masone, G. Averta, Samwise: Infusing wisdom in sam2 for text-driven video segmentation, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 3395\u20133405.","DOI":"10.1109\/CVPR52734.2025.00322"},{"key":"10.1016\/j.knosys.2026.116581_b43","doi-asserted-by":"crossref","unstructured":"Z. Ren, O. Gallo, D. Sun, M.H. Yang, E.B. Sudderth, J. Kautz, A fusion approach for multi-frame optical flow estimation, in: Proceedings of the IEEE Winter Conference on Applications of Computer Vision, WACV, 2019.","DOI":"10.1109\/WACV.2019.00225"},{"key":"10.1016\/j.knosys.2026.116581_b44","doi-asserted-by":"crossref","unstructured":"A. Khoreva, A. Rohrbach, B. Schiele, Video object segmentation with language referring expressions, in: Proceedings of the Asian Conference on Computer Vision, 2018, pp. 123\u2013141.","DOI":"10.1007\/978-3-030-20870-7_8"},{"key":"10.1016\/j.knosys.2026.116581_b45","doi-asserted-by":"crossref","unstructured":"H. Jhuang, J. Gall, S. Zuffi, C. Schmid, M.J. Black, Towards understanding action recognition, in: Proceedings of the IEEE International Conference on Computer Vision, 2013, pp. 3192\u20133199.","DOI":"10.1109\/ICCV.2013.396"},{"key":"10.1016\/j.knosys.2026.116581_b46","series-title":"European Conference on Computer Vision","first-page":"402","article-title":"Raft: Recurrent all-pairs field transforms for optical flow","author":"Teed","year":"2020"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126013079?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126013079?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T20:23:20Z","timestamp":1785961400000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126013079"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":46,"alternative-id":["S0950705126013079"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116581","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A cognitive dual-process framework for Referring Video Object Segmentation","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116581","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116581"}}