{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T19:10:17Z","timestamp":1772651417463,"version":"3.50.1"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T00:00:00Z","timestamp":1772582400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T00:00:00Z","timestamp":1772582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62473341"],"award-info":[{"award-number":["62473341"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Joint Fund Key Project of Science and Technology R&D Plan of Henan Province","award":["235200810022"],"award-info":[{"award-number":["235200810022"]}]},{"name":"the Distinguished Youth Science Foundation of Henan province of China","award":["242300421055"],"award-info":[{"award-number":["242300421055"]}]},{"DOI":"10.13039\/501100006407","name":"Natural Science Foundation of Henan","doi-asserted-by":"crossref","award":["252300420389"],"award-info":[{"award-number":["252300420389"]}],"id":[{"id":"10.13039\/501100006407","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-026-08377-w","type":"journal-article","created":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T17:17:20Z","timestamp":1772644640000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Memory optimization network for enhanced vision-language target tracking"],"prefix":"10.1007","volume":"82","author":[{"given":"Jianwei","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wendi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huanlong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shoukang","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhoujingzi","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,4]]},"reference":[{"issue":"4","key":"8377_CR1","doi-asserted-by":"publisher","first-page":"1268","DOI":"10.1109\/TCSVT.2019.2944654","volume":"31","author":"X Lu","year":"2019","unstructured":"Lu X, Ma C, Ni B, Yang X (2019) Adaptive region proposal with channel regularization for robust object tracking. IEEE Trans Circuits Syst Video Technol 31(4):1268\u20131282","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"8377_CR2","doi-asserted-by":"crossref","unstructured":"Fu Z, Liu Q, Fu Z, Wang Y (2021) Stmtrack: Template-free visual tracking with space-time memory networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 13774\u201313783","DOI":"10.1109\/CVPR46437.2021.01356"},{"key":"8377_CR3","first-page":"146","volume-title":"Eur Conf Comput Vis","author":"S Gao","year":"2022","unstructured":"Gao S, Zhou C, Ma C, Wang X, Yuan J (2022) Aiatrack: attention in attention for transformer visual tracking. Eur Conf Comput Vis. Springer, Cham, pp 146\u2013164"},{"key":"8377_CR4","doi-asserted-by":"crossref","unstructured":"Yan B, Peng H, Fu J, Wang D, Lu H (2021) Learning spatio-temporal transformer for visual tracking. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 10448\u201310457","DOI":"10.1109\/ICCV48922.2021.01028"},{"key":"8377_CR5","first-page":"4838","volume":"38","author":"L Shi","year":"2024","unstructured":"Shi L, Zhong B, Liang Q, Li N, Zhang S, Li X (2024) Explicit visual prompts for visual object tracking. Proc AAAI Conf Artif Intell 38:4838\u20134846","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"8377_CR6","first-page":"7588","volume":"38","author":"Y Zheng","year":"2024","unstructured":"Zheng Y, Zhong B, Liang Q, Mo Z, Zhang S, Li X (2024) Odtrack: Online dense temporal token learning for visual tracking. Proc AAAI Conf Artif Intell 38:7588\u20137596","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"8377_CR7","doi-asserted-by":"crossref","unstructured":"Wei X, Bai Y, Zheng Y, Shi D, Gong Y (2023) bAutoregressive visual tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 9697\u20139706","DOI":"10.1109\/CVPR52729.2023.00935"},{"key":"8377_CR8","first-page":"341","volume-title":"Eur Conf Comput Vis","author":"B Ye","year":"2022","unstructured":"Ye B, Chang H, Ma B, Shan S, Chen X (2022) Joint feature learning and relation modeling for tracking: A one-stream framework. Eur Conf Comput Vis. Springer, Cham, pp 341\u2013357"},{"key":"8377_CR9","first-page":"375","volume-title":"Eur Conf Comput Vis","author":"B Chen","year":"2022","unstructured":"Chen B, Li P, Bai L, Qiao L, Shen Q, Li B, Gan W, Wu W, Ouyang W (2022) Backbone is all your need: A simplified architecture for visual object tracking. Eur Conf Comput Vis. Springer, Cham, pp 375\u2013392"},{"key":"8377_CR10","doi-asserted-by":"crossref","unstructured":"Cui Y, Jiang C, Wang L, Wu G (2022) Mixformer: End-to-end tracking with iterative mixed attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13608\u201313618","DOI":"10.1109\/CVPR52688.2022.01324"},{"key":"8377_CR11","first-page":"16743","volume":"35","author":"L Lin","year":"2022","unstructured":"Lin L, Fan H, Zhang Z, Xu Y, Ling H (2022) Swintrack: A simple and strong baseline for transformer tracking. Adv Neural Inf Process Syst 35:16743\u201316754","journal-title":"Adv Neural Inf Process Syst"},{"key":"8377_CR12","doi-asserted-by":"crossref","unstructured":"Chen X, Yan B, Zhu J, Wang D, Yang X, Lu H (2021) Transformer tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8126\u20138135","DOI":"10.1109\/CVPR46437.2021.00803"},{"key":"8377_CR13","doi-asserted-by":"crossref","unstructured":"Zhang C, Sun X, Yang Y, Liu L, Liu Q, Zhou X, Wang Y (2023) All in one: Exploring unified vision-language tracking with multi-modal alignment. In: Proceedings of the 31st ACM International Conference on Multimedia, 5552\u20135561","DOI":"10.1145\/3581783.3611803"},{"key":"8377_CR14","doi-asserted-by":"publisher","first-page":"1720","DOI":"10.1109\/TMM.2023.3285441","volume":"26","author":"H Zhang","year":"2023","unstructured":"Zhang H, Wang J, Zhang J, Zhang T, Zhong B (2023) One-stream vision-language memory network for object tracking. IEEE Trans Multimedia 26:1720\u20131730","journal-title":"IEEE Trans Multimedia"},{"key":"8377_CR15","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, et al (2021) Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PmLR"},{"key":"8377_CR16","doi-asserted-by":"crossref","unstructured":"Feng Q, Ablavsky V, Bai Q, Sclaroff S (2021) Siamese natural language tracker: Tracking by natural language descriptions with siamese trackers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5851\u20135860","DOI":"10.1109\/CVPR46437.2021.00579"},{"issue":"9","key":"8377_CR17","doi-asserted-by":"publisher","first-page":"4529","DOI":"10.1109\/TCSVT.2023.3288353","volume":"33","author":"R Wang","year":"2023","unstructured":"Wang R, Tang Z, Zhou Q, Liu X, Hui T, Tan Q, Liu S (2023) Unified transformer with isomorphic branches for natural language tracking. IEEE Trans Circuits Syst Video Technol 33(9):4529\u20134541","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"8377_CR18","doi-asserted-by":"crossref","unstructured":"Zhou L, Zhou Z, Mao K, He Z (2023) Joint visual grounding and tracking with natural language specification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 23151\u201323160","DOI":"10.1109\/CVPR52729.2023.02217"},{"issue":"4","key":"8377_CR19","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.1109\/TCSVT.2023.3301933","volume":"34","author":"Y Zheng","year":"2023","unstructured":"Zheng Y, Zhong B, Liang Q, Li G, Ji R, Li X (2023) Toward unified token learning for vision-language tracking. IEEE Trans Circuits Syst Video Technol 34(4):2125\u20132135","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"issue":"9","key":"8377_CR20","doi-asserted-by":"publisher","first-page":"1079","DOI":"10.1007\/s11227-025-07472-8","volume":"81","author":"J Zhang","year":"2025","unstructured":"Zhang J, Yan X, Zhang H, Xu L, Jiang B, Zhong B (2025) Vision-language discriminative fusion network for object tracking. J Supercomput 81(9):1079","journal-title":"J Supercomput"},{"key":"8377_CR21","doi-asserted-by":"crossref","unstructured":"Shao Y, He S, Ye Q, Feng Y, Luo W, Chen J (2024) Context-aware integration of language and visual references for natural language tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 19208\u201319217","DOI":"10.1109\/CVPR52733.2024.01817"},{"key":"8377_CR22","doi-asserted-by":"crossref","unstructured":"Li X, Huang Y, He Z, Wang Y, Lu H, Yang M-H (2023) Citetracker: Correlating image and text for visual tracking. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9974\u20139983","DOI":"10.1109\/ICCV51070.2023.00915"},{"key":"8377_CR23","doi-asserted-by":"crossref","unstructured":"Ma D, Wu X (2023) Tracking by natural language specification with long short-term context decoupling. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 14012\u201314021","DOI":"10.1109\/ICCV51070.2023.01288"},{"key":"8377_CR24","doi-asserted-by":"crossref","unstructured":"Cao Z, Huang Z, Pan L, Zhang S, Liu Z, Fu C (2022) Tctrack: Temporal contexts for aerial tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 14798\u201314808","DOI":"10.1109\/CVPR52688.2022.01438"},{"issue":"2","key":"8377_CR25","doi-asserted-by":"publisher","first-page":"819","DOI":"10.1007\/s13042-024-02296-z","volume":"16","author":"H Zhang","year":"2025","unstructured":"Zhang H, Ma Z, Zhao Y, Wang Y, Jiang B (2025) Reciprocal interlayer-temporal discriminative target model for robust visual tracking. Int J Mach Learn Cybern 16(2):819\u2013834","journal-title":"Int J Mach Learn Cybern"},{"key":"8377_CR26","doi-asserted-by":"crossref","unstructured":"Chen X, Peng H, Wang D, Lu H, Hu H (2023) Seqtrack: Sequence to sequence learning for visual object tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 14572\u201314581","DOI":"10.1109\/CVPR52729.2023.01400"},{"issue":"2","key":"8377_CR27","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1007\/s00138-024-01508-4","volume":"35","author":"H Zhang","year":"2024","unstructured":"Zhang H, Wang P, Chen Z, Zhang J, Li L (2024) Target-distractor memory joint tracking algorithm via credit allocation network. Mach Vis Appl 35(2):28","journal-title":"Mach Vis Appl"},{"key":"8377_CR28","unstructured":"Li XL, Liang P (2021) Prefix-tuning: Optimizing continuous prompts for generation. arXiv preprint arXiv:2101.00190"},{"key":"8377_CR29","doi-asserted-by":"crossref","unstructured":"Khattak MU, Rasheed H, Maaz M, Khan S, Khan FS (2023) Maple: Multi-modal prompt learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 19113\u201319122","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"8377_CR30","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, et al (2021) Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, 8748\u20138763. PmLR"},{"key":"8377_CR31","doi-asserted-by":"crossref","unstructured":"Rao Y, Zhao W, Chen G, Tang Y, Zhu Z, Huang G, Zhou J, Lu J (2022) Denseclip: Language-guided dense prediction with context-aware prompting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 18082\u201318091","DOI":"10.1109\/CVPR52688.2022.01755"},{"issue":"9","key":"8377_CR32","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy CC, Liu Z (2022) Learning to prompt for vision-language models. Int J Comput Vision 130(9):2337\u20132348","journal-title":"Int J Comput Vision"},{"key":"8377_CR33","doi-asserted-by":"crossref","unstructured":"Zhu J, Lai S, Chen X, Wang D, Lu H (2023) Visual prompt multi-modal tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 9516\u20139526","DOI":"10.1109\/CVPR52729.2023.00918"},{"key":"8377_CR34","first-page":"14903","volume":"37","author":"X Feng","year":"2024","unstructured":"Feng X, Li X, Hu S, Zhang D, Zhang J, Chen X, Huang K et al (2024) Memvlt: Vision-language tracking with adaptive memory-based prompts. Adv Neural Inf Process Syst 37:14903\u201314933","journal-title":"Adv Neural Inf Process Syst"},{"key":"8377_CR35","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, et al (2020) An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929"},{"key":"8377_CR36","first-page":"13937","volume":"34","author":"Y Rao","year":"2021","unstructured":"Rao Y, Zhao W, Liu B, Lu J, Zhou J, Hsieh C-J (2021) Dynamicvit: Efficient vision transformers with dynamic token sparsification. Adv Neural Inf Process Syst 34:13937\u201313949","journal-title":"Adv Neural Inf Process Syst"},{"key":"8377_CR37","doi-asserted-by":"crossref","unstructured":"Jiang B, Luo R, Mao J, Xiao T, Jiang Y (2018) Acquisition of localization confidence for accurate object detection. In: Proceedings of the European Conference on Computer Vision (ECCV), 784\u2013799","DOI":"10.1007\/978-3-030-01264-9_48"},{"key":"8377_CR38","doi-asserted-by":"crossref","unstructured":"Danelljan M, Bhat G, Khan FS, Felsberg M (2019) Atom: Accurate tracking by overlap maximization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 4660\u20134669","DOI":"10.1109\/CVPR.2019.00479"},{"key":"8377_CR39","doi-asserted-by":"crossref","unstructured":"Woo S, Park J, Lee J-Y, Kweon IS (2018) Cbam: Convolutional block attention module. In: Proceedings of the European Conference on Computer Vision (ECCV), 3\u201319","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"8377_CR40","doi-asserted-by":"crossref","unstructured":"Cai W, Liu Q, Wang Y (2024) Hiptrack: Visual tracking with historical prompts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 19258\u201319267","DOI":"10.1109\/CVPR52733.2024.01822"},{"key":"8377_CR41","doi-asserted-by":"crossref","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (long and Short Papers), 4171\u20134186","DOI":"10.18653\/v1\/N19-1423"},{"key":"8377_CR42","doi-asserted-by":"crossref","unstructured":"Wu Q, Yang T, Liu Z, Wu B, Shan Y, Chan AB (2023) Dropmae: Masked autoencoders with spatial-attention dropout for tracking tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 14561\u201314571","DOI":"10.1109\/CVPR52729.2023.01399"},{"key":"8377_CR43","doi-asserted-by":"crossref","unstructured":"Fan H, Lin L, Yang F, Chu P, Deng G, Yu S, Bai H, Xu Y, Liao C, Ling H (2019) Lasot: A high-quality benchmark for large-scale single object tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 5374\u20135383","DOI":"10.1109\/CVPR.2019.00552"},{"key":"8377_CR44","doi-asserted-by":"crossref","unstructured":"Wang X, Shu X, Zhang Z, Jiang B, Wang Y, Tian Y, Wu F (2021) Towards more flexible and accurate object tracking with natural language: Algorithms and benchmark. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 13763\u201313773","DOI":"10.1109\/CVPR46437.2021.01355"},{"key":"8377_CR45","doi-asserted-by":"crossref","unstructured":"Li Z, Tao R, Gavves E, Snoek CG, Smeulders AW (2017) Tracking by natural language specification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 6495\u20136503","DOI":"10.1109\/CVPR.2017.777"},{"key":"8377_CR46","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101"},{"issue":"2","key":"8377_CR47","doi-asserted-by":"publisher","first-page":"439","DOI":"10.1007\/s11263-020-01387-y","volume":"129","author":"H Fan","year":"2021","unstructured":"Fan H, Bai H, Lin L, Yang F, Chu P, Deng G, Yu S, Harshit Huang M, Liu J et al (2021) Lasot: A high-quality large-scale single object tracking benchmark. Int J Comput Vision 129(2):439\u2013461","journal-title":"Int J Comput Vision"},{"key":"8377_CR48","doi-asserted-by":"crossref","unstructured":"Liu X, Zhou L, Zhou Z, Chen J, He Z (2025) Mambavlt: Time-evolving multimodal state space model for vision-language tracking. In: Proceedings of the Computer Vision and Pattern Recognition Conference, 8731\u20138741","DOI":"10.1109\/CVPR52734.2025.00816"},{"key":"8377_CR49","doi-asserted-by":"crossref","unstructured":"Ge J, Cao J, Zhu X, Zhang X, Liu C, Wang K, Liu B (2024) Consistencies are all you need for semi-supervised vision-language tracking. In: Proceedings of the 32nd ACM International Conference on Multimedia, 1895\u20131904","DOI":"10.1145\/3664647.3680657"},{"key":"8377_CR50","doi-asserted-by":"crossref","unstructured":"Feng Q, Ablavsky V, Bai Q, Li G, Sclaroff S (2020) Real-time visual object tracking with natural language description. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 700\u2013709","DOI":"10.1109\/WACV45572.2020.9093425"},{"key":"8377_CR51","first-page":"4446","volume":"35","author":"M Guo","year":"2022","unstructured":"Guo M, Zhang Z, Fan H, Jing L (2022) Divert more attention to vision-language tracking. Adv Neural Inf Process Syst 35:4446\u20134460","journal-title":"Adv Neural Inf Process Syst"},{"key":"8377_CR52","doi-asserted-by":"crossref","unstructured":"Wang X, Shu X, Zhang Z, Jiang B, Wang Y, Tian Y, Wu F (2021) Towards more flexible and accurate object tracking with natural language: Algorithms and benchmark. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 13763\u201313773","DOI":"10.1109\/CVPR46437.2021.01355"},{"key":"8377_CR53","doi-asserted-by":"crossref","unstructured":"Li Y, Yu J, Cai Z, Pan Y (2022) Cross-modal target retrieval for tracking by natural language. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 4931\u20134940","DOI":"10.1109\/CVPRW56347.2022.00540"},{"key":"8377_CR54","first-page":"4446","volume":"35","author":"M Guo","year":"2022","unstructured":"Guo M, Zhang Z, Fan H, Jing L (2022) Divert more attention to vision-language tracking. Adv Neural Inf Process Syst 35:4446\u20134460","journal-title":"Adv Neural Inf Process Syst"},{"key":"8377_CR55","doi-asserted-by":"publisher","first-page":"10","DOI":"10.1016\/j.patrec.2023.02.023","volume":"168","author":"H Zhao","year":"2023","unstructured":"Zhao H, Wang X, Wang D, Lu H, Ruan X (2023) Transformer vision-language tracking via proxy token guided cross-modal fusion. Pattern Recogn Lett 168:10\u201316","journal-title":"Pattern Recogn Lett"},{"issue":"10","key":"8377_CR56","doi-asserted-by":"publisher","first-page":"9053","DOI":"10.1109\/TCSVT.2024.3395352","volume":"34","author":"G Zhang","year":"2024","unstructured":"Zhang G, Zhong B, Liang Q, Mo Z, Li N, Song S (2024) One-stream stepwise decreasing for vision-language tracking. IEEE Trans Circuits Syst Video Technol 34(10):9053\u20139063","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"8377_CR57","doi-asserted-by":"crossref","unstructured":"Chen X, Kang B, Geng W, Zhu J, Liu Y, Wang D, Lu H (2025) Sutrack: Towards simple and unified single object tracking. In: Proceedings of the AAAI Conference on Artificial Intelligence, 39, 2239\u20132247","DOI":"10.1609\/aaai.v39i2.32223"},{"key":"8377_CR58","doi-asserted-by":"crossref","unstructured":"Ma Y, Tang Y, Yang W, Zhang T, Zhang J, Kang M (2024) Unifying visual and vision-language tracking via contrastive learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, 38, 4107\u20134116","DOI":"10.1609\/aaai.v38i5.28205"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08377-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-026-08377-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08377-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T17:17:26Z","timestamp":1772644646000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-026-08377-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,4]]},"references-count":58,"journal-issue":{"issue":"4","published-online":{"date-parts":[[2026,3]]}},"alternative-id":["8377"],"URL":"https:\/\/doi.org\/10.1007\/s11227-026-08377-w","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,4]]},"assertion":[{"value":"9 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interest regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The research presented in this paper does not involve any human or animal subjects. All data used in this study are publicly available. Therefore, no ethical approval or informed consent was required for this research.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}}],"article-number":"219"}}