{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T16:57:45Z","timestamp":1780937865263,"version":"3.54.1"},"reference-count":35,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62502180"],"award-info":[{"award-number":["62502180"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132255","type":"journal-article","created":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T16:12:59Z","timestamp":1774714379000},"page":"132255","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Selective distillation of language tokens for redundancy suppression in vision-language tracking"],"prefix":"10.1016","volume":"321","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8060-4725","authenticated-orcid":false,"given":"Tian","family":"Bai","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1831-0653","authenticated-orcid":false,"given":"Shirui","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1513-0313","authenticated-orcid":false,"given":"Guangtong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132255_bib0001","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19048","article-title":"Artrackv2: Prompting autoregressive tracker where to look and how to describe","author":"Bai","year":"2024"},{"key":"10.1016\/j.eswa.2026.132255_bib0002","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"14572","article-title":"Seqtrack: Sequence to sequence learning for visual object tracking","author":"Chen","year":"2023"},{"key":"10.1016\/j.eswa.2026.132255_bib0003","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8126","article-title":"Transformer tracking","author":"Chen","year":"2021"},{"issue":"4","key":"10.1016\/j.eswa.2026.132255_bib0004","first-page":"5158","article-title":"SiamBAN: Target-aware tracking with siamese box adaptive network","volume":"45","author":"Chen","year":"2022","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132255_bib0005","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13608","article-title":"Mixformer: End-to-end tracking with iterative mixed attention","author":"Cui","year":"2022"},{"key":"10.1016\/j.eswa.2026.132255_bib0006","series-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: Human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.132255_bib0007","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5374","article-title":"Lasot: A high-quality benchmark for large-scale single object tracking","author":"Fan","year":"2019"},{"key":"10.1016\/j.eswa.2026.132255_bib0008","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"700","article-title":"Real-time visual object tracking with natural language description","author":"Feng","year":"2020"},{"key":"10.1016\/j.eswa.2026.132255_bib0009","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5851","article-title":"Siamese natural language tracker: Tracking by natural language descriptions with Siamese trackers","author":"Feng","year":"2021"},{"key":"10.1016\/j.eswa.2026.132255_bib0010","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11513","article-title":"H-vit: A hierarchical vision transformer for deformable image registration","author":"Ghahremani","year":"2024"},{"key":"10.1016\/j.eswa.2026.132255_bib0011","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"7201","article-title":"DreamTrack: Dreaming the future for multimodal visual object tracking","author":"Guo","year":"2025"},{"key":"10.1016\/j.eswa.2026.132255_bib0012","first-page":"25007","article-title":"A multi-modal global instance tracking benchmark (MGIT): Better locating target in complex spatio-temporal and causal relationship","volume":"36","author":"Hu","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"5","key":"10.1016\/j.eswa.2026.132255_bib0013","doi-asserted-by":"crossref","first-page":"1562","DOI":"10.1109\/TPAMI.2019.2957464","article-title":"Got-10k: A large high-diversity benchmark for generic object tracking in the wild","volume":"43","author":"Huang","year":"2019","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132255_bib0014","unstructured":"Jiao, X., Yin, Y., Shang, L., Jiang, X., Chen, X., Li, L., Wang, F., & Liu, Q. (2019). TinyBert: Distilling Bert for natural language understanding. arXiv: 1909.10351."},{"key":"10.1016\/j.eswa.2026.132255_bib0015","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"19165","article-title":"Dynamic updates for language adaptation in visual-language tracking","author":"Li","year":"2025"},{"key":"10.1016\/j.eswa.2026.132255_bib0016","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"6495","article-title":"Tracking by natural language specification","author":"Li","year":"2017"},{"key":"10.1016\/j.eswa.2026.132255_bib0017","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"8731","article-title":"MambaVLT: Time-evolving multimodal state space model for vision-language tracking","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132255_bib0018","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). Roberta: A robustly optimized bert pretraining approach. arXiv: 1907.11692."},{"key":"10.1016\/j.eswa.2026.132255_bib0019","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"4107","article-title":"Unifying visual and vision-language tracking via contrastive learning","volume":"38","author":"Ma","year":"2024"},{"key":"10.1016\/j.eswa.2026.132255_bib0020","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"11","article-title":"Generation and comprehension of unambiguous object descriptions","author":"Mao","year":"2016"},{"key":"10.1016\/j.eswa.2026.132255_bib0021","series-title":"Proceedings of the european conference on computer vision (ECCV)","first-page":"300","article-title":"TrackingNet: A large-scale dataset and benchmark for object tracking in the wild","author":"Muller","year":"2018"},{"key":"10.1016\/j.eswa.2026.132255_bib0022","unstructured":"Sanh, V., Debut, L., Chaumond, J., & Wolf, T. (2019). DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter. arXiv: 1910.01108."},{"key":"10.1016\/j.eswa.2026.132255_bib0023","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19208","article-title":"Context-aware integration of language and visual references for natural language tracking","author":"Shao","year":"2024"},{"key":"10.1016\/j.eswa.2026.132255_bib0024","unstructured":"Sun, S., Cheng, Y., Gan, Z., & Liu, J. (2019). Patient knowledge distillation for bert model compression. arXiv: 1908.09355."},{"key":"10.1016\/j.eswa.2026.132255_bib0025","doi-asserted-by":"crossref","unstructured":"Sun, Z., Yu, H., Song, X., Liu, R., Yang, Y., & Zhou, D. (2020). MobileBert: A compact task-agnostic bert for resource-limited devices. arXiv: 2004.02984.","DOI":"10.18653\/v1\/2020.acl-main.195"},{"key":"10.1016\/j.eswa.2026.132255_bib0026","unstructured":"Tang, R., Lu, Y., Liu, L., Mou, L., Vechtomova, O., & Lin, J. (2019). Distilling task-specific knowledge from Bert into simple neural networks. arXiv: 1903.12136."},{"key":"10.1016\/j.eswa.2026.132255_bib0027","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13763","article-title":"Towards more flexible and accurate object tracking with natural language: Algorithms and benchmark","author":"Wang","year":"2021"},{"key":"10.1016\/j.eswa.2026.132255_bib0028","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19113","article-title":"DiffusionTrack: Point set diffusion model for visual object tracking","author":"Xie","year":"2024"},{"key":"10.1016\/j.eswa.2026.132255_bib0029","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19300","article-title":"Autoregressive queries for adaptive tracking with spatio-temporal transformers","author":"Xie","year":"2024"},{"key":"10.1016\/j.eswa.2026.132255_bib0030","series-title":"European conference on computer vision","first-page":"341","article-title":"Joint feature learning and relation modeling for tracking: A one-stream framework","author":"Ye","year":"2022"},{"key":"10.1016\/j.eswa.2026.132255_bib0031","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"5552","article-title":"All in one: Exploring unified vision-language tracking with multi-modal alignment","author":"Zhang","year":"2023"},{"issue":"10","key":"10.1016\/j.eswa.2026.132255_bib0032","doi-asserted-by":"crossref","first-page":"9053","DOI":"10.1109\/TCSVT.2024.3395352","article-title":"One-stream stepwise decreasing for vision-language tracking","volume":"34","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132255_bib0033","series-title":"Proc. 33rd int. joint conf. artif. intell","first-page":"1652","article-title":"Diffusion mask-driven visual-language tracking","author":"Zhang","year":"2024"},{"issue":"4","key":"10.1016\/j.eswa.2026.132255_bib0034","doi-asserted-by":"crossref","first-page":"2125","DOI":"10.1109\/TCSVT.2023.3301933","article-title":"Toward unified token learning for vision-language tracking","volume":"34","author":"Zheng","year":"2023","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.132255_bib0035","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"23151","article-title":"Joint visual grounding and tracking with natural language specification","author":"Zhou","year":"2023"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426011681?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426011681?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T15:59:38Z","timestamp":1780934378000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426011681"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":35,"alternative-id":["S0957417426011681"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132255","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Selective distillation of language tokens for redundancy suppression in vision-language tracking","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132255","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132255"}}