{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T02:59:59Z","timestamp":1780973999594,"version":"3.54.1"},"reference-count":65,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100006247","name":"Anhui University of Science and Technology","doi-asserted-by":"publisher","award":["YZ2023H2C005"],"award-info":[{"award-number":["YZ2023H2C005"]}],"id":[{"id":"10.13039\/501100006247","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003995","name":"Natural Science Foundation of Anhui Province","doi-asserted-by":"publisher","award":["2508085MF168"],"award-info":[{"award-number":["2508085MF168"]}],"id":[{"id":"10.13039\/501100003995","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132045","type":"journal-article","created":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T17:09:00Z","timestamp":1773680940000},"page":"132045","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["TIKD: Where text meets vision for knowledge distillation"],"prefix":"10.1016","volume":"319","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8674-7302","authenticated-orcid":false,"given":"Xingzhu","family":"Liang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3503-5170","authenticated-orcid":false,"given":"Chun","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9285-0693","authenticated-orcid":false,"given":"Mengyuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3337-563X","authenticated-orcid":false,"given":"Yu-e","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132045_bib0001","unstructured":"Aladago, M. M., & Piergiovanni, A. J. (2022). Compound tokens: Channel fusion for vision-language representation learning. arXiv preprint arXiv: 2212.01447."},{"key":"10.1016\/j.eswa.2026.132045_bib0002","doi-asserted-by":"crossref","first-page":"23716","DOI":"10.52202\/068431-1723","article-title":"Flamingo: a visual language model for few-shot learning","volume":"35","author":"Alayrac","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132045_bib0003","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"738","article-title":"Local-global multi-modal distillation for weakly-supervised temporal video grounding","volume":"vol. 38","author":"Bao","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0004","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"19846","article-title":"Move-kd: Knowledge distillation for vlms with mixture of visual encoders","author":"Cao","year":"2025"},{"key":"10.1016\/j.eswa.2026.132045_bib0005","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5008","article-title":"Distilling knowledge via knowledge review","author":"Chen","year":"2021"},{"key":"10.1016\/j.eswa.2026.132045_bib0006","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12052","article-title":"Dearkd: data-efficient early knowledge distillation for vision transformers","author":"Chen","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0007","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv: 2010.11929."},{"key":"10.1016\/j.eswa.2026.132045_bib0008","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19135","article-title":"ikun: Speak to trackers without retraining","author":"Du","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0009","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107429","article-title":"Neighborhood relation-based knowledge distillation for image classification","volume":"188","author":"Gou","year":"2025","journal-title":"Neural Networks"},{"key":"10.1016\/j.eswa.2026.132045_bib0010","first-page":"9164","article-title":"Learning efficient vision transformers via fine-grained manifold distillation","volume":"35","author":"Hao","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132045_bib0011","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.eswa.2026.132045_bib0012","unstructured":"Hinton, G., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv preprint arXiv: 1503.02531."},{"key":"10.1016\/j.eswa.2026.132045_bib0013","series-title":"International conference on medical image computing and computer-assisted intervention","first-page":"772","article-title":"Knowledge distillation from multi-modal to mono-modal segmentation networks","author":"Hu","year":"2020"},{"key":"10.1016\/j.eswa.2026.132045_bib0014","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"24222","article-title":"Multi-modal knowledge distillation-based human trajectory forecasting","author":"Jeong","year":"2025"},{"key":"10.1016\/j.eswa.2026.132045_bib0015","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16876","article-title":"Structural and statistical texture knowledge distillation for semantic segmentation","author":"Ji","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0016","unstructured":"Jie, S., Tang, Y., Ding, N., Deng, Z.-H., Han, K., & Wang, Y. (2024). Memory-space visual prompting for efficient vision-language fine-tuning. arXiv preprint arXiv: 2405.05615."},{"key":"10.1016\/j.eswa.2026.132045_bib0017","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"24276","article-title":"Multi-level logit distillation","author":"Jin","year":"2023"},{"key":"10.1016\/j.eswa.2026.132045_bib0018","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"15033","article-title":"Distilling spectral graph for object-context aware open-vocabulary semantic segmentation","author":"Kim","year":"2025"},{"key":"10.1016\/j.eswa.2026.132045_bib0019","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"14690","article-title":"Cosmos: Cross-modality self-distillation for vision language pre-training","author":"Kim","year":"2025"},{"key":"10.1016\/j.eswa.2026.132045_bib0020","series-title":"International conference on machine learning","first-page":"5583","article-title":"Vilt: Vision-and-language transformer without convolution or region supervision","author":"Kim","year":"2021"},{"key":"10.1016\/j.eswa.2026.132045_bib0021","series-title":"International conference on machine learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0022","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume":"34","author":"Li","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132045_bib0023","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"24408","article-title":"One prompt word is enough to boost adversarial robustness for pre-trained vision-language models","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0024","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"26617","article-title":"Promptkd: Unsupervised prompt distillation for vision-language models","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0025","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"1504","article-title":"Curriculum temperature for knowledge distillation","volume":"vol. 37","author":"Li","year":"2023"},{"issue":"8","key":"10.1016\/j.eswa.2026.132045_bib0026","doi-asserted-by":"crossref","first-page":"10055","DOI":"10.1109\/TPAMI.2023.3262578","article-title":"Local-global context aware transformer for language-guided video segmentation","volume":"45","author":"Liang","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132045_bib0027","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10915","article-title":"Knowledge distillation via the target-aware transformer","author":"Lin","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0028","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"14420","article-title":"Efficientvit: Memory efficient vision transformer with cascaded group attention","author":"Liu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132045_bib0029","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"10012","article-title":"Swin transformer: Hierarchical vision transformer using shifted windows","author":"Liu","year":"2021"},{"key":"10.1016\/j.eswa.2026.132045_bib0030","first-page":"65445","article-title":"Wasserstein distance rivals kullback-leibler divergence for knowledge distillation","volume":"37","author":"Lv","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132045_bib0031","first-page":"14200","article-title":"Attention bottlenecks for multimodal fusion","volume":"34","author":"Nagrani","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"2","key":"10.1016\/j.eswa.2026.132045_bib0032","doi-asserted-by":"crossref","first-page":"2887","DOI":"10.1007\/s11042-020-08836-3","article-title":"Deep learning-based late fusion of multimodal information for emotion classification of music video","volume":"80","author":"Pandeya","year":"2021","journal-title":"Multimedia Tools and Applications"},{"key":"10.1016\/j.eswa.2026.132045_bib0033","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"5213","article-title":"Multimodal distillation for egocentric action recognition","author":"Radevski","year":"2023"},{"key":"10.1016\/j.eswa.2026.132045_bib0034","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.132045_bib0035","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3687","article-title":"Tinymim: An empirical study of distilling mim pre-trained models","author":"Ren","year":"2023"},{"key":"10.1016\/j.eswa.2026.132045_bib0036","unstructured":"Romero, A., Ballas, N., Kahou, S. E., Chassang, A., Gatta, C., & Bengio, Y. (2014). Fitnets: hints for thin deep nets; 2014. arXiv preprint arXiv: 1412.6550, 3."},{"key":"10.1016\/j.eswa.2026.132045_bib0037","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"17959","article-title":"End-to-end generative pretraining for multimodal video captioning","author":"Seo","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0038","unstructured":"Simonyan, K., & Zisserman, A. (2014). Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv: 1409.1556."},{"key":"10.1016\/j.eswa.2026.132045_bib0039","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15638","article-title":"Flava: A foundational language and vision alignment model","author":"Singh","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0040","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15731","article-title":"Logit standardization in knowledge distillation","author":"Sun","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0041","series-title":"International conference on machine learning","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","author":"Touvron","year":"2021"},{"key":"10.1016\/j.eswa.2026.132045_bib0042","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"32","article-title":"Going deeper with image transformers","author":"Touvron","year":"2021"},{"key":"10.1016\/j.eswa.2026.132045_bib0043","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15909","article-title":"Repvit: Revisiting mobile cnn from vit perspective","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0044","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16520","article-title":"CrossKD: Cross-head knowledge distillation for object detection","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0045","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12186","article-title":"Multimodal token fusion for vision transformers","author":"Wang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0046","unstructured":"Wei, S., Luo, Y., & Luo, C. (2023). One-stage modality distillation for incomplete multimodal learning. arXiv preprint arXiv: 2309.08204."},{"key":"10.1016\/j.eswa.2026.132045_bib0047","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"14633","article-title":"Referring multi-object tracking","author":"Wu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132045_bib0048","series-title":"European conference on computer vision","first-page":"68","article-title":"Tinyvit: Fast pretraining distillation for small vision transformers","author":"Wu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0049","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110959","article-title":"Multimodal self-supervised learning for remote sensing data land cover classification","volume":"157","author":"Xue","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132045_bib0050","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.128647","article-title":"Weak galaxy object detection using improved YOLOX model with feature map knowledge distillation","author":"Yan","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132045_bib0051","unstructured":"Yang, Q., Zhao, Y., & Cheng, H. (2024a). Mmlf: Multi-modal multi-class late fusion for object detection with uncertainty estimation. arXiv preprint arXiv: 2410.08739."},{"key":"10.1016\/j.eswa.2026.132045_bib0052","series-title":"European conference on computer vision","first-page":"53","article-title":"Masked generative distillation","author":"Yang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0053","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"1379","article-title":"Vitkd: Feature-based knowledge distillation for vision transformers","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0054","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"4210","article-title":"Effective whole-body pose estimation with two-stages distillation","author":"Yang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132045_bib0055","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11794","article-title":"Gated channel transformation for visual recognition","author":"Yang","year":"2020"},{"key":"10.1016\/j.eswa.2026.132045_bib0056","unstructured":"Yao, Y., Yu, T., Zhang, A., Wang, C., Cui, J., Zhu, H., Cai, T., Li, H., Zhao, W., He, Z. et al. (2024). Minicpm-v: A gpt-4v level mllm on your phone. arXiv preprint arXiv: 2408.01800."},{"key":"10.1016\/j.eswa.2026.132045_bib0057","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125670","article-title":"Attention correction feature and boundary constraint knowledge distillation for efficient 3d medical image segmentation","volume":"262","author":"Yu","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132045_bib0058","unstructured":"Zagoruyko, S., & Komodakis, N. (2016). Paying more attention to attention: Improving the performance of convolutional neural networks via attention transfer. arXiv preprint arXiv: 1612.03928."},{"key":"10.1016\/j.eswa.2026.132045_bib0059","doi-asserted-by":"crossref","unstructured":"Zhang, B., Qin, J., Xiang, X., & Tan, Y. (2024a). Learning continuation: Integrating past knowledge for contrastive distillation. Knowledge-Based Systems, 304, 112573.","DOI":"10.1016\/j.knosys.2024.112573"},{"issue":"8","key":"10.1016\/j.eswa.2026.132045_bib0060","doi-asserted-by":"crossref","first-page":"5625","DOI":"10.1109\/TPAMI.2024.3369699","article-title":"Vision-language models for vision tasks: A survey","volume":"46","author":"Zhang","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine intelligence"},{"key":"10.1016\/j.eswa.2026.132045_bib0061","series-title":"2024 IEEE international joint conference on biometrics (IJCB)","first-page":"1","article-title":"Cpl-clip: Compound prompt learning for flexible-modal face anti-spoofing","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132045_bib0062","article-title":"Fdbpl: Faster distillation-based prompt learning for region-aware vision-language models adaptation","author":"Zhang","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132045_bib0063","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11953","article-title":"Decoupled knowledge distillation","author":"Zhao","year":"2022"},{"key":"10.1016\/j.eswa.2026.132045_bib0064","series-title":"Proceedings of the 2023 ACM international conference on multimedia retrieval","first-page":"622","article-title":"Clap: Contrastive language-audio pre-training model for multi-modal sentiment analysis","author":"Zhao","year":"2023"},{"key":"10.1016\/j.eswa.2026.132045_bib0065","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111029","article-title":"Distilling efficient vision transformers from cnns for semantic segmentation","volume":"158","author":"Zheng","year":"2025","journal-title":"Pattern Recognition"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426009589?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426009589?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T02:36:48Z","timestamp":1780972608000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426009589"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":65,"alternative-id":["S0957417426009589"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132045","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"TIKD: Where text meets vision for knowledge distillation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132045","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132045"}}