{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T08:11:53Z","timestamp":1783757513733,"version":"3.55.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100007224","name":"National Foundation for Science and Technology Development","doi-asserted-by":"publisher","award":["102.05-2025.16"],"award-info":[{"award-number":["102.05-2025.16"]}],"id":[{"id":"10.13039\/100007224","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s10994-026-07076-0","type":"journal-article","created":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:05:52Z","timestamp":1780661152000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SAMD: Span-Aware Matryoshka Distillation for Cross-Tokenizer Embedding Models"],"prefix":"10.1007","volume":"115","author":[{"given":"Thang Duc","family":"Tran","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Minh-Phuc","family":"Truong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Thuan Do","family":"Phan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linh Ngo","family":"Van","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,5]]},"reference":[{"key":"7076_CR1","unstructured":"Ba, J., & Caruana, R. (2014). Do deep nets really need to be deep? In Advances in neural information processing systems (p. 27)."},{"key":"7076_CR2","doi-asserted-by":"crossref","unstructured":"Barbieri, F., Camacho-Collados, J., Anke, L. E., & Neves, L. (2020). Tweeteval: Unified benchmark and comparative evaluation for tweet classification. In Findings of the Association for Computational Linguistics: EMNLP 2020 (pp. 1644\u20131650)","DOI":"10.18653\/v1\/2020.findings-emnlp.148"},{"key":"7076_CR3","unstructured":"BehnamGhader, P., Adlakha, V., Mosbach, M., Bahdanau, D., Chapados, N., & Reddy, S. (2024). LLM2Vec: Large language models are secretly powerful text encoders. In First conference on language modeling. https:\/\/openreview.net\/forum?id=IW1PR7vEBf"},{"key":"7076_CR4","unstructured":"Bhattarai, P., Amjad, M., Zhylko, D., & Alhanai, T. (2025). Knowledge distillation through geometry-aware representational alignment. arXiv preprint. arXiv:2509.25253"},{"key":"7076_CR5","unstructured":"Boizard, N., El Haddad, K., Hudelot, C., & Colombo, P. (2024). Towards cross-tokenizer distillation: the universal logit distillation loss for llms. arXiv preprint arXiv:2402.12030."},{"key":"7076_CR6","doi-asserted-by":"crossref","unstructured":"Bucilua, C., Caruana, R., & Niculescu-Mizil, A. (2006). Model compression. Proceedings of the 12th ACM SIGKDD international conference on knowledge discovery and data mining (pp. 535\u2013541)","DOI":"10.1145\/1150402.1150464"},{"key":"7076_CR7","unstructured":"Cai, M., Yang, J., Gao, J., & Lee, Y. J. (2024). Matryoshka multimodal models. arXiv:2405.17430"},{"key":"7076_CR8","doi-asserted-by":"crossref","unstructured":"Chen, Y., Liu, Y., Meng, F., Chen, Y., Xu, J., & Zhou, J. (2025). Enhancing cross-tokenizer knowledge distillation with contextual dynamical mapping. In Findings of the Association for Computational Linguistics: ACL 2025 (pp. 8005\u20138018).","DOI":"10.18653\/v1\/2025.findings-acl.419"},{"key":"7076_CR9","doi-asserted-by":"crossref","unstructured":"Conneau, A., & Kiela, D. (2018). SentEval: An evaluation toolkit for universal sentence representations. arXiv.org\/abs\/1803.05449","DOI":"10.63317\/2gscircifffd"},{"key":"7076_CR10","doi-asserted-by":"crossref","unstructured":"Cui, X., Zhu, M., Qin, Y., Xie, L., Zhou, W., & Li, H. (2025). Multi-level optimal transport for universal cross-tokenizer knowledge distillation on language models. In Proceedings of the AAAI conference on artificial intelligence (Vol. 39, pp. 23724\u201323732).","DOI":"10.1609\/aaai.v39i22.34543"},{"key":"7076_CR11","unstructured":"Dasgupta, S., & Cohn, T. (2025). Improving language model distillation through hidden state matching. In International Conference on learning representations (ICLR)."},{"key":"7076_CR12","unstructured":"Gao, C., Wu, X., Wang, P., Wang, J., Zang, L., Wang, Z., & Hu, S. (2021). Distilcse: Effective knowledge distillation for contrastive sentence embeddings. arXiv preprint. arXiv:2112.05638"},{"key":"7076_CR13","doi-asserted-by":"crossref","unstructured":"Gao, T., Yao, X., & Chen, D. (2021). Simcse: Simple contrastive learning of sentence embeddings. In Proceedings of the 2021 conference on empirical methods in natural language processing (pp. 6894\u20136910).","DOI":"10.18653\/v1\/2021.emnlp-main.552"},{"key":"7076_CR14","doi-asserted-by":"crossref","unstructured":"Gretton, A., Bousquet, O., Smola, A., & Sch\u00f6lkopf, B. (2005). Measuring statistical dependence with hilbert-schmidt norms. In Proceedings of algorithmic learning theory (ALT).","DOI":"10.1007\/11564089_7"},{"key":"7076_CR15","unstructured":"Gu, Y., Dong, L., Wei, F., & Huang, M. (2024). MiniLLM: Knowledge distillation of large language models. arXiv Preprint. https:\/\/arxiv.org\/abs\/2306.08543"},{"key":"7076_CR16","unstructured":"Gu, J., Zhai, S., Zhang, Y., Susskind, J., & Jaitly, N. (2024). Matryoshka diffusion models. arXiv:2310.15111."},{"key":"7076_CR17","unstructured":"Hinton, G., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv preprint. arXiv:1503.02531."},{"key":"7076_CR18","doi-asserted-by":"crossref","unstructured":"Hu, W., Dou, Z.-Y., Li, L. H., Kamath, A., Peng, N., & Chang, K.-W. (2024). Matryoshka query transformer for large vision-language models. arXiv:2405.19315","DOI":"10.52202\/079017-1588"},{"key":"7076_CR19","doi-asserted-by":"crossref","unstructured":"Jawahar, G., Sagot, B., & Seddah, D. (2019). What does Bert learn about the structure of language? In ACL 2019-57th annual meeting of the Association for Computational Linguistics","DOI":"10.18653\/v1\/P19-1356"},{"key":"7076_CR20","doi-asserted-by":"crossref","unstructured":"Jiao, X., Yin, Y., Shang, L., Jiang, X., Chen, X., Li, L., Wang, F., & Liu, Q. (2020). Tinybert: Distilling bert for natural language understanding. In Findings of the Association for Computational Linguistics: EMNLP 2020 (pp. 4163\u20134174).","DOI":"10.18653\/v1\/2020.findings-emnlp.372"},{"key":"7076_CR21","doi-asserted-by":"crossref","unstructured":"Khot, T., Sabharwal, A., & Clark, P. (2018). Scitail: A textual entailment dataset from science question answering. In Proceedings of the AAAI conference on artificial intelligence (Vol. 32).","DOI":"10.1609\/aaai.v32i1.12022"},{"key":"7076_CR22","doi-asserted-by":"crossref","unstructured":"Kim, Y., & Rush, A.M. (2016). Sequence-level knowledge distillation. In: Proceedings of the 2016 conference on empirical methods in natural language processing (EMNLP).","DOI":"10.18653\/v1\/D16-1139"},{"key":"7076_CR23","unstructured":"Ko, J., Chen, T., Kim, S., Ding, T., Liang, L., Zharkov, I., & Yun, S.-Y. (2025). Distillm-2: A contrastive approach boosts the distillation of LLMS. In International conference on machine learning (pp. 31044\u201331062). PMLR."},{"key":"7076_CR24","unstructured":"Kornblith, S., Norouzi, M., Lee, H., & Hinton, G. (2019). Similarity of neural network representations revisited. In International conference on machine learning (pp. 3519\u20133529). PMlR."},{"key":"7076_CR25","doi-asserted-by":"publisher","first-page":"30233","DOI":"10.52202\/068431-2192","volume":"35","author":"A Kusupati","year":"2022","unstructured":"Kusupati, A., Bhatt, G., Rege, A., Wallingford, M., Sinha, A., Ramanujan, V., Howard-Snyder, W., Chen, K., Kakade, S., Jain, P., et al. (2022). Matryoshka representation learning. Advances in Neural Information Processing Systems, 35, 30233\u201330249.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"7076_CR26","unstructured":"LI, X., Li, Z., Li, J., Xie, H., & Li, Q. (2025). ESE: Espresso sentence embeddings. In The Thirteenth international conference on learning representations. https:\/\/openreview.net\/forum?id=plgLA2YBLH"},{"key":"7076_CR27","doi-asserted-by":"crossref","unstructured":"Le, A. D., Le\u00a0Hai, N., Nguyen, T. X., Van, L. N., Diep, N. T. N., Dinh, S., & Nguyen, T. H. (2025). Enhancing discriminative representation in similar relation clusters for few-shot continual relation extraction. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers, pp. 2450\u20132467).","DOI":"10.18653\/v1\/2025.naacl-long.123"},{"key":"7076_CR28","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, L.L., Xu, J., Wang, Y., Yan, S., Xia, Y., Yang, Y., Cao, T., Sun, H., Deng, W. et al. (2023). Constraint-aware and ranking-distilled token pruning for efficient transformer inference. In: Proceedings of the 29th ACM SIGKDD Conference on knowledge discovery and data mining (pp. 1280\u20131290).","DOI":"10.1145\/3580305.3599284"},{"key":"7076_CR29","doi-asserted-by":"crossref","unstructured":"Marelli, M., Bentivogli, L., Baroni, M., Bernardi, R., Menini, S., & Zamparelli, R. (2014) Semeval-2014 task 1: Evaluation of compositional distributional semantic models on full sentences through semantic relatedness and textual entailment. In Proceedings of the 8th international workshop on semantic evaluation (SemEval 2014) (pp. 1\u20138).","DOI":"10.3115\/v1\/S14-2001"},{"key":"7076_CR30","unstructured":"Minixhofer, B., Ponti, E. M., & Vuli\u0107, I. (2025). Cross-tokenizer distillation via approximate likelihood matching. arXiv preprint. arXiv:2503.20083."},{"key":"7076_CR31","doi-asserted-by":"publisher","unstructured":"Mu, Y., Wu, Y., Fan, Y., Wang, C., Li, H., Zeng, J., He, Q., Yang, M., Meng, F., Zhou, J., Xiao, T., & Zhu, J. (2024). Cross-layer attention sharing for pre-trained large language models. arXiv preprint. arXiv:2408.01890. https:\/\/doi.org\/10.48550\/arXiv.2408.01890 .","DOI":"10.48550\/arXiv.2408.01890"},{"key":"7076_CR32","doi-asserted-by":"crossref","unstructured":"Muennighoff, N., Tazi, N., Magne, L., & Reimers, N. (2023). Mteb: Massive text embedding benchmark. In Proceedings of the 17th conference of the European Chapter of the Association for Computational Linguistics (pp. 2014\u20132037).","DOI":"10.18653\/v1\/2023.eacl-main.148"},{"key":"7076_CR33","unstructured":"On, F. F. (2026). Knowledge distillation for large language models through residual learning. In International conference on learning representations (ICLR)."},{"key":"7076_CR34","doi-asserted-by":"crossref","unstructured":"Pilehvar, M.T., & Camacho-Collados, J. (2019) Wic: The word-in-context dataset for evaluating context-sensitive meaning representations. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1 (Long and Short Papers), pp. 1267\u20131273).","DOI":"10.18653\/v1\/N19-1128"},{"key":"7076_CR35","unstructured":"Raghu, M., Gilmer, J., Yosinski, J., & Sohl-Dickstein, J. (2017). Svcca: Singular vector canonical correlation analysis for deep learning dynamics and interpretability. In Advances in neural information processing systems (p. 30)."},{"key":"7076_CR36","unstructured":"Romero, A., Ballas, N., Kahou, S. E., Chassang, A., Gatta, C., & Bengio, Y. (2015). FITNETS: Hints for thin deep nets. In International conference on learning representations (ICLR)."},{"key":"7076_CR37","unstructured":"Saadi, K., & Wang, D. (2025). Flexible feature distillation for large language models. arXiv preprint. arXiv:2507.10155."},{"key":"7076_CR38","unstructured":"Sanh, V., Debut, L., Chaumond, J., & Wolf, T. (2019). Distilbert, a distilled version of BERT: Smaller, faster, cheaper and lighter. arXiv preprint. arXiv:1910.01108."},{"key":"7076_CR39","doi-asserted-by":"crossref","unstructured":"Sun, S., Cheng, Y., Gan, Z., & Liu, J. (2019). Patient knowledge distillation for BERT model compression. In Proceedings of EMNLP-IJCNLP.","DOI":"10.18653\/v1\/D19-1441"},{"key":"7076_CR40","doi-asserted-by":"crossref","unstructured":"Truong, M.-P., Vu, H. A., Vu, T., Diep, N. T. N., Van, L. N., Nguyen, T. H., & Le, T. (2025). EMO: Embedding model distillation via intra-model relation and optimal transport alignments. In Proceedings of the 2025 conference on empirical methods in natural language processing (pp. 7605\u20137617).","DOI":"10.18653\/v1\/2025.emnlp-main.385"},{"key":"7076_CR41","unstructured":"Wan, F., Huang, X., Cai, D., & Quan, X. (2024). Knowledge fusion of large language models. International Conference on Learning Representations (ICLR)"},{"key":"7076_CR42","doi-asserted-by":"crossref","unstructured":"Wang, A., Singh, A., Michael, J., Hill, F., Levy, O., & Bowman, S. (2018). Glue: A multi-task benchmark and analysis platform for natural language understanding. In Proceedings of the 2018 EMNLP workshop BlackboxNLP: Analyzing and interpreting neural networks for NLP (pp. 353\u2013355).","DOI":"10.18653\/v1\/W18-5446"},{"key":"7076_CR43","unstructured":"Wang, G., Yang, Z., Wang, Z., Wang, S., Xu, Q., & Huang, Q. (2025). ABKD: Pursuing a proper allocation of the probability mass in knowledge distillation via $$\\alpha -\\beta$$-divergence. arXiv preprint. arXiv:2505.04560"},{"key":"7076_CR44","first-page":"5776","volume":"33","author":"W Wang","year":"2020","unstructured":"Wang, W., Wei, F., Dong, L., Bao, H., Yang, N., & Zhou, M. (2020). Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. Advances in Neural Information Processing Systems, 33, 5776\u20135788.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"7076_CR45","unstructured":"Zhang, D., Li, J., Zeng, Z., & Wang, F. (2024). Jasper and stella: distillation of sota embedding models. arXiv preprint. arXiv:2412.19048."},{"key":"7076_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, S., Zhang, X., Sun, Z., Chen, Y., & Xu, J. (2024). Dual-space knowledge distillation for large language models. In Proceedings of the 2024 conference on empirical methods in natural language processing (EMNLP).","DOI":"10.18653\/v1\/2024.emnlp-main.1010"},{"key":"7076_CR47","unstructured":"Zhou, Z., Shen, Y., Shao, S., Gong, L., & Lin, S. (2024) Rethinking centered kernel alignment in knowledge distillation. In Proceedings of the 33rd international joint conference on artificial intelligence (IJCAI). arXiv:2401.11824."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-026-07076-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-026-07076-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-026-07076-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T07:13:33Z","timestamp":1783754013000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-026-07076-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":47,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["7076"],"URL":"https:\/\/doi.org\/10.1007\/s10994-026-07076-0","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"17 January 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 April 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 May 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 June 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to Participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}}],"article-number":"143"}}