{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T10:58:49Z","timestamp":1774522729434,"version":"3.50.1"},"reference-count":88,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T00:00:00Z","timestamp":1771200000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T00:00:00Z","timestamp":1771200000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s13042-025-02952-y","type":"journal-article","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T11:34:35Z","timestamp":1771241675000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A training-inference consistent framework for early exiting in language models with parallel decoding"],"prefix":"10.1007","volume":"17","author":[{"given":"Ziqian","family":"Zeng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zelin","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huiping","family":"Zhuang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,16]]},"reference":[{"key":"2952_CR1","doi-asserted-by":"crossref","unstructured":"Devlin J, Chang M, Lee K, Toutanova K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, NAACL-HLT 2019, pp 4171\u20134186","DOI":"10.18653\/v1\/N19-1423"},{"key":"2952_CR2","unstructured":"Liu Y, Ott M, Goyal N, Du J, Joshi M, Chen D, Levy O, Lewis M, Zettlemoyer L, Stoyanov V (2019) Roberta: a robustly optimized bert pretraining approach. ArXiv:abs\/1907.11692"},{"key":"2952_CR3","unstructured":"Yang Z, Dai Z, Yang Y, Carbonell J, Salakhutdinov R, Le QV (2019) Xlnet: generalized autoregressive pretraining for language understanding. In: Proceedings of the 33rd international conference on neural information processing systems, NeurIPS 2019"},{"key":"2952_CR4","doi-asserted-by":"crossref","unstructured":"Nallapati R, Zhou B, Santos C, Xiang B (2016) Abstractive text summarization using sequence-to-sequence rnns and beyond. In: Proceedings of the 20th SIGNLL conference on computational natural language learning, pp 280\u2013290","DOI":"10.18653\/v1\/K16-1028"},{"key":"2952_CR5","doi-asserted-by":"crossref","unstructured":"Fabbri AR, Li I, She T, Li S, Radev D (2019) Multi-news: a large-scale multi-document summarization dataset and abstractive hierarchical model. In: Proceedings of the 57th annual meeting of the association for computational linguistics, ACL 2019, pp 1074\u20131084","DOI":"10.18653\/v1\/P19-1102"},{"key":"2952_CR6","doi-asserted-by":"crossref","unstructured":"Rajpurkar P, Zhang J, Lopyrev K, Liang P (2016) Squad: 100, 000+ questions for machine comprehension of text. In: Proceedings of the 2016 conference on empirical methods in natural language processing, EMNLP 2016, pp 2383\u20132392","DOI":"10.18653\/v1\/D16-1264"},{"key":"2952_CR7","unstructured":"Cettolo M, Federico M, Bentivogli L, Niehues J, St\u00fcker S, Sudoh K, Yoshino K, Federmann C (2017) Overview of the iwslt 2017 evaluation campaign. In: Proceedings of the 14th international workshop on spoken language translation, pp 2\u201314"},{"key":"2952_CR8","unstructured":"Touvron H, Martin L, Stone KR, Albert P, Almahairi A, Babaei Y, Bashlykov N, Batra S, Bhargava P, Bhosale S, et al (2023) Llama 2: open foundation and fine-tuned chat models. ArXiv:abs\/2307.09288"},{"key":"2952_CR9","unstructured":"Achiam J, Adler S, Agarwal S, Ahmad L, Akkaya I, Aleman FL, Almeida D, Altenschmidt J, Altman S, Anadkat S et al (2023) Gpt-4 technical report. ArXiv:abs\/2303.08774"},{"key":"2952_CR10","doi-asserted-by":"crossref","unstructured":"Hoffmann J, Borgeaud S, Mensch A, Buchatskaya E, Cai T, Rutherford E, Casas DdL, Hendricks LA, Welbl J, Clark A et al (2022) Training compute-optimal large language models. ArXiv:abs\/2203.15556","DOI":"10.52202\/068431-2176"},{"issue":"240","key":"2952_CR11","first-page":"1","volume":"24","author":"A Chowdhery","year":"2023","unstructured":"Chowdhery A, Narang S, Devlin J, Bosma M, Mishra G, Roberts A, Barham P, Chung HW, Sutton C, Gehrmann S et al (2023) Palm: scaling language modeling with pathways. J Mach Learn Res 24(240):1\u2013113","journal-title":"J Mach Learn Res"},{"key":"2952_CR12","unstructured":"Zhang B, Haddow B, Birch A (2023) Prompting large language model for machine translation: A case study. In: Proceedings of the 40th international conference on machine learning, ICML 2023, pp 41092\u201341110"},{"key":"2952_CR13","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1162\/tacl_a_00632","volume":"12","author":"T Zhang","year":"2024","unstructured":"Zhang T, Ladhak F, Durmus E, Liang P, McKeown K, Hashimoto TB (2024) Benchmarking large language models for news summarization. Trans Assoc Comput Linguist 12:39\u201357","journal-title":"Trans Assoc Comput Linguist"},{"key":"2952_CR14","doi-asserted-by":"publisher","first-page":"453","DOI":"10.1162\/tacl_a_00276","volume":"7","author":"T Kwiatkowski","year":"2019","unstructured":"Kwiatkowski T, Palomaki J, Redfield O, Collins M, Parikh A, Alberti C, Epstein D, Polosukhin I, Devlin J, Lee K et al (2019) Natural questions: a benchmark for question answering research. Trans Assoc Comput Linguist 7:453\u2013466","journal-title":"Trans Assoc Comput Linguist"},{"key":"2952_CR15","doi-asserted-by":"crossref","unstructured":"Lin X, Bertasius G, Wang J, Chang S-F, Parikh D, Torresani L (2021) Vx2text: end-to-end learning of video-based text generation from multimodal inputs. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, CVPR 2021, pp 7005\u20137015","DOI":"10.1109\/CVPR46437.2021.00693"},{"key":"2952_CR16","doi-asserted-by":"crossref","unstructured":"Kwon W, Li Z, Zhuang S, Sheng Y, Zheng L, Yu CH, Gonzalez J, Zhang H, Stoica I (2023) Efficient memory management for large language model serving with pagedattention. In: Proceedings of the 29th symposium on operating systems principles, pp 611\u2013626","DOI":"10.1145\/3600006.3613165"},{"key":"2952_CR17","unstructured":"Chelba C, Chen M, Bapna A, Shazeer N (2020) Faster transformer decoding: N-gram masked self-attention. ArXiv:abs\/2001.04589"},{"issue":"6","key":"2952_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3660522","volume":"42","author":"H Peng","year":"2024","unstructured":"Peng H, Zhang J, Huang X, Hao Z, Li A, Yu Z, Yu PS (2024) Unsupervised social bot detection via structural information theory. ACM Trans Inf Syst 42(6):1\u201342","journal-title":"ACM Trans Inf Syst"},{"key":"2952_CR19","unstructured":"Liu Z, Wang J, Dao T, Zhou T, Yuan B, Song Z, Shrivastava A, Zhang C, Tian Y, Re C et al (2023) Deja vu: contextual sparsity for efficient llms at inference time. In: Proceedings of the 40th international conference on machine learning, ICML 2023, pp 22137\u201322176"},{"key":"2952_CR20","first-page":"606","volume":"5","author":"R Pope","year":"2023","unstructured":"Pope R, Douglas S, Chowdhery A, Devlin J, Bradbury J, Heek J, Xiao K, Agrawal S, Dean J (2023) Efficiently scaling transformer inference. Proc Mach Learn Syst 5:606\u2013624","journal-title":"Proc Mach Learn Syst"},{"key":"2952_CR21","unstructured":"Frantar E, Ashkboos S, Hoefler T, Alistarh D (2023) Gptq: accurate post-training quantization for generative pre-trained transformers. In: Proceedings of the 11th international conference on learning representations, ICLR 2023"},{"key":"2952_CR22","first-page":"87","volume":"6","author":"J Lin","year":"2024","unstructured":"Lin J, Tang J, Tang H, Yang S, Chen W-M, Wang W-C, Xiao G, Dang X, Gan C, Han S (2024) Awq: activation-aware weight quantization for on-device llm compression and acceleration. Proc Mach Learn Syst 6:87\u2013100","journal-title":"Proc Mach Learn Syst"},{"key":"2952_CR23","unstructured":"Park G, Kim M, Lee S, Kim J, Kwon B, Kwon SJ, Kim B, Lee Y, Lee D et al (2022) Lut-gemm: quantized matrix multiplication based on luts for efficient inference in large-scale generative language models. In: Proceedings of the 10th international conference on learning representations, ICLR 2022"},{"key":"2952_CR24","unstructured":"Xiao G, Tian Y, Chen B, Han S, Lewis M (2024) Efficient streaming language models with attention sinks. In: Proceedings of the 12th international conference on learning representations, ICLR 2024"},{"key":"2952_CR25","doi-asserted-by":"crossref","unstructured":"Ma X, Fang G, Wang X (2023) Llm-pruner: on the structural pruning of large language models. Proceedings of the 37th international conference on neural information processing systems, NeurIPS 2023, 36, 21702\u201321720","DOI":"10.52202\/075280-0950"},{"key":"2952_CR26","unstructured":"Xia M, Gao T, Zeng Z, Chen D (2024) Sheared llama: accelerating language model pre-training via structured pruning. In: Proceedings of the 12th international conference on learning representations, ICLR, 2024"},{"key":"2952_CR27","unstructured":"Zhang Z, Sheng Y, Zhou T, Chen T, Zheng L, Cai R, Song Z, Tian Y, R\u00e9 C, Barrett CW, Wang Z, Chen B (2023) H2O: heavy-hitter oracle for efficient generative inference of large language models. In: Proceedings of the 37th international conference on neural information processing systems, NeurIPS, 2023"},{"key":"2952_CR28","unstructured":"Li Y, Yu Y, Zhang Q, Liang C, He P, Chen W, Zhao T (2023) Losparse: structured compression of large language models based on low-rank and sparse approximation. In: Proceedings of the 40th international conference on machine learning, ICML, 2023, vol. 202, pp 20336\u201320350"},{"key":"2952_CR29","unstructured":"Javaheripi M, Rosa G, Mukherjee S, Shah S, Religa T, Mendes CCT, Bubeck S, Koushanfar F, Dey D (2022) Litetransformersearch: training-free neural architecture search for efficient language models. In: Proceedings of the 36th international conference on neural information processing systems, NeurIPS, 2022"},{"key":"2952_CR30","unstructured":"Xu D, Mukherjee S, Liu X, Dey D, Wang W, Zhang X, Awadallah AH, Gao J (2022) Few-shot task-agnostic neural architecture search for distilling large language models. In: Proceedings of the 36th international conference on neural information processing systems, NeurIPS, 2022"},{"key":"2952_CR31","doi-asserted-by":"crossref","unstructured":"Xin J, Tang R, Lee J, Yu Y, Lin J (2020) Deebert: dynamic early exiting for accelerating BERT inference. In: Proceedings of the 58th annual meeting of the association for computational linguistics, ACL, 2020, pp 2246\u20132251","DOI":"10.18653\/v1\/2020.acl-main.204"},{"key":"2952_CR32","doi-asserted-by":"crossref","unstructured":"Sun T, Liu X, Zhu W, Geng Z, Wu L, He Y, Ni Y, Xie G, Huang X, Qiu X (2022) A simple hash-based early exiting approach for language understanding and generation. In: Findings of the association for computational linguistics, ACL, 2022, pp 2409\u20132421","DOI":"10.18653\/v1\/2022.findings-acl.189"},{"key":"2952_CR33","unstructured":"Schuster T, Fisch A, Gupta J, Dehghani M, Bahri D, Tran V, Tay Y, Metzler D (2022) Confident adaptive language modeling. In: Proceedings of the 36th international conference on neural information processing systems, NeurIPS, 2022"},{"key":"2952_CR34","unstructured":"Corro LD, Giorno AD, Agarwal S, Yu B, Awadallah A, Mukherjee S (2023) Skipdecode: autoregressive skip decoding with batching and caching for efficient LLM inference. Arxiv:abs\/2307.02628"},{"key":"2952_CR35","doi-asserted-by":"crossref","unstructured":"Geva M, Caciularu A, Wang KR, Goldberg Y (2022) Transformer feed-forward layers build predictions by promoting concepts in the vocabulary space. In: Proceedings of the 2022 conference on empirical methods in natural language processing, EMNLP 2022, pp 30\u201345","DOI":"10.18653\/v1\/2022.emnlp-main.3"},{"key":"2952_CR36","doi-asserted-by":"crossref","unstructured":"Liu Y, Meng F, Zhou J, Chen Y, Xu J (2021) Faster depth-adaptive transformers. In: Proceedings of the 35th AAAI conference on artificial intelligence, AAAI, 2021, pp 13424\u201313432","DOI":"10.1609\/aaai.v35i15.17584"},{"key":"2952_CR37","doi-asserted-by":"crossref","unstructured":"Zeng Z, Hong Y, Dai H, Zhuang H, Chen C (2024) Consistentee: a consistent and hardness-guided early exiting method for accelerating language models inference. In: Proceedings of the 38th AAAI conference on artificial intelligence, AAAI, 2024, vol. 38, pp 19506\u201319514","DOI":"10.1609\/aaai.v38i17.29922"},{"key":"2952_CR38","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. In: Proceedings of the 31st international conference on neural information processing systems, NeurIPS, 2017, pp 5998\u20136008"},{"key":"2952_CR39","doi-asserted-by":"crossref","unstructured":"Ma J, Huang Y, Wang L, Huang X, Peng H, Yu Z, Yu P (2024) Augmenting low-resource cross-lingual summarization with progression-grounded training and prompting. ACM Trans Asian Low-Resour Lang Inf Process 23(9):1-22","DOI":"10.1145\/3675167"},{"key":"2952_CR40","doi-asserted-by":"crossref","unstructured":"Yang Z, Peng H, Jiang Y, Li X, Du H, Wang S, Liu J (2025) Chathttpfuzz: large language model-assisted iot http fuzzing. International Journal of Machine Learning and Cybernetics 16:4577\u20134598","DOI":"10.1007\/s13042-024-02527-3"},{"issue":"10","key":"2952_CR41","doi-asserted-by":"publisher","first-page":"4456","DOI":"10.1007\/s11263-024-02097-5","volume":"132","author":"M Li","year":"2024","unstructured":"Li M, Zhou P, Liu J-W, Keppo J, Lin M, Yan S, Xu X (2024) Instant3d: instant text-to-3d generation. Int J Comput Vision 132(10):4456\u20134472","journal-title":"Int J Comput Vision"},{"key":"2952_CR42","doi-asserted-by":"publisher","first-page":"6297","DOI":"10.1109\/TMM.2023.3347849","volume":"26","author":"M Li","year":"2023","unstructured":"Li M, Fu H, He S, Fan H, Liu J, Keppo J, Shou MZ (2023) Dr-fer: discriminative and robust representation learning for facial expression recognition. IEEE Trans Multimed 26:6297\u20136309","journal-title":"IEEE Trans Multimed"},{"issue":"6","key":"2952_CR43","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1145\/3530811","volume":"55","author":"Y Tay","year":"2023","unstructured":"Tay Y, Dehghani M, Bahri D, Metzler D (2023) Efficient transformers: a survey. ACM Comput Surv 55(6):109\u2013110928","journal-title":"ACM Comput Surv"},{"key":"2952_CR44","unstructured":"Zhu X, Li J, Liu Y, Ma C, Wang W (2023) A survey on model compression for large language models. Arxiv:abs\/2308.07633"},{"key":"2952_CR45","unstructured":"Zhou Z, Ning X, Hong K, Fu T, Xu J, Li S, Lou Y, Wang L, Yuan Z, Li X, Yan S, Dai G, Zhang X, Dong Y, Wang Y (2024) A survey on efficient inference for large language models. Arxiv:abs\/2404.14294"},{"issue":"3","key":"2952_CR46","doi-asserted-by":"publisher","first-page":"628","DOI":"10.1109\/TC.2021.3057082","volume":"71","author":"H Peng","year":"2022","unstructured":"Peng H, Yang R, Wang Z, Li J, He L, Yu PS, Zomaya AY, Ranjan R (2022) Lime: low-cost and incremental learning for dynamic heterogeneous information networks. IEEE Trans Comput 71(3):628\u2013642","journal-title":"IEEE Trans Comput"},{"issue":"1","key":"2952_CR47","doi-asserted-by":"publisher","first-page":"1029","DOI":"10.1109\/TCE.2023.3323373","volume":"70","author":"C Li","year":"2023","unstructured":"Li C, Peng Y, Liu G, Li Y, Yang X, Chen C (2023) Efficient vision transformer for human-centric aiot applications through token tracking assignment. IEEE Trans Consum Electron 70(1):1029\u20131039","journal-title":"IEEE Trans Consum Electron"},{"key":"2952_CR48","doi-asserted-by":"crossref","unstructured":"Zou X, Chen C, Lin P, Zhang L, Xu Y, Zhang W (2024) Scalable heterogeneous scheduling based model parallelism for real-time inference of large-scale deep neural networks. IEEE Transactions on Emerging Topics in Computational Intelligence 8(4):2962-2973","DOI":"10.1109\/TETCI.2024.3369628"},{"key":"2952_CR49","unstructured":"Hou L, Huang Z, Shang L, Jiang X, Chen X, Liu Q (2020) Dynabert: dynamic BERT with adaptive width and depth. In: Proceedings of the 34th international conference on neural information processing systems, NeurIPS, 2020"},{"key":"2952_CR50","doi-asserted-by":"crossref","unstructured":"Hsieh C, Li C, Yeh C, Nakhost H, Fujii Y, Ratner A, Krishna R, Lee C, Pfister T (2023) Distilling step-by-step! outperforming larger language models with less training data and smaller model sizes. In: Findings of the association for computational linguistics, ACL, 2023, pp 8003\u20138017","DOI":"10.18653\/v1\/2023.findings-acl.507"},{"key":"2952_CR51","unstructured":"Liang C, Zuo S, Zhang Q, He P, Chen W, Zhao T (2023) Less is more: Task-aware layer-wise distillation for language model compression. In: Proceedings of the 40th international conference on machine learning, ICML, 2023, vol. 202, pp 20852\u201320867"},{"key":"2952_CR52","doi-asserted-by":"crossref","unstructured":"Magister LC, Mallinson J, Ad\u00e1mek J, Malmi E, Severyn A (2023) Teaching small language models to reason. In: Proceedings of the 61st annual meeting of the association for computational linguistics, ACL, 2023, pp 1773\u20131781","DOI":"10.18653\/v1\/2023.acl-short.151"},{"key":"2952_CR53","doi-asserted-by":"crossref","unstructured":"Wang L, Huang X, Yu Z, Peng H, Gao S, Mao C, Huang Y, Dong L, Philip SY (2024) Zero-shot text normalization via cross-lingual knowledge distillation. Speech, and Language Processing, IEEE\/ACM Transactions on Audio","DOI":"10.1109\/TASLP.2024.3407509"},{"key":"2952_CR54","unstructured":"Frantar E, Alistarh D (2023) Sparsegpt: Massive language models can be accurately pruned in one-shot. In: Proceedings of the 40th international conference on machine learning, ICML 2013, vol. 202, pp 10323\u201310337"},{"key":"2952_CR55","doi-asserted-by":"crossref","unstructured":"Xia M, Zhong Z, Chen D (2022) Structured pruning learns compact and accurate models. In: Proceedings of the 60th annual meeting of the association for computational linguistics, ACL, 2022, pp 1513\u20131528","DOI":"10.18653\/v1\/2022.acl-long.107"},{"issue":"8","key":"2952_CR56","first-page":"4035","volume":"44","author":"J Liu","year":"2022","unstructured":"Liu J, Zhuang B, Zhuang Z, Guo Y, Huang J, Zhu J, Tan M (2022) Discrimination-aware network pruning for deep model compression. IEEE Trans Pattern Anal Mach Intell 44(8):4035\u20134051","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"2952_CR57","doi-asserted-by":"crossref","unstructured":"Yao Z, Yazdani\u00a0Aminabadi R, Zhang M, Wu X, Li C, He Y (2022) Zeroquant: efficient and affordable post-training quantization for large-scale transformers. Proceedings of the 36th international conference on neural information processing systems, NeurIPS, 2022 35, 27168\u201327183","DOI":"10.52202\/068431-1970"},{"key":"2952_CR58","unstructured":"Xiao G, Lin J, Seznec M, Wu H, Demouth J, Han S (2023) Smoothquant: accurate and efficient post-training quantization for large language models. In: Proceedings of the 40th international conference on machine learning, ICML, 2013, pp 38087\u201338099"},{"key":"2952_CR59","doi-asserted-by":"crossref","unstructured":"Liu W, Zhou P, Wang Z, Zhao Z, Deng H, Ju Q (2020) Fastbert: a self-distilling BERT with adaptive inference time. In: Proceedings of the 58th annual meeting of the association for computational linguistics, ACL 2020, Online, July 5-10, 2020, pp 6035\u20136044","DOI":"10.18653\/v1\/2020.acl-main.537"},{"key":"2952_CR60","unstructured":"Geng S, Gao P, Fu Z, Zhang Y (2021) Romebert: robust training of multi-exit BERT. Arxiv:abs\/2101.09755"},{"key":"2952_CR61","doi-asserted-by":"crossref","unstructured":"Schwartz R, Stanovsky G, Swayamdipta S, Dodge J, Smith NA (2020) The right tool for the job: Matching model and instance complexities. In: Proceedings of the 58th annual meeting of the association for computational linguistics, ACL 2020, pp 6640\u20136651","DOI":"10.18653\/v1\/2020.acl-main.593"},{"key":"2952_CR62","doi-asserted-by":"crossref","unstructured":"Wang J, Chen K, Chen G, Shou L, McAuley JJ (2022) Skipbert: efficient inference with shallow layer skipping. In: Proceedings of the 60th annual meeting of the association for computational linguistics, ACL 2022, Dublin, Ireland, May 22-27, 2022, pp 7287\u20137301","DOI":"10.18653\/v1\/2022.acl-long.503"},{"key":"2952_CR63","unstructured":"Zhou W, Xu C, Ge T, McAuley J, Xu K, Wei F (2020) Bert loses patience: Fast and robust inference with early exit. Proceedings of the 34th international conference on neural information processing systems, NeurIPS 2020, 33, 18330\u201318341"},{"key":"2952_CR64","doi-asserted-by":"crossref","unstructured":"Zhang Z, Zhu W, Zhang J, Wang P, Jin R, Chung T-S (2022) Pcee-bert: accelerating bert inference via patient and confident early exiting. In: Findings of the association for computational linguistics, NAACL 2022, pp 327\u2013338","DOI":"10.18653\/v1\/2022.findings-naacl.25"},{"key":"2952_CR65","doi-asserted-by":"crossref","unstructured":"Xin J, Tang R, Yu Y, Lin J (2021) Berxit: Early exiting for bert with better fine-tuning and extension to regression. In: Proceedings of the 16th conference of the European chapter of the association for computational linguistics, EACL 2021, pp 91\u2013104","DOI":"10.18653\/v1\/2021.eacl-main.8"},{"key":"2952_CR66","doi-asserted-by":"crossref","unstructured":"Schuster T, Fisch A, Jaakkola TS, Barzilay R (2021) Consistent accelerated inference via confident adaptive transformers. In: Proceedings of the 2021 conference on empirical methods in natural language processing, EMNLP, 2021, pp 4962\u20134979","DOI":"10.18653\/v1\/2021.emnlp-main.406"},{"key":"2952_CR67","doi-asserted-by":"crossref","unstructured":"Bae S, Ko J, Song H, Yun S-Y (2023) Fast and robust early-exiting framework for autoregressive language models with synchronized parallel decoding. In: Proceedings of the 2023 conference on empirical methods in natural language processing, EMNLP, 2023, pp 5910\u20135924","DOI":"10.18653\/v1\/2023.emnlp-main.362"},{"key":"2952_CR68","unstructured":"Chen Y, Pan X, Li Y, Ding B, Zhou J (2024) EE-LLM: large-scale training and inference of early-exit large language models with 3d parallelism. In: Proceedings of the 41st international conference on machine learning, ICML, 2024"},{"key":"2952_CR69","unstructured":"Song J, Oh K, Kim T, Kim H, Kim Y, Kim J (2024) SLEB: streamlining llms through redundancy verification and elimination of transformer blocks. In: Proceedings of the 41st international conference on machine learning, ICML, 2024"},{"key":"2952_CR70","doi-asserted-by":"crossref","unstructured":"Fan S, Jiang X, Li X, Meng X, Han P, Shang S, Sun A, Wang Y, Wang Z (2024) Not all layers of llms are necessary during inference. Arxiv:abs\/2403.02181","DOI":"10.24963\/ijcai.2024\/566"},{"key":"2952_CR71","unstructured":"Kaya Y, Hong S, Dumitras T (2019) Shallow-deep networks: understanding and mitigating network overthinking. In: Proceedings of the 36th international conference on machine learning, ICML, 2019, vol. 97, pp 3301\u20133310"},{"key":"2952_CR72","unstructured":"Arpit D, Jastrzebski S, Ballas N, Krueger D, Bengio E, Kanwal MS, Maharaj T, Fischer A, Courville AC, Bengio Y, Lacoste-Julien S (2017) A closer look at memorization in deep networks. In: Proceedings of the 34th international conference on machine learning, ICML, 2017, vol. 70, pp 233\u2013242"},{"key":"2952_CR73","unstructured":"Toneva M, Sordoni A, Combes RT, Trischler A, Bengio Y, Gordon GJ (2019) An empirical study of example forgetting during deep neural network learning. In: Proceedings of the 7th international conference on learning representations, ICLR 2019, New Orleans, LA, USA, May 6-9, 2019"},{"key":"2952_CR74","unstructured":"G\u00fcl\u00e7ehre \u00c7, Paine TL, Shahriari B, Denil M, Hoffman M, Soyer H, Tanburn R, Kapturowski S, Rabinowitz NC, Williams D, Barth-Maron G, Wang Z, Freitas N, Team W (2020) Making efficient use of demonstrations to solve hard exploration problems. In: Proceedings of the 8th international conference on learning representations, ICLR, 2020"},{"key":"2952_CR75","unstructured":"Gu J, Bradbury J, Xiong C, Li VOK, Socher R (2018) Non-autoregressive neural machine translation. In: Proceedings of the 6th international conference on learning representations, ICLR, 2018"},{"key":"2952_CR76","doi-asserted-by":"crossref","unstructured":"Ghazvininejad M, Levy O, Liu Y, Zettlemoyer L (2019) Mask-predict: parallel decoding of conditional masked language models. In: Proceedings of the 2019 conference on empirical methods in natural language processing and the 9th international joint conference on natural language processing, EMNLP-IJCNLP 2019, pp 6111\u20136120","DOI":"10.18653\/v1\/D19-1633"},{"key":"2952_CR77","doi-asserted-by":"crossref","unstructured":"Gu J, Kong X (2021) Fully non-autoregressive neural machine translation: tricks of the trade. In: Findings of the association for computational linguistics, ACL, 2021, pp 120\u2013133","DOI":"10.18653\/v1\/2021.findings-acl.11"},{"key":"2952_CR78","doi-asserted-by":"crossref","unstructured":"Santilli A, Severino S, Postolache E, Maiorca V, Mancusi M, Marin R, Rodol\u00e0 E (2023) Accelerating transformer inference for translation via parallel decoding. In: Proceedings of the 61st annual meeting of the association for computational linguistics, ACL, 2023, pp 12336\u201312355","DOI":"10.18653\/v1\/2023.acl-long.689"},{"key":"2952_CR79","unstructured":"Wang A, Singh A, Michael J, Hill F, Levy O, Bowman SR (2019) GLUE: a multi-task benchmark and analysis platform for natural language understanding. In: Proceedings of the 7th international conference on learning representations, ICLR, 2019"},{"key":"2952_CR80","doi-asserted-by":"crossref","unstructured":"Al-Garadi MA, Yang Y-C, Sarker A (2022) The role of natural language processing during the covid-19 pandemic: health applications, opportunities, and challenges. In: Healthcare, vol. 10, p 2270","DOI":"10.3390\/healthcare10112270"},{"key":"2952_CR81","doi-asserted-by":"crossref","unstructured":"Xu J, Wang P, Tian G, Xu B, Zhao J, Wang F, Hao H (2015) Short text clustering via convolutional neural networks. In: Proceedings of the 1st workshop on vector space modeling for natural language processing, NAACL-HLT, 2015, pp 62\u201369","DOI":"10.3115\/v1\/W15-1509"},{"key":"2952_CR82","doi-asserted-by":"crossref","unstructured":"Ye D, Lin Y, Huang Y, Sun M (2021) TR-BERT: dynamic token reduction for accelerating BERT inference. In: Proceedings of the 2021 conference of the North American Chapter of the association for computational linguistics: human language technologies, NAACL-HLT, 2021, pp 5798\u20135809","DOI":"10.18653\/v1\/2021.naacl-main.463"},{"key":"2952_CR83","unstructured":"Taori R, Gulrajani I, Zhang T, Dubois Y, Li X, Guestrin C, Liang P, Hashimoto TB (2023) Alpaca: a strong, replicable instruction-following model. Stanford Center for Research on Foundation Models 3(6):7"},{"key":"2952_CR84","unstructured":"Conover M, Hayes M, Mathur A, Xie J, Wan J, Shah S, Ghodsi A, Wendell P, Zaharia M, Xin R (2023) Free dolly: Introducing the world\u2019s first truly open instruction-tuned llm. Company Blog of Databricks"},{"key":"2952_CR85","doi-asserted-by":"crossref","unstructured":"Narayan S, Cohen SB, Lapata M (2018) Don\u2019t give me the details, just the summary! topic-aware convolutional neural networks for extreme summarization. In: Proceedings of the 2018 Conference on empirical methods in natural language processing, EMNLP, 2018, pp 1797\u20131807","DOI":"10.18653\/v1\/D18-1206"},{"key":"2952_CR86","unstructured":"Hu EJ, Shen Y, Wallis P, Allen-Zhu Z, Li Y, Wang S, Wang L, Chen W (2022) Lora: Low-rank adaptation of large language models. In: Proceedings of the 10th international conference on learning representations, ICLR, 2022"},{"key":"2952_CR87","doi-asserted-by":"crossref","unstructured":"Zheng L, Chiang W, Sheng Y, Zhuang S, Wu Z, Zhuang Y, Lin Z, Li Z, Li D, Xing EP, Zhang H, Gonzalez JE, Stoica I (2023) Judging llm-as-a-judge with mt-bench and chatbot arena. In: NeurIPS proceedings of the 37th international conference on neural information processing systems, NeurIPS, 2023","DOI":"10.52202\/075280-2020"},{"key":"2952_CR88","doi-asserted-by":"crossref","unstructured":"Zhou Y, Lei T, Liu H, Du N, Huang Y, Zhao V, Dai AM, Le QV, Laudon J et al (2022) Mixture-of-experts with expert choice routing. Proceedings of the 36th international conference on neural information processing systems, NeurIPS 2022, 35, 7103\u20137114","DOI":"10.52202\/068431-0515"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-025-02952-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-025-02952-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-025-02952-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T10:02:27Z","timestamp":1774519347000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-025-02952-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,16]]},"references-count":88,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["2952"],"URL":"https:\/\/doi.org\/10.1007\/s13042-025-02952-y","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,16]]},"assertion":[{"value":"5 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"121"}}