{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T15:48:10Z","timestamp":1775144890832,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":20,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819512324","type":"print"},{"value":"9789819512331","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,8,19]],"date-time":"2025-08-19T00:00:00Z","timestamp":1755561600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,19]],"date-time":"2025-08-19T00:00:00Z","timestamp":1755561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-1233-1_7","type":"book-chapter","created":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T06:05:12Z","timestamp":1755842712000},"page":"73-84","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Robust and\u00a0Efficient Early Exit for\u00a0Large Language Models: Mitigating KV Cache Loss and\u00a0Enhancing Exit Stability"],"prefix":"10.1007","author":[{"given":"Long","family":"Meng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruiqing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weiqiao","family":"Shan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,19]]},"reference":[{"key":"7_CR1","unstructured":"Elbayad, M., Gu, J., Grave, E., Auli, M.: Depth-adaptive transformer. arXiv preprint arXiv:1910.10073 (2019)"},{"key":"7_CR2","doi-asserted-by":"crossref","unstructured":"Schuster, T., Fisch, A., Jaakkola, T., Barzilay R.: Consistent accelerated inference via confident adaptive transformers. arXiv preprint arXiv:2104.08803 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.406"},{"key":"7_CR3","first-page":"17456","volume":"35","author":"T Schuster","year":"2022","unstructured":"Schuster, T.: Confident adaptive language modeling. Adv. Neural Inf. Process. Syst. 35, 17456\u201317472 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Gao, X., Zhu, W., Gao, J., Yin, C.: F-PABEE: flexible-patience-based early exiting for single-label and multi-label text classification tasks. In: ICASSP 2023 - IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10095864"},{"key":"7_CR5","unstructured":"Ji, X., Tang, R., Lee, J., Yu, Y., Lin, J.: DeeBERT: dynamic early exiting for accelerating BERT inference. arXiv preprint arXiv:2004.12993 (2020)"},{"key":"7_CR6","doi-asserted-by":"crossref","unstructured":"Liao, K., Zhang, Y., Ren, X., Su, Q., Sun, X., He, B.: A global past-future early exit method for accelerating inference of pre-trained language models. In: 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT 2021), pp. 2013\u20132023 (2021)","DOI":"10.18653\/v1\/2021.naacl-main.162"},{"key":"7_CR7","doi-asserted-by":"crossref","unstructured":"Liu, Y., Meng, F., Zhou, J., Chen, Y., Xu, J.: Faster depth-adaptive transformers. In: AAAI Conference on Artificial Intelligence, vol. 35, no. 15, pp. 13424\u201313432 (2021)","DOI":"10.1609\/aaai.v35i15.17584"},{"key":"7_CR8","unstructured":"Shan, W., et al.: Early exit is a natural capability in transformer-based models: an empirical study on early exit without joint optimization. arXiv preprint arXiv:2412.01455 (2024)"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Bae, S., Ko, J., Song, H., Yun S.Y.: Fast and robust early-exiting framework for autoregressive language models with synchronized parallel decoding. arXiv preprint arXiv:2310.05424 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.362"},{"key":"7_CR10","doi-asserted-by":"crossref","unstructured":"Liu, J., Wang, Q., Wang, J., Cai X.: Speculative decoding via early-exiting for faster LLM inference with thompson sampling control mechanism. arXiv preprint arXiv:2406.03853 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.179"},{"key":"7_CR11","unstructured":"Del Corro, L., Del Giorno, A., Agarwal, S., Yu B., Awadallah A., Mukherjee S.: SkipDecode: autoregressive skip decoding with batching and caching for efficient LLM inference. arXiv preprint arXiv:2307.02628 (2023)"},{"key":"7_CR12","doi-asserted-by":"crossref","unstructured":"Varshney, N., Chatterjee, A., Parmar, M., Baral C.: Investigating acceleration of LLaMA inference by enabling intermediate layer decoding via instruction tuning with \u2018LITE\u2019. In: Findings of the Association for Computational Linguistics: NAACL 2024, pp. 3656\u20133677 (2024)","DOI":"10.18653\/v1\/2024.findings-naacl.232"},{"key":"7_CR13","unstructured":"Chen, Y., Pan, X., Li, Y., Ding B., Zhou J.: EE-LLM: large-scale training and inference of early-exit large language models with 3D parallelism. arXiv preprint arXiv:2312.04916 (2023)"},{"key":"7_CR14","doi-asserted-by":"crossref","unstructured":"Fan, S., et al.: Not all layers of LLMs are necessary during inference. arxiv preprint arXiv:2403.02181 (2024)","DOI":"10.24963\/ijcai.2024\/566"},{"key":"7_CR15","doi-asserted-by":"crossref","unstructured":"Xin, J., Tang, R., Yu, Y., Lin J.: BERxiT: early exiting for BERT with better fine-tuning and extension to regression. In: 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume (EACL), pp. 91\u2013104 (2021)","DOI":"10.18653\/v1\/2021.eacl-main.8"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Zheng, Y., et al.: LlamaFactory: unified efficient fine-tuning of 100+ language models. arXiv preprint arXiv:2403.13372 (2024)","DOI":"10.18653\/v1\/2024.acl-demos.38"},{"key":"7_CR17","doi-asserted-by":"publisher","unstructured":"Leo G., et al.: A framework for few-shot language model evaluation. Zenodo, v0.4.3 (2024). https:\/\/doi.org\/10.5281\/zenodo.12608602, https:\/\/zenodo.org\/records\/12608602","DOI":"10.5281\/zenodo.12608602"},{"key":"7_CR18","unstructured":"Cobbe, K., et al.: Training verifiers to solve math word problems. arXiv preprint arXiv:2110.14168 (2021)"},{"key":"7_CR19","unstructured":"Austin, J., et al.: Program synthesis with large language models. arXiv preprint arXiv:2108.07732 (2021)"},{"key":"7_CR20","doi-asserted-by":"crossref","unstructured":"Kwiatkowski, T., et al.: Natural questions: a benchmark for question answering research. Trans. Assoc. Comput. Linguist. 7, 453\u2013466 (2019)","DOI":"10.1162\/tacl_a_00276"}],"container-title":["Lecture Notes in Computer Science","Advances in Neural Networks \u2013 ISNN 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-1233-1_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,2]],"date-time":"2026-04-02T14:50:44Z","timestamp":1775141444000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-1233-1_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,19]]},"ISBN":["9789819512324","9789819512331"],"references-count":20,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-1233-1_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,8,19]]},"assertion":[{"value":"19 August 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ISNN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Zhangye","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"isnn2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conference.cs.cityu.edu.hk\/isnn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}