{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T22:56:39Z","timestamp":1742943399999,"version":"3.40.3"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031702389"},{"type":"electronic","value":"9783031702396"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-70239-6_19","type":"book-chapter","created":{"date-parts":[[2024,9,19]],"date-time":"2024-09-19T06:02:09Z","timestamp":1726725729000},"page":"271-284","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Adaptive Greedy Layer Pruning: Iterative Layer Pruning with Subsequent Model Repurposing"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8442-6652","authenticated-orcid":false,"given":"Tam\u00e1s","family":"Ficsor","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3845-4978","authenticated-orcid":false,"given":"G\u00e1bor","family":"Berend","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,20]]},"reference":[{"key":"19_CR1","unstructured":"Agirre, E., M\u2018arquez, L., Wicentowski, R. (eds.): Proceedings of the Fourth International Workshop on Semantic Evaluations (SemEval-2007). Association for Computational Linguistics, Prague, Czech Republic (2007)"},{"key":"19_CR2","unstructured":"Bentivogli, L., Magnini, B., Dagan, I., Dang, H.T., Giampiccolo, D.: The fifth PASCAL recognizing textual entailment challenge. In: Proceedings of the Second Text Analysis Conference, TAC 2009, Gaithersburg, Maryland, USA, November 16-17, 2009. NIST (2009). https:\/\/tac.nist.gov\/publications\/2009\/additional.papers\/RTE5_overview.proceedings.pdf"},{"key":"19_CR3","unstructured":"Cheng, H., Zhang, M., Shi, J.Q.: A survey on deep neural network pruning-taxonomy, comparison, analysis, and recommendations (2023). http:\/\/arxiv.org\/abs\/2308.06767"},{"key":"19_CR4","unstructured":"Clark, K., Luong, M.T., Le, Q.V., Manning, C.D.: ELECTRA: pre-training text encoders as discriminators rather than generators. In: International Conference on Learning Representations (2020). https:\/\/openreview.net\/forum?id=r1xMH1BtvB"},{"key":"19_CR5","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1007\/11736790_9","volume-title":"Machine Learning Challenges. Evaluating Predictive Uncertainty, Visual Object Classification, and Recognising Tectual Entailment","author":"I Dagan","year":"2006","unstructured":"Dagan, I., Glickman, O., Magnini, B.: The PASCAL recognising textual entailment challenge. In: Qui\u00f1onero-Candela, J., Dagan, I., Magnini, B., d\u2019Alch\u00e9-Buc, F. (eds.) MLCW 2005. LNCS (LNAI), vol. 3944, pp. 177\u2013190. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11736790_9"},{"key":"19_CR6","doi-asserted-by":"publisher","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4171\u20134186. Association for Computational Linguistics, Minneapolis, Minnesota (2019). https:\/\/doi.org\/10.18653\/v1\/N19-1423, https:\/\/aclanthology.org\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"19_CR7","unstructured":"Dolan, W.B., Brockett, C.: Automatically constructing a corpus of sentential paraphrases. In: Proceedings of the Third International Workshop on Paraphrasing (IWP2005) (2005). https:\/\/aclanthology.org\/I05-5002"},{"key":"19_CR8","unstructured":"Frantar, E., Ashkboos, S., Hoefler, T., Alistarh, D.: GPTQ: accurate post-training quantization for generative pre-trained transformers. CoRR abs\/2210.17323 (2022). http:\/\/dblp.uni-trier.de\/db\/journals\/corr\/corr2210.html#abs-2210-17323"},{"key":"19_CR9","doi-asserted-by":"crossref","unstructured":"Giampiccolo, D., Magnini, B., Dagan, I., Dolan, B.: The third PASCAL recognizing textual entailment challenge. In: Proceedings of the ACL-PASCAL Workshop on Textual Entailment and Paraphrasing, pp.\u00a01\u20139. Association for Computational Linguistics, Prague (2007). https:\/\/aclanthology.org\/W07-1401","DOI":"10.3115\/1654536.1654538"},{"key":"19_CR10","doi-asserted-by":"publisher","unstructured":"Gordon, M., Duh, K., Andrews, N.: Compressing BERT: studying the effects of weight pruning on transfer learning. In: Proceedings of the 5th Workshop on Representation Learning for NLP, pp. 143\u2013155. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.repl4nlp-1.18, https:\/\/aclanthology.org\/2020.repl4nlp-1.18","DOI":"10.18653\/v1\/2020.repl4nlp-1.18"},{"key":"19_CR11","unstructured":"He, P., Gao, J., Chen, W.: DeBERTav3: improving DeBERTa using ELECTRA-style pre-training with gradient-disentangled embedding sharing. In: The Eleventh International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=sE7-XhLxHA"},{"key":"19_CR12","doi-asserted-by":"publisher","unstructured":"H\u00e9der, M., et al.: The past, present and future of the ELKH cloud. Inform\u00e1ci\u00f3s T\u00e1rsadalom 22(2), 128 (2022). https:\/\/doi.org\/10.22503\/inftars.xxii.2022.2.8","DOI":"10.22503\/inftars.xxii.2022.2.8"},{"key":"19_CR13","doi-asserted-by":"publisher","unstructured":"Hu, B., Zhu, Y., Li, J., Tang, S.: SmartBERT: a promotion of dynamic early exiting mechanism for accelerating BERT inference. In: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence. IJCAI \u201923 (2023). https:\/\/doi.org\/10.24963\/ijcai.2023\/563","DOI":"10.24963\/ijcai.2023\/563"},{"key":"19_CR14","unstructured":"Kitaev, N., Kaiser, L., Levskaya, A.: Reformer: the efficient transformer. In: International Conference on Learning Representations (2020). https:\/\/openreview.net\/forum?id=rkgNKkHtvB"},{"key":"19_CR15","unstructured":"Lan, Z., Chen, M., Goodman, S., Gimpel, K., Sharma, P., Soricut, R.: ALBERT: a lite BERT for self-supervised learning of language representations. In: International Conference on Learning Representations (2020). https:\/\/openreview.net\/forum?id=H1eA7AEtvS"},{"key":"19_CR16","unstructured":"Lepikhin, D., et al.: GShard: scaling giant models with conditional computation and automatic sharding. In: International Conference on Learning Representations (2021). https:\/\/openreview.net\/forum?id=qrwe7XHTmYb"},{"key":"19_CR17","doi-asserted-by":"crossref","unstructured":"Liang, T., Glossner, J., Wang, L., Shi, S., Zhang, X.: Pruning and quantization for deep neural network acceleration: a survey. Neurocomputing 461, 370\u2013403 (2021). http:\/\/dblp.uni-trier.de\/db\/journals\/ijon\/ijon461.html#LiangGWSZ21","DOI":"10.1016\/j.neucom.2021.07.045"},{"key":"19_CR18","doi-asserted-by":"publisher","unstructured":"Lin, Z., Liu, J., Yang, Z., Hua, N., Roth, D.: Pruning redundant mappings in transformer models via spectral-normalized identity prior. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 719\u2013730. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.findings-emnlp.64, https:\/\/aclanthology.org\/2020.findings-emnlp.64","DOI":"10.18653\/v1\/2020.findings-emnlp.64"},{"key":"19_CR19","unstructured":"Liu, Y., et al.: RoBERTa: a robustly optimized BERT pretraining approach (2019)"},{"key":"19_CR20","unstructured":"Mehta, S., Koncel-Kedziorski, R., Rastegari, M., Hajishirzi, H.: DeFINE: deep factorized input token embeddings for neural sequence modeling. In: ICLR, OpenReview.net (2020). http:\/\/dblp.uni-trier.de\/db\/conf\/iclr\/iclr2020.html#MehtaKRH20"},{"key":"19_CR21","unstructured":"Mosbach, M., Andriushchenko, M., Klakow, D.: On the stability of fine-tuning BERT: misconceptions, explanations, and strong baselines. In: Proceedings of the International Conference on Learning Representations (2021). https:\/\/openreview.net\/forum?id=nzpLWnVAyah"},{"key":"19_CR22","doi-asserted-by":"crossref","unstructured":"Nakkiran, P., Kaplun, G., Bansal, Y., Yang, T., Barak, B., Sutskever, I.: Deep double descent: where bigger models and more data hurt. In: ICLR, OpenReview.net (2020). http:\/\/dblp.uni-trier.de\/db\/conf\/iclr\/iclr2020.html#NakkiranKBYBS20","DOI":"10.1088\/1742-5468\/ac3a74"},{"key":"19_CR23","doi-asserted-by":"publisher","first-page":"76","DOI":"10.1016\/j.patrec.2022.03.023","volume":"157","author":"D Peer","year":"2022","unstructured":"Peer, D., Stabinger, S., Engl, S., Rodr\u00edguez-S\u00e1nchez, A.: Greedy-layer pruning: speeding up transformer models for natural language processing. Pattern Recogn. Lett. 157, 76\u201382 (2022). https:\/\/doi.org\/10.1016\/j.patrec.2022.03.023","journal-title":"Pattern Recogn. Lett."},{"key":"19_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2022.101429","volume":"77","author":"H Sajjad","year":"2023","unstructured":"Sajjad, H., Dalvi, F., Durrani, N., Nakov, P.: On the effect of dropping layers of pre-trained transformer models. Comput. Speech Lang. 77, 101429 (2023). https:\/\/doi.org\/10.1016\/j.csl.2022.101429","journal-title":"Comput. Speech Lang."},{"key":"19_CR25","unstructured":"Sanh, V., Debut, L., Chaumond, J., Wolf, T.: DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. CoRR abs\/1910.01108 (2019). http:\/\/dblp.uni-trier.de\/db\/journals\/corr\/corr1910.html#abs-1910-01108"},{"key":"19_CR26","doi-asserted-by":"publisher","unstructured":"Schwartz, R., Stanovsky, G., Swayamdipta, S., Dodge, J., Smith, N.A.: The right tool for the job: matching model and instance complexities. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 6640\u20136651. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.593","DOI":"10.18653\/v1\/2020.acl-main.593"},{"key":"19_CR27","doi-asserted-by":"crossref","unstructured":"Socher, R., Perelygin, A., Wu, J., Chuang, J., Manning, C.D., Ng, A., Potts, C.: Recursive deep models for semantic compositionality over a sentiment treebank. In: Proceedings of EMNLP, pp. 1631\u20131642 (2013)","DOI":"10.18653\/v1\/D13-1170"},{"key":"19_CR28","unstructured":"Stanton, S.D., Izmailov, P., Kirichenko, P., Alemi, A.A., Wilson, A.G.: Does knowledge distillation really work? In: Beygelzimer, A., Dauphin, Y., Liang, P., Vaughan, J.W. (eds.) Advances in Neural Information Processing Systems (2021). https:\/\/openreview.net\/forum?id=7J-fKoXiReA"},{"key":"19_CR29","doi-asserted-by":"publisher","unstructured":"Strubell, E., Ganesh, A., McCallum, A.: Energy and policy considerations for deep learning in NLP. In: Korhonen, A., Traum, D., M\u00e0rquez, L. (eds.) Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 3645\u20133650. Association for Computational Linguistics, Florence, Italy (2019). https:\/\/doi.org\/10.18653\/v1\/P19-1355, https:\/\/aclanthology.org\/P19-1355","DOI":"10.18653\/v1\/P19-1355"},{"key":"19_CR30","doi-asserted-by":"publisher","unstructured":"Sun, S., Gan, Z., Fang, Y., Cheng, Y., Wang, S., Liu, J.: Contrastive distillation on intermediate representations for language model compression. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 498\u2013508. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.emnlp-main.36","DOI":"10.18653\/v1\/2020.emnlp-main.36"},{"key":"19_CR31","doi-asserted-by":"crossref","unstructured":"Tjong Kim\u00a0Sang, E.F., De\u00a0Meulder, F.: Introduction to the CoNLL-2003 shared task: Language-independent named entity recognition. In: Proceedings of the Seventh Conference on Natural Language Learning at HLT-NAACL 2003, pp. 142\u2013147 (2003). https:\/\/www.aclweb.org\/anthology\/W03-0419","DOI":"10.3115\/1119176.1119195"},{"key":"19_CR32","unstructured":"Turc, I., Chang, M.W., Lee, K., Toutanova, K.: Well-read students learn better: on the importance of pre-training compact models (2020). https:\/\/openreview.net\/forum?id=BJg7x1HFvB"},{"key":"19_CR33","doi-asserted-by":"publisher","unstructured":"Wang, A., Singh, A., Michael, J., Hill, F., Levy, O., Bowman, S.: GLUE: a multi-task benchmark and analysis platform for natural language understanding. In: Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP, pp. 353\u2013355. Association for Computational Linguistics, Brussels, Belgium (2018). https:\/\/doi.org\/10.18653\/v1\/W18-5446","DOI":"10.18653\/v1\/W18-5446"},{"key":"19_CR34","doi-asserted-by":"publisher","unstructured":"Wang, X., Weissweiler, L., Sch\u00fctze, H., Plank, B.: How to distill your BERT: an empirical study on the impact of weight initialisation and distillation objectives. In: Rogers, A., Boyd-Graber, J., Okazaki, N. (eds.) Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers), pp. 1843\u20131852. Association for Computational Linguistics, Toronto, Canada (2023). https:\/\/doi.org\/10.18653\/v1\/2023.acl-short.157","DOI":"10.18653\/v1\/2023.acl-short.157"},{"key":"19_CR35","doi-asserted-by":"publisher","unstructured":"Warstadt, A., Singh, A., Bowman, S.R.: Neural network acceptability judgments. Trans. Assoc. Comput. Linguit. 7, 625\u2013641 (2019). https:\/\/doi.org\/10.1162\/tacl_a_00290, https:\/\/aclanthology.org\/Q19-1040","DOI":"10.1162\/tacl_a_00290"},{"key":"19_CR36","doi-asserted-by":"publisher","unstructured":"Xin, J., Tang, R., Lee, J., Yu, Y., Lin, J.: DeeBERT: dynamic early exiting for accelerating BERT inference. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 2246\u20132251. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.204","DOI":"10.18653\/v1\/2020.acl-main.204"},{"key":"19_CR37","doi-asserted-by":"publisher","unstructured":"Yao, Y., Huang, S., Wang, W., Dong, L., Wei, F.: Adapt-and-distill: developing small, fast and effective pretrained language models for domains. In: Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021, pp. 460\u2013470. Association for Computational Linguistics, Online (2021). https:\/\/doi.org\/10.18653\/v1\/2021.findings-acl.40","DOI":"10.18653\/v1\/2021.findings-acl.40"},{"key":"19_CR38","unstructured":"Zhang, T., Wu, F., Katiyar, A., Weinberger, K.Q., Artzi, Y.: Revisiting few-sample BERT fine-tuning. CoRR abs\/2006.05987 (2020). http:\/\/dblp.uni-trier.de\/db\/journals\/corr\/corr2006.html#abs-2006-05987"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Information Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-70239-6_19","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T10:23:52Z","timestamp":1732789432000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-70239-6_19"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031702389","9783031702396"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-70239-6_19","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"20 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLDB","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Applications of Natural Language to Information Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Turin","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 June 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 June 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nldb2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/nldb2024.di.unito.it\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}