{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T15:43:57Z","timestamp":1787672637423,"version":"build-2736575974"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2023,7,1]],"date-time":"2023-07-01T00:00:00Z","timestamp":1688169600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,7,1]],"date-time":"2023-07-01T00:00:00Z","timestamp":1688169600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. Comput. Sci. Technol."],"published-print":{"date-parts":[[2023,7]]},"DOI":"10.1007\/s11390-021-1119-0","type":"journal-article","created":{"date-parts":[[2023,11,9]],"date-time":"2023-11-09T08:23:54Z","timestamp":1699518234000},"page":"853-866","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":16,"title":["Improving BERT Fine-Tuning via Self-Ensemble and Self-Distillation"],"prefix":"10.1007","volume":"38","author":[{"given":"Yi-Ge","family":"Xu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xi-Peng","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li-Gao","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuan-Jing","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,7,31]]},"reference":[{"key":"1119_CR1","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang M W, Lee K et al. BERT: Pre-training of deep bidirectional transformers for language understanding. In Proc. the 2019 Conference of the North American Chapter of the Association for Computational Linguistics (NAACL): Human Language Technologies, Jun. 2019, pp.4171\u20134186. https:\/\/doi.org\/10.18653\/v1\/N19-1423.","DOI":"10.18653\/v1\/N19-1423"},{"key":"1119_CR2","unstructured":"Yang Z L, Dai Z H, Yang Y M et al. XLNet: Generalized autoregressive pretraining for language understanding. In Proc. the 33rd International Conference on Neural Information Processing Systems (NIPS), Dec. 2019, Article No. 517."},{"key":"1119_CR3","unstructured":"Liu Y H, Ott M, Goyal N et al. RoBERTa: A robustly optimized BERT pretraining approach. arXiv: 1907.11692, 2019. https:\/\/arxiv.org\/abs\/1907.11692, Aug. 2023."},{"key":"1119_CR4","doi-asserted-by":"publisher","unstructured":"Rajpurkar P, Zhang J, Lopyrev K et al. SQuAD: 100,000+ questions for machine comprehension of text. In Proc. the 2016 Conference on Empirical Methods in Natural Language Processing (EMNLP), Nov. 2016, pp.2383\u20132392. https:\/\/doi.org\/10.18653\/v1\/D16-1264.","DOI":"10.18653\/v1\/D16-1264"},{"key":"1119_CR5","doi-asserted-by":"publisher","unstructured":"Bowman S R, Angeli G, Potts C et al. A large annotated corpus for learning natural language inference. In Proc. the 2015 EMNLP, Sept. 2015, pp.632\u2013642. https:\/\/doi.org\/10.18653\/v1\/D15-1075.","DOI":"10.18653\/v1\/D15-1075"},{"issue":"10","key":"1119_CR6","doi-asserted-by":"publisher","first-page":"1872","DOI":"10.1007\/s11431-020-1647-3","volume":"63","author":"XP Qiu","year":"2020","unstructured":"Qiu X P, Sun T X, Xu Y G et al. Pre-trained models for natural language processing: A survey. Science China Technological Sciences, 2020, 63(10): 1872\u20131897. https:\/\/doi.org\/10.1007\/s11431-020-1647-3.","journal-title":"Science China Technological Sciences"},{"key":"1119_CR7","doi-asserted-by":"publisher","unstructured":"Peters M E, Ruder S, Smith N A. To tune or not to tune? Adapting pretrained representations to diverse tasks. In Proc. the 4th Workshop on Representation Learning for NLP, Aug. 2019, pp.7\u201314. https:\/\/doi.org\/10.18653\/v1\/W19-4302.","DOI":"10.18653\/v1\/W19-4302"},{"key":"1119_CR8","unstructured":"Stickland A C, Murray I. BERT and PALs: Projected attention layers for efficient adaptation in multi-task learning. In Proc. the 36th International Conference on Machine Learning (ICML), Jun. 2019, pp.5986\u20135995."},{"key":"1119_CR9","unstructured":"Houlsby N, Giurgiu A, Jastrzebski S et al. Parameter-efficient transfer learning for NLP. In Proc. the 36th ICML, Jun. 2019, pp.2790\u20132799."},{"key":"1119_CR10","unstructured":"Dong L, Yang N, Wang W H et al. Unified language model pre-training for natural language understanding and generation. arXiv: 1905.03197, 2019. https:\/\/arxiv.org\/abs\/1905.03197, Aug. 2023."},{"key":"1119_CR11","unstructured":"Liu X D, He P C, Chen W Z et al. Multi-task deep neural networks for natural language understanding. arXiv: 1901.11504, 2019. https:\/\/arxiv.org\/pdf\/1901.11504.pdf, Aug. 2023."},{"key":"1119_CR12","unstructured":"Raffel C, Shazeer N, Roberts A et al. Exploring the limits of transfer learning with a unified text-to-text transformer. arXiv: 1910.10683, 2019. https:\/\/arxiv.org\/abs\/1910.10683, Aug. 2023."},{"key":"1119_CR13","doi-asserted-by":"publisher","unstructured":"Sun C, Qiu X P, Xu Y G et al. How to fine-tune BERT for text classification? In Proc. the 18th China National Conference on Chinese Computational Linguistics, Oct. 2019, pp.194\u2013206. https:\/\/doi.org\/10.1007\/978-3-030-32381-3_16.","DOI":"10.1007\/978-3-030-32381-3_16"},{"issue":"4","key":"1119_CR14","doi-asserted-by":"publisher","first-page":"455","DOI":"10.1007\/s10462-016-9535-1","volume":"49","author":"H Li","year":"2018","unstructured":"Li H, Wang X S, Ding S F. Research and development of neural network ensembles: A survey. Artificial Intelligence Review, 2018, 49(4): 455\u2013479. https:\/\/doi.org\/10.1007\/s10462-016-9535-1.","journal-title":"Artificial Intelligence Review"},{"issue":"4","key":"1119_CR15","doi-asserted-by":"publisher","first-page":"838","DOI":"10.1137\/0330046","volume":"30","author":"BT Polyak","year":"1992","unstructured":"Polyak B T, Juditsky A B. Acceleration of stochastic approximation by averaging. SIAM Journal on Control and Optimization, 1992, 30(4): 838\u2013855. https:\/\/doi.org\/10.1137\/0330046.","journal-title":"SIAM Journal on Control and Optimization"},{"key":"1119_CR16","unstructured":"Schaul T, Quan J, Antonoglou I et al. Prioritized experience replay. In Proc. the 4th International Conference on Learning Representations (ICLR), May 2016."},{"key":"1119_CR17","unstructured":"Hinton G, Vinyals O, Dean J. Distilling the knowledge in a neural network. arXiv: 1503.02531, 2015. https:\/\/arxiv.org\/abs\/1503.02531, Aug. 2023."},{"key":"1119_CR18","unstructured":"Laine S, Aila T. Temporal ensembling for semi-supervised learning. In Proc. the 5th ICLR, Apr. 2017."},{"key":"1119_CR19","unstructured":"Tarvainen A, Valpola H. Mean teachers are better role models: Weight-averaged consistency targets improve semi-supervised deep learning results. In Proc. the 31st NIPS, Dec. 2017, pp.1195\u20131204."},{"key":"1119_CR20","doi-asserted-by":"publisher","unstructured":"Wei H R, Huang S J, Wang R et al. Online distilling from checkpoints for neural machine translation. In Proc. the 2019 NAACL: Human Language Technologies, Jun. 2019, pp.1932\u20131941. https:\/\/doi.org\/10.18653\/v1\/N19-1192.","DOI":"10.18653\/v1\/N19-1192"},{"key":"1119_CR21","doi-asserted-by":"publisher","unstructured":"Liu W J, Zhou P, Wang Z R et al. FastBERT: A self-distilling BERT with adaptive inference time. In Proc. the 58th Annual Meeting of the Association for Computational Linguistics (ACL), Jul. 2020, pp.6035\u20136044. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.537.","DOI":"10.18653\/v1\/2020.acl-main.537"},{"key":"1119_CR22","doi-asserted-by":"publisher","unstructured":"Wang A, Singh A, Michael J et al. GLUE: A multi-task benchmark and analysis platform for natural language understanding. In Proc. the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP, Nov. 2018, pp.353\u2013355. https:\/\/doi.org\/10.18653\/v1\/W18-5446.","DOI":"10.18653\/v1\/W18-5446"},{"key":"1119_CR23","unstructured":"Vaswani A, Shazeer N, Parmar N et al. Attention is all you need. In Proc. the 31st NIPS, Dec. 2017, pp.5998\u20136008."},{"key":"1119_CR24","unstructured":"Sanh V, Debut L, Chaumond J et al. DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter. arXiv: 1910.01108, 2019. https:\/\/arxiv.org\/abs\/1910.01108, Aug. 2023."},{"key":"1119_CR25","doi-asserted-by":"publisher","unstructured":"Jiao X Q, Yin Y C, Shang L F et al. TinyBERT: Distilling BERT for natural language understanding. In Proc. the 2020 Findings of the Association for Computational Linguistics, Nov. 2020, pp.4163\u20134174. https:\/\/doi.org\/10.18653\/v1\/2020.findings-emnlp.372.","DOI":"10.18653\/v1\/2020.findings-emnlp.372"},{"key":"1119_CR26","doi-asserted-by":"publisher","unstructured":"Sun Z Q, Yu H K, Song X D et al. MobileBERT: A compact task-agnostic BERT for resource-limited devices. In Proc. the 58th ACL, Jul. 2020, pp.2158\u20132170. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.195.","DOI":"10.18653\/v1\/2020.acl-main.195"},{"key":"1119_CR27","unstructured":"Wang W H, Wei F R, Dong L et al. MINILM: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. In Proc. the 34th NIPS, Dec. 2020, Article No. 485."},{"key":"1119_CR28","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1016\/j.engappai.2022.105151","volume":"105","author":"MA Ganaie","year":"2022","unstructured":"Ganaie M A, Hu M H, Malik A K et al. Ensemble deep learning: A review. Eng. Appl. Artif. Intell., 2022, 105: 105\u2013151. https:\/\/doi.org\/10.1016\/j.engappai.2022.105151.","journal-title":"Eng. Appl. Artif. Intell."},{"key":"1119_CR29","unstructured":"Andrychowicz M, Wolski F, Ray A et al. Hindsight experience replay. In Proc. the 31st NIPS, Dec. 2017, pp.5055\u20135065."},{"key":"1119_CR30","unstructured":"Horgan D, Quan J, Budden D et al. Distributed prioritized experience replay. In Proc. the 6th ICLR, Apr. 30\u2013May 3, 2018."},{"key":"1119_CR31","doi-asserted-by":"publisher","unstructured":"Sun S Q, Cheng Y, Gan Z et al. Patient knowledge distillation for BERT model compression. In Proc. the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing, Nov. 2019, pp.4323\u20134332. https:\/\/doi.org\/10.18653\/v1\/D19-1441.","DOI":"10.18653\/v1\/D19-1441"},{"key":"1119_CR32","unstructured":"Liu X D, He P C, Chen W Z et al. Improving multi-task deep neural networks via knowledge distillation for natural language understanding. arXiv: 1904.09482, 2019. https:\/\/arxiv.org\/abs\/1904.09482, Aug. 2023."},{"key":"1119_CR33","doi-asserted-by":"publisher","first-page":"625","DOI":"10.1162\/tacl_a_00290","volume":"7","author":"A Warstadt","year":"2019","unstructured":"Warstadt A, Singh A, Bowman S R. Neural network acceptability judgments. Trans. Association for Computational Linguistics, 2019, 7: 625\u2013641. https:\/\/doi.org\/10.1162\/tacl_a_00290.","journal-title":"Trans. Association for Computational Linguistics"},{"key":"1119_CR34","unstructured":"Socher R, Perelygin A, Wu J et al. Recursive deep models for semantic compositionality over a sentiment Treebank. In Proc. EMNLP, Oct. 2013, pp.1631\u20131642."},{"key":"1119_CR35","unstructured":"Dolan W B, Brockett C. Automatically constructing a corpus of sentential paraphrases. In Proc. the 3rd International Workshop on Paraphrasing, Oct. 2005, pp.9\u201316."},{"key":"1119_CR36","doi-asserted-by":"crossref","unstructured":"Cer D, Diab M, Agirre E et al. SemEval-2017 task 1: Semantic textual similarity-multilingual and cross-lingual focused evaluation. arXiv: 1708.00055, 2017. https:\/\/arxiv.org\/abs\/1708.00055, Aug. 2023.","DOI":"10.18653\/v1\/S17-2001"},{"key":"1119_CR37","doi-asserted-by":"crossref","unstructured":"Williams A, Nangia N, Bowman S. A broad-coverage challenge corpus for sentence understanding through inference. In Proc. the 2018 NAACL: Human Language Technologies, Jun. 2018, pp.1112\u20131122. 10.18653\/v1\/N18-1101.","DOI":"10.18653\/v1\/N18-1101"},{"key":"1119_CR38","doi-asserted-by":"publisher","unstructured":"Matthews B W. Comparison of the predicted and observed secondary structure of T4 phage lysozyme. Biochimica et Biophysica Acta (BBA)-Protein Structure, 1975, 405(2): 442\u2013451. https:\/\/doi.org\/10.1016\/0005-2795(75)90109-9.","DOI":"10.1016\/0005-2795(75)90109-9"},{"key":"1119_CR39","unstructured":"Maas A L, Daly R E, Pham P T et al. Learning word vectors for sentiment analysis. In Proc. the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies, Jun. 2011, pp.142\u2013150."},{"key":"1119_CR40","unstructured":"Zhang X, Zhao J B, LeCun Y. Character-level convolutional networks for text classification. In Proc. the 28th NIPS, Dec. 2015, pp.649\u2013657."},{"key":"1119_CR41","unstructured":"Pilault J, Elhattami A, Pal C. Conditionally adaptive multi-task learning: Improving transfer learning in NLP using fewer parameters & less data. arXiv: 2009.09139, 2020. https:\/\/arxiv.org\/abs\/2009.09139, Aug. 2023."},{"key":"1119_CR42","doi-asserted-by":"publisher","unstructured":"Howard J, Ruder S. Universal language model fine-tuning for text classification. In Proc. the 56th ACL, Jul. 2018, pp.328\u2013339. https:\/\/doi.org\/10.18653\/v1\/P18-1031.","DOI":"10.18653\/v1\/P18-1031"}],"container-title":["Journal of Computer Science and Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-021-1119-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11390-021-1119-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-021-1119-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,9]],"date-time":"2023-11-09T08:36:36Z","timestamp":1699518996000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11390-021-1119-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7]]},"references-count":42,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2023,7]]}},"alternative-id":["1119"],"URL":"https:\/\/doi.org\/10.1007\/s11390-021-1119-0","relation":{},"ISSN":["1000-9000","1860-4749"],"issn-type":[{"value":"1000-9000","type":"print"},{"value":"1860-4749","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,7]]},"assertion":[{"value":"28 October 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 June 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 July 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}