{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T17:39:33Z","timestamp":1785605973753,"version":"3.56.0"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,12,27]],"date-time":"2022-12-27T00:00:00Z","timestamp":1672099200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,12,27]],"date-time":"2022-12-27T00:00:00Z","timestamp":1672099200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["New Gener. Comput."],"published-print":{"date-parts":[[2023,3]]},"DOI":"10.1007\/s00354-022-00198-8","type":"journal-article","created":{"date-parts":[[2022,12,27]],"date-time":"2022-12-27T06:02:41Z","timestamp":1672120961000},"page":"109-134","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Length-Based Curriculum Learning for Efficient Pre-training of Language Models"],"prefix":"10.1007","volume":"41","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0918-5294","authenticated-orcid":false,"given":"Koichi","family":"Nagatsuka","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Clifford","family":"Broni-Bediako","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masayasu","family":"Atsumi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,12,27]]},"reference":[{"key":"198_CR1","doi-asserted-by":"crossref","unstructured":"Howard, J., Ruder, S.: Universal language model fine-tuning for text classification. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics, Vol. 1: Long Papers, pp. 328\u2013339 (2018)","DOI":"10.18653\/v1\/P18-1031"},{"key":"198_CR2","first-page":"3079","volume":"28","author":"AM Dai","year":"2015","unstructured":"Dai, A.M., Le, Q.V.: Semi-supervised sequence learning. Adv. Neural Inf. Process. Syst. 28, 3079\u20133087 (2015)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"198_CR3","doi-asserted-by":"crossref","unstructured":"Peters, M.E., Neumann, M., Iyyer, M., Gardner, M., Clark, C., Lee, K., Zettlemoyer, L.: Deep contextualized word representations. In: Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol. 1 (Long Papers), pp. 2227\u20132237 (2018)","DOI":"10.18653\/v1\/N18-1202"},{"key":"198_CR4","unstructured":"Radford, A., Narasimhan, K., Salimans, T., Sutskever, I.: Improving language understanding by generative pre-training (2018)"},{"key":"198_CR5","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol. 1 (Long and Short Papers), pp. 4171\u20134186 (2019)"},{"key":"198_CR6","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., Stoyanov, V.: Roberta: a robustly optimized BERT pretraining approach (2019). arXiv preprint. arXiv:1907.11692"},{"key":"198_CR7","unstructured":"Lan, Z., Chen, M., Goodman, S., Gimpel, K., Sharma, P., Soricut, R.: ALBERT: a lite BERT for self-supervised learning of language representations (2020). arXiv:1909.11942 [cs]"},{"key":"198_CR8","unstructured":"Sanh, V., Debut, L., Chaumond, J., Wolf, T.: DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter (2019). arXiv preprint. arXiv:1910.01108"},{"key":"198_CR9","unstructured":"Clark, K., Luong, M.T., Le, Q.V., Manning, C.D.: Electra: pre-training text encoders as discriminators rather than generators (2020). arXiv preprint. arXiv:2003.10555"},{"key":"198_CR10","first-page":"415","volume":"30","author":"WL Taylor","year":"1953","unstructured":"Taylor, W.L.: \u201ccloze procedure\u2019\u2019: a new tool for measuring readability. Journal. Mass Commun. Q. 30, 415\u2013433 (1953)","journal-title":"Journal. Mass Commun. Q."},{"key":"198_CR11","doi-asserted-by":"crossref","unstructured":"Voita, E., Talbot, D., Moiseev, F., Sennrich, R., Titov, I.: Analyzing multi-head self-attention: specialized heads do the heavy lifting, the rest can be pruned. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 5797\u20135808 (2019)","DOI":"10.18653\/v1\/P19-1580"},{"key":"198_CR12","doi-asserted-by":"crossref","unstructured":"Sukhbaatar, S., Grave, E., Bojanowski, P., Joulin, A.: Adaptive attention span in transformers. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 331\u2013335 (2019)","DOI":"10.18653\/v1\/P19-1032"},{"key":"198_CR13","unstructured":"de\u00a0Wynter, A., Perry, D.J.: Optimal subarchitecture extraction for BERT (2020). CoRR. arXiv:2010.10499"},{"key":"198_CR14","unstructured":"Yang, Z., Dai, Z., Yang, Y., Carbonell, J., Salakhutdinov, R.R., Le, Q.V.: Xlnet: generalized autoregressive pretraining for language understanding. In: Advances in Neural Information Processing Systems, vol.\u00a032 (2019)"},{"issue":"1","key":"198_CR15","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1016\/0010-0277(93)90058-4","volume":"48","author":"JL Elman","year":"1993","unstructured":"Elman, J.L.: Learning and development in neural networks: the importance of starting small. Cognition 48(1), 71\u201399 (1993)","journal-title":"Cognition"},{"key":"198_CR16","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: Proceedings of the 26th Annual International Conference on Machine Learning, pp. 41\u201348 (2009)","DOI":"10.1145\/1553374.1553380"},{"key":"198_CR17","unstructured":"Moore, R.C., Lewis, W.: Intelligent selection of language model training data. In: Proceedings of the ACL 2010 Conference Short Papers, pp. 220\u2013224 (2010)"},{"key":"198_CR18","doi-asserted-by":"crossref","unstructured":"Gururangan, S., Marasovi\u0107, A., Swayamdipta, S., Lo, K., Beltagy, I., Downey, D., Smith, N.A.: Don\u2019t stop pretraining: adapt language models to domains and tasks (2020). arXiv preprint. arXiv:2004.10964","DOI":"10.18653\/v1\/2020.acl-main.740"},{"key":"198_CR19","doi-asserted-by":"crossref","unstructured":"Soviany, P., Ionescu, R.T., Rota, P., Sebe, N.: Curriculum learning: a survey (2021). CoRR. arXiv:2101.10382","DOI":"10.1007\/s11263-022-01611-x"},{"issue":"09","key":"198_CR20","doi-asserted-by":"crossref","first-page":"4555","DOI":"10.1109\/TPAMI.2021.3072422","volume":"44","author":"X Wang","year":"2022","unstructured":"Wang, X., Chen, Y., Zhu, W.: A survey on curriculum learning. IEEE Trans. Pattern Anal. Mach. Intell. 44(09), 4555\u20134576 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"198_CR21","doi-asserted-by":"crossref","unstructured":"Shi, M., Ferrari, V.: Weakly supervised object localization using size estimates. In: European Conference on Computer Vision, pp. 105\u2013121 (2016)","DOI":"10.1007\/978-3-319-46454-1_7"},{"key":"198_CR22","doi-asserted-by":"crossref","unstructured":"Ionescu, R.T., Alexe, B., Leordeanu, M., Popescu, M., Papadopoulos, D.P., Ferrari, V.: How hard can it be? Estimating the difficulty of visual search in an image. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2157\u20132166 (2016)","DOI":"10.1109\/CVPR.2016.237"},{"key":"198_CR23","unstructured":"Spitkovsky, V.I., Alshawi, H., Jurafsky, D.: From baby steps to leapfrog: how \u201cless is more\u201d in unsupervised dependency parsing. In: Human Language Technologies: The 2010 Annual Conference of the North American Chapter of the Association for Computational Linguistics, pp. 751\u2013759 (2010)"},{"key":"198_CR24","doi-asserted-by":"crossref","unstructured":"Nagatsuka, K., Broni-Bediako, C., Atsumi, M.: Pre-training a BERT with curriculum learning by increasing block-size of input text. In: Proceedings of the International Conference on Recent Advances in Natural Language Processing (RANLP 2021), pp. 989\u2013996 (2021)","DOI":"10.26615\/978-954-452-072-4_112"},{"key":"198_CR25","unstructured":"Hacohen, G., Weinshall, D.: On the power of curriculum learning in training deep networks. In: Proceedings of the 36th International Conference on Machine Learning, vol.\u00a097, pp. 2535\u20132544 (2019)"},{"key":"198_CR26","unstructured":"Merity, S., Xiong, C., Bradbury, J., Socher, R.: Pointer sentinel mixture models (2016). arXiv preprint. arXiv:1609.07843"},{"key":"198_CR27","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., Agarwal, S., Herbert-Voss, A., Krueger, G., Henighan, T., Child, R., Ramesh, A., Ziegler, D., Wu, J., Winter, C., Hesse, C., Chen, M., Sigler, E., Litwin, M., Gray, S., Chess, B., Clark, J., Berner, C., McCandlish, S., Radford, A., Sutskever, I., Amodei, D.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol.\u00a033, pp. 1877\u20131901 (2020)"},{"issue":"140","key":"198_CR28","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., Shazeer, N., Roberts, A., Lee, K., Narang, S., Matena, M., Zhou, Y., Li, W., Liu, P.J.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res 21(140), 1\u201367 (2020)","journal-title":"J. Mach. Learn. Res"},{"key":"198_CR29","unstructured":"Wu, X., Dyer, E., Neyshabur, B.: When do curricula work? In: International Conference on Learning Representations (2021)"},{"key":"198_CR30","unstructured":"Li, C., Zhang, M., He, Y.: Curriculum learning: a regularization method for efficient and stable billion-scale GPT model pre-training (2021). arXiv:2108.06084"},{"key":"198_CR31","doi-asserted-by":"crossref","unstructured":"Xu, B., Zhang, L., Mao, Z., Wang, Q., Xie, H., Zhang, Y.: Curriculum learning for natural language understanding. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 6095\u20136104 (2020)","DOI":"10.18653\/v1\/2020.acl-main.542"},{"key":"198_CR32","doi-asserted-by":"crossref","unstructured":"Kocmi, T., Bojar, O.: Curriculum learning and minibatch bucketing in neural machine translation, pp. 379\u2013386 (2017)","DOI":"10.26615\/978-954-452-049-6_050"},{"key":"198_CR33","doi-asserted-by":"crossref","unstructured":"Platanios, E.A., Stretcu, O., Neubig, G., Poczos, B., Mitchell, T.: Competence-based curriculum learning for neural machine translation. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol. 1 (Long and Short Papers), pp. 1162\u20131172 (2019)","DOI":"10.18653\/v1\/N19-1119"},{"key":"198_CR34","doi-asserted-by":"crossref","unstructured":"Rajeswar, S., Subramanian, S., Dutil, F., Pal, C., Courville, A.: Adversarial generation of natural language (2017). arXiv preprint. arXiv:1705.10929","DOI":"10.18653\/v1\/W17-2629"},{"key":"198_CR35","doi-asserted-by":"crossref","unstructured":"Tay, Y., Wang, S., Luu, A.T., Fu, J., Phan, M.C., Yuan, X., Rao, J., Hui, S.C., Zhang, A.: Simple and effective curriculum pointer-generator networks for reading comprehension over long narratives (2019). arXiv:1905.10847","DOI":"10.18653\/v1\/P19-1486"},{"issue":"8","key":"198_CR36","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"198_CR37","doi-asserted-by":"crossref","unstructured":"Penha, G., Hauff, C.: Curriculum learning strategies for IR: an empirical study on conversation response ranking (2019). arXiv:1912.08555","DOI":"10.1007\/978-3-030-45439-5_46"},{"key":"198_CR38","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (2019)"},{"issue":"2","key":"198_CR39","first-page":"313","volume":"19","author":"MP Marcus","year":"1993","unstructured":"Marcus, M.P., Santorini, B., Marcinkiewicz, M.A.: Building a large annotated corpus of English: the Penn Treebank. Comput. Linguist. 19(2), 313\u2013330 (1993)","journal-title":"Comput. Linguist."},{"key":"198_CR40","doi-asserted-by":"crossref","unstructured":"Wang, A., Singh, A., Michael, J., Hill, F., Levy, O., Bowman, S.: GLUE: a multi-task benchmark and analysis platform for natural language understanding. In: Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP, pp. 353\u2013355 (2018)","DOI":"10.18653\/v1\/W18-5446"},{"key":"198_CR41","unstructured":"Zhang, X., Kumar, G., Khayrallah, H., Murray, K., Gwinnup, J., Martindale, M.J., McNamee, P., Duh, K., Carpuat, M.: An empirical exploration of curriculum learning for neural machine translation (2018). arXiv preprint. arXiv:1811.00739"},{"key":"198_CR42","doi-asserted-by":"crossref","unstructured":"Zhang, X., Shapiro, P., Kumar, G., McNamee, P., Carpuat, M., Duh, K.: Curriculum learning for domain adaptation in neural machine translation. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol. 1 (Long and Short Papers), pp. 1903\u20131915 (2019)","DOI":"10.18653\/v1\/N19-1189"},{"key":"198_CR43","doi-asserted-by":"crossref","unstructured":"Rajpurkar, P., Jia, R., Liang, P.: Know what you don\u2019t know: unanswerable questions for squad. In: ACL (2018)","DOI":"10.18653\/v1\/P18-2124"},{"key":"198_CR44","doi-asserted-by":"crossref","unstructured":"Rajpurkar, P., Zhang, J., Lopyrev, K., Liang, P.: SQuAD: 100,000+ questions for machine comprehension of text. In: Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing, pp. 2383\u20132392 (2016)","DOI":"10.18653\/v1\/D16-1264"}],"container-title":["New Generation Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00354-022-00198-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00354-022-00198-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00354-022-00198-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,4,19]],"date-time":"2023-04-19T10:45:21Z","timestamp":1681901121000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00354-022-00198-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,27]]},"references-count":44,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2023,3]]}},"alternative-id":["198"],"URL":"https:\/\/doi.org\/10.1007\/s00354-022-00198-8","relation":{},"ISSN":["0288-3635","1882-7055"],"issn-type":[{"value":"0288-3635","type":"print"},{"value":"1882-7055","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,27]]},"assertion":[{"value":"12 March 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 December 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 December 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}