{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:14Z","timestamp":1779228374684,"version":"3.51.4"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101969","type":"journal-article","created":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T08:10:19Z","timestamp":1772093419000},"page":"101969","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Improve NNLMs by text generation from pre-trained language models"],"prefix":"10.1016","volume":"100","author":[{"given":"Minguang","family":"Song","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5511-3692","authenticated-orcid":false,"given":"Yunxin","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101969_b1","series-title":"Cynical selection of language model training data","author":"Axelrod","year":"2017"},{"key":"10.1016\/j.csl.2026.101969_b2","series-title":"Layer normalization","author":"Ba","year":"2016"},{"key":"10.1016\/j.csl.2026.101969_b3","first-page":"1137","article-title":"A neural probabilistic language model","volume":"3","author":"Bengio","year":"2003","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.csl.2026.101969_b4","series-title":"Language models are few-shot learners","author":"Brown","year":"2020"},{"key":"10.1016\/j.csl.2026.101969_b5","article-title":"The AMI meeting corpus: A pre-announcement","author":"Carletta","year":"2005","journal-title":"MLMI Int. Workshop"},{"key":"10.1016\/j.csl.2026.101969_b6","series-title":"NeurIPS","first-page":"11011","article-title":"GroupReduce: Block-wise low-rank approximation for neural language model shrinking","volume":"vol. 31","author":"Chen","year":"2018"},{"key":"10.1016\/j.csl.2026.101969_b7","series-title":"NAACL","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b8","series-title":"EMNLP","first-page":"9522","article-title":"Automatic document selection for efficient encoder pretraining","author":"Feng","year":"2022"},{"key":"10.1016\/j.csl.2026.101969_b9","series-title":"ASRU","first-page":"428","article-title":"Using web text to improve keyword spotting in speech","author":"Gandhe","year":"2013"},{"key":"10.1016\/j.csl.2026.101969_b10","unstructured":"Hajimolahoseini, H., Rezagholizadeh, M., Partovinia, V., Tahaei, M., Awad, O.M., Liu, Y., 2021. Compressing pre-trained language models using progressive low rank decomposition. In: NeurIPS Workshop."},{"key":"10.1016\/j.csl.2026.101969_b11","unstructured":"Hinton, G., Vinyals, O., Dean, J., 2014. Distilling the Knowledge in a Neural Network. In: NIPS Workshop."},{"issue":"8","key":"10.1016\/j.csl.2026.101969_b12","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Comput."},{"key":"10.1016\/j.csl.2026.101969_b13","series-title":"ICLR","article-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2022"},{"key":"10.1016\/j.csl.2026.101969_b14","series-title":"DistilGPT2","author":"HuggingFace","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b15","series-title":"INTERSPEECH","first-page":"3905","article-title":"Language modeling with deep transformers","author":"Irie","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b16","series-title":"ACL","first-page":"497","article-title":"Low-rank RNN adaptation for context-aware language modeling","author":"Jaech","year":"2018"},{"key":"10.1016\/j.csl.2026.101969_b17","series-title":"ICASSP","first-page":"1695","article-title":"Selecting articles from the language model training corpus","author":"Klakow","year":"2000"},{"key":"10.1016\/j.csl.2026.101969_b18","series-title":"ICLR","article-title":"ALBERT: A lite BERT for self-supervised learning of language representations","author":"Lan","year":"2020"},{"issue":"9","key":"10.1016\/j.csl.2026.101969_b19","doi-asserted-by":"crossref","first-page":"1","DOI":"10.59782\/aai.v1i1.276","article-title":"Pre-trained language models for text generation: A survey","volume":"56","author":"Li","year":"2024","journal-title":"AAAI"},{"key":"10.1016\/j.csl.2026.101969_b20","series-title":"ICML","first-page":"20336","article-title":"LoSparse: Structured compression of large language models based on low-rank and sparse approximation","author":"Li","year":"2023"},{"key":"10.1016\/j.csl.2026.101969_b21","series-title":"INTERSPEECH","first-page":"829","article-title":"Improving speech recognition and keyword search for low resource languages using web data","author":"Mendels","year":"2015"},{"key":"10.1016\/j.csl.2026.101969_b22","series-title":"Interspeech","first-page":"1045","article-title":"Recurrent neural network based language model","author":"Mikolov","year":"2010"},{"key":"10.1016\/j.csl.2026.101969_b23","series-title":"ACL","first-page":"220","article-title":"Intelligent selection of language model training data","author":"Moore","year":"2010"},{"key":"10.1016\/j.csl.2026.101969_b24","series-title":"INTERSPEECH","first-page":"4926","article-title":"Language model data augmentation based on text domain transfer","author":"Ogawa","year":"2020"},{"key":"10.1016\/j.csl.2026.101969_b25","series-title":"NeurIPS","first-page":"8024","article-title":"PyTorch: An imperative style, high-performance deep learning library","author":"Paszke","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b26","doi-asserted-by":"crossref","unstructured":"Paul, D.B., Baker, J.M., 1992. The Design for the Wall Street Journal-based CSR Corpus. In: DARPA Speech and Language Workshop. HLT \u201991, pp. 357\u2013362.","DOI":"10.3115\/1075527.1075614"},{"key":"10.1016\/j.csl.2026.101969_b27","series-title":"ASRU","article-title":"The Kaldi speech recognition toolkit","author":"Povey","year":"2011"},{"key":"10.1016\/j.csl.2026.101969_b28","series-title":"Interspeech","first-page":"2751","article-title":"Purely sequence-trained neural networks for ASR based on lattice-free MMI","author":"Povey","year":"2016"},{"key":"10.1016\/j.csl.2026.101969_b29","series-title":"EACL","first-page":"157","article-title":"Using the output embedding to improve language models","author":"Press","year":"2017"},{"key":"10.1016\/j.csl.2026.101969_b30","series-title":"Qwen3 technical report","author":"Qwen Team","year":"2025"},{"key":"10.1016\/j.csl.2026.101969_b31","series-title":"OpenAI Blog","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"10.1016\/j.csl.2026.101969_b32","series-title":"OpenAI Blog","article-title":"Language models are unsupervised multitask learners","author":"Radford","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b33","series-title":"ICASSP","first-page":"6655","article-title":"Low-rank matrix factorization for deep neural network training with high-dimensional output targets","author":"Sainath","year":"2013"},{"key":"10.1016\/j.csl.2026.101969_b34","series-title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","author":"Sanh","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b35","series-title":"ACML","first-page":"1081","article-title":"Effective sentence scoring method using BERT for speech recognition","author":"Shin","year":"2019"},{"issue":"3","key":"10.1016\/j.csl.2026.101969_b36","doi-asserted-by":"crossref","first-page":"517","DOI":"10.1109\/TASLP.2015.2400218","article-title":"From feedforward to recurrent LSTM neural networks for language modeling","volume":"23","author":"Sundermeyer","year":"2015","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101969_b37","series-title":"ICASSP","first-page":"7245","article-title":"Improvements to N-gram language model using text generated from neural language model","author":"Suzuki","year":"2019"},{"issue":"4","key":"10.1016\/j.csl.2026.101969_b38","first-page":"581","article-title":"Morphology aware data augmentation with neural language models for online hybrid ASR","volume":"69","author":"Tarj\u00e1n","year":"2022","journal-title":"Acta Linguist. Acad."},{"key":"10.1016\/j.csl.2026.101969_b39","series-title":"Deep transformer based data augmentation with subword units for morphologically rich online ASR","author":"Tarj\u00e1n","year":"2020"},{"key":"10.1016\/j.csl.2026.101969_b40","series-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.csl.2026.101969_b41","series-title":"NIPS","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.csl.2026.101969_b42","series-title":"Improving N-gram language models with pre-trained deep transformer","author":"Wang","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b43","series-title":"Linformer: Self-attention with linear complexity","author":"Wang","year":"2020"},{"key":"10.1016\/j.csl.2026.101969_b44","series-title":"EMNLP","first-page":"38","article-title":"Transformers: State-of-the-art natural language processing","author":"Wolf","year":"2020"},{"key":"10.1016\/j.csl.2026.101969_b45","series-title":"NeurIPS","first-page":"34201","article-title":"Data selection for language models via importance resampling","author":"Xie","year":"2023"},{"key":"10.1016\/j.csl.2026.101969_b46","series-title":"NeurIPS","first-page":"5753","article-title":"XLNet: Generalized autoregressive pretraining for language understanding","volume":"vol. 32","author":"Yang","year":"2019"},{"key":"10.1016\/j.csl.2026.101969_b47","series-title":"ICLR","article-title":"BERTScore: Evaluating text generation with BERT","author":"Zhang","year":"2020"},{"key":"10.1016\/j.csl.2026.101969_b48","series-title":"ASRU","first-page":"162","article-title":"Adapting GPT, GPT-2 and BERT language models for speech recognition","author":"Zheng","year":"2021"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S088523082600032X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S088523082600032X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:12:51Z","timestamp":1779225171000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S088523082600032X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":48,"alternative-id":["S088523082600032X"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101969","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Improve NNLMs by text generation from pre-trained language models","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101969","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101969"}}