{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T02:15:40Z","timestamp":1783131340082,"version":"3.54.6"},"reference-count":58,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["12274221"],"award-info":[{"award-number":["12274221"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["2024CSJGG1100"],"award-info":[{"award-number":["2024CSJGG1100"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Speech Communication"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.specom.2026.103428","type":"journal-article","created":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T00:08:43Z","timestamp":1780445323000},"page":"103428","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["DDSE: Efficient Neural Codec Language Models for speech enhancement with disentangled representations"],"prefix":"10.1016","volume":"182","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0087-7200","authenticated-orcid":false,"given":"Qinwen","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaobin","family":"Rong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mansur","family":"Yesilbursa","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kamil","family":"Wojcicki","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tong","family":"Lei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jing","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.specom.2026.103428_b1","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)","first-page":"11789","article-title":"Robust and unbounded length generalization in autoregressive transformer-based text-to-speech","author":"Battenberg","year":"2025"},{"key":"10.1016\/j.specom.2026.103428_b2","doi-asserted-by":"crossref","unstructured":"Bie, X., Liu, X., Richard, G., 2025. Learning Source Disentanglement in Neural Audio Codec. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP.","DOI":"10.1109\/ICASSP49660.2025.10888065"},{"key":"10.1016\/j.specom.2026.103428_b3","doi-asserted-by":"crossref","first-page":"2523","DOI":"10.1109\/TASLP.2023.3288409","article-title":"AudioLM: A language modeling approach to audio generation","volume":"31","author":"Borsos","year":"2023","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Proc."},{"key":"10.1016\/j.specom.2026.103428_b4","unstructured":"Botinhao, C.V., Wang, X., Takaki, S., Yamagishi, J., 2016. Investigating RNN-based speech enhancement methods for noise-robust text-to-speech. In: 9th ISCA Speech Synthesis Workshop. pp. 159\u2013165."},{"issue":"6","key":"10.1016\/j.specom.2026.103428_b5","doi-asserted-by":"crossref","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","article-title":"WavLM: Large-scale self-supervised pre-training for full stack speech processing","volume":"16","author":"Chen","year":"2022","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103428_b6","doi-asserted-by":"crossref","first-page":"705","DOI":"10.1109\/TASLPRO.2025.3530270","article-title":"Neural codec language models are zero-shot text to speech synthesizers","volume":"33","author":"Chen","year":"2025","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Proc."},{"key":"10.1016\/j.specom.2026.103428_b7","series-title":"Interspeech 2021","first-page":"2426","article-title":"Unsupervised cross-lingual representation learning for speech recognition","author":"Conneau","year":"2021"},{"key":"10.1016\/j.specom.2026.103428_b8","doi-asserted-by":"crossref","first-page":"47704","DOI":"10.52202\/075280-2066","article-title":"Simple and controllable music generation","volume":"36","author":"Copet","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103428_b9","series-title":"The entropy mechanism of reinforcement learning for reasoning language models","author":"Cui","year":"2025"},{"key":"10.1016\/j.specom.2026.103428_b10","article-title":"High fidelity neural audio compression","author":"D\u00e9fossez","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.specom.2026.103428_b11","series-title":"Moshi: A speech-text foundation model for real-time dialogue","author":"D\u00e9fossez","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b12","series-title":"Interspeech 2020","first-page":"3830","article-title":"ECAPA-TDNN: Emphasized channel attention, propagation and aggregation in TDNN based speaker verification","author":"Desplanques","year":"2020"},{"key":"10.1016\/j.specom.2026.103428_b13","series-title":"Findings of the Association for Computational Linguistics: ACL 2025","first-page":"16563","article-title":"Mitigating hallucination in multimodal large language model via hallucination-targeted direct preference optimization","author":"Fu","year":"2025"},{"key":"10.1016\/j.specom.2026.103428_b14","series-title":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6633","article-title":"Fullsubnet: A full-band and sub-band fusion model for real-time single-channel speech enhancement","author":"Hao","year":"2021"},{"issue":"1","key":"10.1016\/j.specom.2026.103428_b15","doi-asserted-by":"crossref","first-page":"13","DOI":"10.1186\/s13636-015-0054-9","article-title":"Visqol: An objective speech quality model","volume":"2015","author":"Hines","year":"2015","journal-title":"EURASIP J. Audio, Speech, Music. Process."},{"key":"10.1016\/j.specom.2026.103428_b16","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Proc."},{"issue":"2","key":"10.1016\/j.specom.2026.103428_b17","doi-asserted-by":"crossref","DOI":"10.1145\/3703155","article-title":"A survey on hallucination in large language models: Principles, taxonomy, challenges, and open questions","volume":"43","author":"Huang","year":"2025","journal-title":"ACM Trans. Inf. Syst."},{"key":"10.1016\/j.specom.2026.103428_b18","series-title":"NaturalSpeech 3: Zero-shot speech synthesis with factorized codec and diffusion models","author":"Ju","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b19","series-title":"LLaSE-G1: Incentivizing generalization capability for llama-based speech enhancement","author":"Kang","year":"2025"},{"key":"10.1016\/j.specom.2026.103428_b20","series-title":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5220","article-title":"A study on data augmentation of reverberant speech for robust speech recognition","author":"Ko","year":"2017"},{"key":"10.1016\/j.specom.2026.103428_b21","series-title":"Interspeech 2019","first-page":"321","article-title":"Building the singapore english national speech corpus","author":"Koh","year":"2019"},{"key":"10.1016\/j.specom.2026.103428_b22","series-title":"Interspeech 2024","first-page":"4144","article-title":"Understanding sounds, missing the questions: The challenge of object hallucination in large audio-language models","author":"Kuan","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b23","series-title":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Can large audio-language models truly hear? Tackling hallucinations with multi-task assessment and stepwise audio reasoning","author":"Kuan","year":"2025"},{"key":"10.1016\/j.specom.2026.103428_b24","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Using RLHF to align speech enhancement approaches to mean-opinion quality scores","author":"Kumar","year":"2025"},{"key":"10.1016\/j.specom.2026.103428_b25","doi-asserted-by":"crossref","unstructured":"Kumar, R., Seetharaman, P., Luebs, A., Kumar, I., Kumar, K., 2023. High-Fidelity Audio Compression with Improved RVQGAN. In: Thirty-Seventh Conference on Neural Information Processing Systems. Vol. 36, New Orleans, LA, USA, pp. 27980\u201327993, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/58d0e78cf042af5876e12661087bea12-Paper-Conference.pdf.","DOI":"10.52202\/075280-1214"},{"key":"10.1016\/j.specom.2026.103428_b26","series-title":"Interspeech 2021","first-page":"2811","article-title":"DPCRN: Dual-path convolution recurrent network for single channel speech enhancement","author":"Le","year":"2021"},{"key":"10.1016\/j.specom.2026.103428_b27","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1441","article-title":"Crowdsourced multilingual speech intelligibility testing","author":"Lechler","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b28","doi-asserted-by":"crossref","first-page":"2724","DOI":"10.1109\/TASLP.2023.3294692","article-title":"StoRM: A diffusion-based stochastic regeneration model for speech enhancement and dereverberation","volume":"31","author":"Lemercier","year":"2023","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Proc."},{"issue":"8","key":"10.1016\/j.specom.2026.103428_b29","doi-asserted-by":"crossref","first-page":"1448","DOI":"10.1109\/JSTSP.2024.3506286","article-title":"SemantiCodec: An ultra low bitrate semantic audio codec for general sound","volume":"18","author":"Liu","year":"2024","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"10.1016\/j.specom.2026.103428_b30","series-title":"ICML 2024 Workshop on Models of Human Feedback for AI Alignment","article-title":"Filtered direct preference optimization","author":"Morimura","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b31","doi-asserted-by":"crossref","unstructured":"Omran, A., Zeghidour, N., Borsos, Z., de Chaumont Quitry, F., Slaney, M., Tagliasacchi, M., 2023. Disentangling speech from surroundings with neural embeddings. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, Rhodes, Greece, pp. 1\u20135.","DOI":"10.1109\/ICASSP49357.2023.10096435"},{"key":"10.1016\/j.specom.2026.103428_b32","series-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5206","article-title":"Librispeech: An asr corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"key":"10.1016\/j.specom.2026.103428_b33","doi-asserted-by":"crossref","unstructured":"Perez, E., Strub, F., De Vries, H., Dumoulin, V., Courville, A., 2018. Film: Visual reasoning with a general conditioning layer. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 32, New Orleans, USA.","DOI":"10.1609\/aaai.v32i1.11671"},{"key":"10.1016\/j.specom.2026.103428_b34","series-title":"Speech Communication; 15th ITG Conference","first-page":"265","article-title":"Evaluation metrics for generative speech enhancement methods: Issues and perspectives","author":"Pirklbauer","year":"2023"},{"key":"10.1016\/j.specom.2026.103428_b35","series-title":"International Conference on Learning Representations","article-title":"Train short, test long: Attention with linear biases enables input length extrapolation","author":"Press","year":"2022"},{"key":"10.1016\/j.specom.2026.103428_b36","series-title":"International Conference on Machine Learning","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2023"},{"key":"10.1016\/j.specom.2026.103428_b37","doi-asserted-by":"crossref","first-page":"53728","DOI":"10.52202\/075280-2338","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume":"36","author":"Rafailov","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103428_b38","series-title":"ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"886","article-title":"DNSMOS p. 835: A non-intrusive perceptual objective speech quality metric to evaluate noise suppressors","author":"Reddy","year":"2022"},{"key":"10.1016\/j.specom.2026.103428_b39","series-title":"The interspeech 2020 deep noise suppression challenge: Datasets, subjective testing framework, and challenge results","author":"Reddy","year":"2020"},{"key":"10.1016\/j.specom.2026.103428_b40","doi-asserted-by":"crossref","unstructured":"Saeki, T., Xin, D., Nakata, W., Koriyama, T., Takamichi, S., Saruwatari, H., 2022. UTMOS: UTokyo-SaruLab System for VoiceMOS Challenge 2022. In: Interspeech 2022. Incheon, Korea, pp. 4521\u20134525. http:\/\/dx.doi.org\/10.21437\/Interspeech.2022-439.","DOI":"10.21437\/Interspeech.2022-439"},{"key":"10.1016\/j.specom.2026.103428_b41","series-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"10.1016\/j.specom.2026.103428_b42","first-page":"36","article-title":"Method for the subjective assessment of intermediate quality level of audio systems","volume":"2","author":"Series","year":"2014","journal-title":"Int. Telecommun. Union Radiocommun. Assem."},{"key":"10.1016\/j.specom.2026.103428_b43","series-title":"TSELM: Target speaker extraction using discrete tokens and language models","author":"Tang","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b44","series-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.specom.2026.103428_b45","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.specom.2026.103428_b46","series-title":"ICASSP 2024 - 2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"11561","article-title":"SELM: Speech enhancement using discrete tokens and language models","author":"Wang","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b47","series-title":"Interspeech 2022","first-page":"2928","article-title":"Speech enhancement with score-based generative models in the complex STFT domain","author":"Welker","year":"2022"},{"key":"10.1016\/j.specom.2026.103428_b48","series-title":"ICASSP 2026-2026 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"21987","article-title":"Low-resource audio codec (lrac): 2025 challenge description","author":"Wojcicki","year":"2026"},{"key":"10.1016\/j.specom.2026.103428_b49","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024","first-page":"13258","article-title":"V-DPO: Mitigating hallucination in large vision language models via vision-guided direct preference optimization","author":"Xie","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b50","series-title":"Simple and effective zero-shot cross-lingual phoneme recognition","author":"Xu","year":"2021"},{"key":"10.1016\/j.specom.2026.103428_b51","series-title":"Interspeech 2024","first-page":"1170","article-title":"Genhancer: High-fidelity speech enhancement via generative modeling on discrete codec tokens","author":"Yang","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b52","series-title":"UniAudio: An audio foundation model toward universal audio generation","author":"Yang","year":"2023"},{"key":"10.1016\/j.specom.2026.103428_b53","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"GenSE: Generative speech enhancement via language models using hierarchical modeling","author":"Yao","year":"2025"},{"issue":"5","key":"10.1016\/j.specom.2026.103428_b54","doi-asserted-by":"crossref","first-page":"272","DOI":"10.3390\/a18050272","article-title":"Speech enhancement algorithms: A systematic literature review","volume":"18","author":"Yousif","year":"2025","journal-title":"Algorithms"},{"key":"10.1016\/j.specom.2026.103428_b55","doi-asserted-by":"crossref","first-page":"495","DOI":"10.1109\/TASLP.2021.3129994","article-title":"SoundStream: An end-to-end neural audio codec","volume":"30","author":"Zeghidour","year":"2021","journal-title":"IEEE\/ACM Trans. Audio, Speech Lang. Proc."},{"key":"10.1016\/j.specom.2026.103428_b56","series-title":"The Twelfth International Conference on Learning Representations","article-title":"SpeechTokenizer: Unified speech tokenizer for speech language models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.specom.2026.103428_b57","series-title":"Beyond hallucinations: Enhancing LVLMs through hallucination-aware direct preference optimization","author":"Zhao","year":"2023"},{"key":"10.1016\/j.specom.2026.103428_b58","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"1764","article-title":"Generative pre-trained speech language model with efficient hierarchical transformer","author":"Zhu","year":"2024"}],"container-title":["Speech Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000762?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639326000762?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T01:18:21Z","timestamp":1783127901000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167639326000762"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":58,"alternative-id":["S0167639326000762"],"URL":"https:\/\/doi.org\/10.1016\/j.specom.2026.103428","relation":{},"ISSN":["0167-6393"],"issn-type":[{"value":"0167-6393","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"DDSE: Efficient Neural Codec Language Models for speech enhancement with disentangled representations","name":"articletitle","label":"Article Title"},{"value":"Speech Communication","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.specom.2026.103428","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"103428"}}