{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:58:43Z","timestamp":1783439923795,"version":"3.54.6"},"reference-count":226,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"China NSFC","award":["92370206"],"award-info":[{"award-number":["92370206"]}]},{"name":"Shanghai Municipal Science and Technology Major Project","award":["2021SHZDZX0102"],"award-info":[{"award-number":["2021SHZDZX0102"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/tpami.2025.3643619","type":"journal-article","created":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T18:35:38Z","timestamp":1765564538000},"page":"4184-4204","source":"Crossref","is-referenced-by-count":18,"title":["Recent Advances in Discrete Speech Tokens: A Review"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-8114-2085","authenticated-orcid":false,"given":"Yiwei","family":"Guo","sequence":"first","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhihan","family":"Li","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7959-0336","authenticated-orcid":false,"given":"Hankun","family":"Wang","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bohan","family":"Li","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1157-4032","authenticated-orcid":false,"given":"Chongtian","family":"Shao","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7130-0500","authenticated-orcid":false,"given":"Hanglei","family":"Zhang","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5329-0847","authenticated-orcid":false,"given":"Chenpeng","family":"Du","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7423-617X","authenticated-orcid":false,"given":"Xie","family":"Chen","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2599-6752","authenticated-orcid":false,"given":"Shujie","family":"Liu","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia (MSRA), Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7102-9826","authenticated-orcid":false,"given":"Kai","family":"Yu","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, Jiangsu Key Lab of Language Computing; X-LANCE Lab, School of Computer Science, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.682"},{"key":"ref2","article-title":"WavChat: A survey of spoken dialogue models","author":"Ji","year":"2024"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00430"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3288409"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3530270"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3434425"},{"key":"ref8","article-title":"Towards audio language modelling: An overview","author":"Wu","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2024.3444318"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-023-16665-3"},{"key":"ref11","article-title":"CodecFake-Omni: A large-scale codec-based deepfake speech dataset","author":"Du","year":"2025"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3207050"},{"key":"ref13","article-title":"SpeechTokenizer: Unified speech tokenizer for speech language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang","year":"2024"},{"key":"ref14","article-title":"Moshi: A speech-text foundation model for real-time dialogue","author":"D\u00e9fossez","year":"2024"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00618"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447751"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446556"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2024.3506286"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1214"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-921"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094723"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448454"},{"key":"ref23","article-title":"SNAC: Multi-scale neural audio codec","volume-title":"Proc. NeurIPS 2024 Workshop","author":"Siuzdak","year":"2024"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10889508"},{"key":"ref25","first-page":"56802","article-title":"UniAudio 1.5: Large language model-driven audio codec is a few-shot audio task learner","volume-title":"Proc. NeurIPS","volume":"37","author":"Yang","year":"2024"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i24.34761"},{"key":"ref27","article-title":"Enhancing the stability of LLM-based speech generation systems through self-supervised representations","author":"Mart\u0131\u0144-Cortinas","year":"2024"},{"key":"ref28","first-page":"22605","article-title":"NaturalSpeech 3: Zero-shot speech synthesis with factorized codec and diffusion models","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"235","author":"Ju","year":"2024"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-1106"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888065"},{"key":"ref31","article-title":"DeCodec: Rethinking audio codecs as universal disentangled representation learners","author":"Luo","year":"2025"},{"key":"ref32","article-title":"vq-wav2vec: Self-supervised learning of discrete speech representations","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Baevski","year":"2020"},{"key":"ref33","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. NeurIPS","volume":"33","author":"Baevski","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref36","first-page":"18003","article-title":"ContentVec: An improved self-supervised speech representation by disentangling speakers","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Qian","year":"2022"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-288"},{"key":"ref38","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2023"},{"key":"ref39","article-title":"CosyVoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens","author":"Du","year":"2024"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2022.11.139"},{"key":"ref41","first-page":"1027","article-title":"K-means++: The advantages of careful seeding","volume-title":"Proc. SODA. SIAM","author":"Arthur","year":"2007"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446062"},{"key":"ref43","article-title":"SyllableLM: Learning coarse semantic units for speech language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Baade","year":"2025"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/MASSP.1984.1162229"},{"key":"ref45","article-title":"Neural discrete representation learning","volume-title":"Proc. NeurIPS","volume":"30","author":"Den","year":"2017"},{"key":"ref46","article-title":"Estimating or propagating gradients through stochastic neurons for conditional computation","author":"Bengio","year":"2013"},{"key":"ref47","article-title":"Generating diverse high-fidelity images with VQ-VAE-2","volume-title":"Proc. NeurIPS","volume":"32","author":"Razavi","year":"2019"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN48605.2020.9207145"},{"key":"ref49","article-title":"Jukebox: A generative model for music","author":"Dhariwal","year":"2020"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01103"},{"key":"ref51","article-title":"Language model beats diffusion: Tokenizer is key to visual generation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yu","year":"2024"},{"key":"ref52","article-title":"Vector-quantized image modeling with improved VQGAN","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yu","year":"2022"},{"key":"ref53","first-page":"22968","article-title":"Addressing representation collapse in vector quantized models with one linear layer","volume-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis.","author":"Zhu","year":"2025"},{"key":"ref54","article-title":"Categorical reparameterization with gumbel-softmax","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jang","year":"2017"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3277693"},{"key":"ref56","article-title":"Finite scalar quantization: VQ-VAE made simple","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Mentzer","year":"2024"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2010.57"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1982.1171604"},{"key":"ref59","article-title":"HiFi-Codec: Group-residual vector quantization for high fidelity audio codec","author":"Yang","year":"2023"},{"key":"ref60","article-title":"Scaling transformers for low-bitrate high-quality speech coding","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Parker","year":"2025"},{"key":"ref61","first-page":"1746","article-title":"Learning ordered representations with nested dropout","volume-title":"Proc. Proc. Int. Conf. Mach. Learn.","author":"Rippel","year":"2014"},{"key":"ref62","article-title":"Improving discrete optimisation via decoupled straight-through gumbel-softmax","author":"Shah","year":"2024"},{"key":"ref63","doi-asserted-by":"crossref","DOI":"10.17487\/rfc5219","article-title":"A more loss-tolerant RTP payload format for MP3 audio","author":"Finlayson","year":"2008"},{"key":"ref64","first-page":"1","article-title":"Definition of the opus audio codec","volume-title":"RFC","volume":"6716","author":"Valin","year":"2012"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7179063"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"ref67","article-title":"MelGAN: Generative adversarial networks for conditional waveform synthesis","volume-title":"Proc. NeurIPS","volume":"32","author":"Kumar","year":"2019"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1016"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3129994"},{"key":"ref70","article-title":"High fidelity neural audio compression","author":"D\u00e9fossez","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447523"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1559"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10889794"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.562"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096509"},{"key":"ref77","first-page":"1526","article-title":"From discrete tokens to high-fidelity audio using multi-band diffusion","volume-title":"Proc. NeurIPS","volume":"36","author":"Roman","year":"2023"},{"key":"ref78","article-title":"Vocos: Closing the gap between time-domain and fourier-based neural vocoders for high-quality audio synthesis","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Siuzdak","year":"2024"},{"key":"ref79","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume-title":"Proc. NeurIPS","volume":"33","author":"Ho","year":"2020"},{"key":"ref80","article-title":"Score-based generative modeling through stochastic differential equations","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Song","year":"2021"},{"key":"ref81","article-title":"Flow matching for generative modeling","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Lipman","year":"2023"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3417347"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-108"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2024.3469530"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447744"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP63861.2024.10800013"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3579310"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832324"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1392"},{"key":"ref90","article-title":"WavTokenizer: An efficient acoustic discrete codec tokenizer for audio language modeling","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ji","year":"2025"},{"key":"ref91","article-title":"BigCodec: Pushing the limits of low-bitrate neural speech codec","author":"Xin","year":"2024"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832247"},{"key":"ref93","article-title":"Spark-TTS: An efficient LLM-based text-to-speech model with single-stream decoupled speech tokens","author":"Wang","year":"2025"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-1440"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746883"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095442"},{"issue":"140","key":"ref97","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref98","article-title":"SecoustiCodec: Cross-modal aligned streaming single-codebook speech codec","author":"Qiang","year":"2025"},{"key":"ref99","article-title":"LLaMa 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10988"},{"key":"ref101","first-page":"28708","article-title":"Masked autoencoders that listen","volume-title":"Proc. NeurIPS","volume":"35","author":"Huang","year":"2022"},{"key":"ref102","article-title":"Llasa: Scaling train-time and inference-time compute for llama-based speech synthesis","author":"Ye","year":"2025"},{"key":"ref103","article-title":"Seamless: Multilingual expressive and streaming speech translation","author":"Barrault","year":"2023"},{"key":"ref104","article-title":"XY-Tokenizer: Mitigating the semantic-acoustic conflict in low-bitrate speech codecs","author":"Gong","year":"2025"},{"key":"ref105","first-page":"1180","article-title":"Unsupervised domain adaptation by backpropagation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ganin","year":"2015"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2003.819861"},{"key":"ref107","first-page":"5210","article-title":"AutoVC: Zero-shot voice style transfer with only autoencoder loss","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Qian","year":"2019"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29747"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446160"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1157"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10884"},{"key":"ref112","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10097097"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832198"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-1775"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref116","article-title":"Pushing the limits of semi-supervised learning for automatic speech recognition","author":"Zhang","year":"2020"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1345"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-486"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747870"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1016\/j.iswa.2023.200266"},{"key":"ref121","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.931"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3094"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1835"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-143"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-2051"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-489"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3546559"},{"key":"ref128","first-page":"6348","article-title":"Textually pretrained speech language models","volume-title":"Proc. NeurIPS","volume":"36","author":"Hassid","year":"2023"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.593"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-2251"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-2135"},{"key":"ref132","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.314"},{"key":"ref133","article-title":"MaskGCT: Zero-shot text-to-speech with masked generative codec transformer","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wang","year":"2025"},{"key":"ref134","first-page":"3915","article-title":"Self-supervised learning with random-projection quantizer for speech recognition","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Chiu","year":"2022"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3602320"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-329"},{"issue":"97","key":"ref137","first-page":"1","article-title":"Scaling speech technology to 1,000 languages","volume":"25","author":"Pratap","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"ref138","article-title":"LAST: Language model aware speech tokenization","author":"Turetzky","year":"2024"},{"key":"ref139","article-title":"NEST-RQ: Next token prediction for speech self-supervised pre-training","author":"Han","year":"2024"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.iwslt-1.46"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10022552"},{"key":"ref142","article-title":"SPIRAL: Self-supervised perturbation-invariant representation learning for speech pre-training","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Huang","year":"2022"},{"key":"ref143","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-847"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-246"},{"key":"ref145","article-title":"Removing speaker information from speech representation using variable-length soft pooling","author":"Hwang","year":"2024"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-475"},{"key":"ref147","article-title":"DASB\u2013Discrete audio and speech benchmark","author":"Mousavi","year":"2024"},{"key":"ref148","article-title":"Scaling speech-text pre-training with synthetic interleaved data","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zeng","year":"2025"},{"key":"ref149","article-title":"GLM-4-Voice: Towards intelligent and human-like end-to-end spoken chatbot","author":"Mousavi","year":"2024"},{"key":"ref150","article-title":"CosyVoice 2: Scalable streaming speech synthesis with large language models","author":"Du","year":"2024"},{"key":"ref151","article-title":"CosyVoice 3: Towards in-the-wild speech generation via scaling-up and post-training","author":"Du","year":"2025"},{"key":"ref152","article-title":"LLaMA-Omni: Seamless speech interaction with large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Fang","year":"2025"},{"key":"ref153","first-page":"17022","article-title":"Hifi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis","volume-title":"Proc. NeurIPS","volume":"33","author":"Kong","year":"2020"},{"key":"ref154","article-title":"vec2wav 2.0: Advancing voice conversion via discrete token vocoders","author":"Guo","year":"2024"},{"key":"ref155","article-title":"Better speech synthesis through scaling","author":"Betker","year":"2023"},{"key":"ref156","article-title":"Seed-TTS: A family of high-quality versatile speech generation models","author":"Anastassiou","year":"2024"},{"key":"ref157","article-title":"Why do speech language models fail to generate semantically coherent outputs? a modality evolving perspective","author":"Wang","year":"2024"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447929"},{"key":"ref159","article-title":"DiscreTalk: Text-to-speech as a machine translation problem","author":"Hayashi","year":"2020"},{"key":"ref160","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-981"},{"key":"ref161","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446063"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-533"},{"key":"ref163","first-page":"23","article-title":"A new algorithm for data compression","volume-title":"C Users J. Arch.","volume":"12","author":"Gage","year":"1994"},{"key":"ref164","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-2375"},{"key":"ref165","article-title":"Variable-rate discrete representation learning","author":"Dieleman","year":"2021"},{"key":"ref166","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1518"},{"key":"ref167","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2743"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1693"},{"key":"ref169","article-title":"The zero resource speech benchmark 2021: Metrics and baselines for unsupervised spoken language modelling","volume-title":"Proc. NeurIPS Workshop","author":"Nguyen","year":"2020"},{"key":"ref170","article-title":"Sylber: Syllabic embedding representation of speech from raw audio","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Cho","year":"2025"},{"key":"ref171","first-page":"34995","article-title":"Variable-rate hierarchical CPC leads to acoustic unit discovery in speech","volume-title":"Proc. NeurIPS","volume":"35","author":"Cuervo","year":"2022"},{"key":"ref172","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2025-1289"},{"key":"ref173","article-title":"CodecSlime: Temporal redundancy compression of neural speech codec via dynamic frame rate","author":"Wang","year":"2025"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832258"},{"key":"ref175","article-title":"TASTE: Text-aligned speech tokenization and embedding for spoken language modeling","author":"Tseng","year":"2025"},{"key":"ref176","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"ref177","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2114881"},{"key":"ref178","doi-asserted-by":"publisher","DOI":"10.21437\/eurospeech.1993-241"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.616"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832289"},{"key":"ref181","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-demo.19"},{"key":"ref182","article-title":"STAB: Speech tokenizer assessment benchmark","author":"Vashishth","year":"2024"},{"key":"ref183","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3200909"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-2131"},{"key":"ref185","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2025-1280"},{"key":"ref186","article-title":"Exploring SSL discrete tokens for multilingual ASR","author":"Cui","year":"2024"},{"key":"ref187","article-title":"A comparative study of discrete speech tokens for semantic-related tasks with large language models","author":"Wang","year":"2024"},{"key":"ref188","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1905"},{"key":"ref189","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC63619.2025.10849259"},{"key":"ref190","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890096"},{"key":"ref191","article-title":"Towards general discrete speech codec for complex acoustic environments: A study of reconstruction and downstream task consistency","volume-title":"Proc. IEEE ASRU","author":"Wang","year":"2025"},{"key":"ref192","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"ref193","first-page":"2709","article-title":"YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone","volume-title":"Proc. Proc. Int. Conf. Mach. Learn.","author":"Casanova","year":"2022"},{"key":"ref194","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.588"},{"key":"ref195","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref196","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i16.17684"},{"key":"ref197","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.447"},{"key":"ref198","article-title":"Direct speech to speech translation: A review","author":"Sarim","year":"2025"},{"key":"ref199","article-title":"LauraGPT: Listen, attend, understand, and regenerate audio with GPT","author":"Chen","year":"2023"},{"key":"ref200","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10610"},{"key":"ref201","article-title":"SpeechPrompt v2: Prompt tuning for speech classification tasks","author":"Chang","year":"2023"},{"key":"ref202","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.673"},{"key":"ref203","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-2016"},{"key":"ref204","article-title":"BASE TTS: Lessons from building a billion-parameter text-to-speech model on 100 K hours of data","author":"\u0141ajszczak","year":"2024"},{"key":"ref205","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i24.34703"},{"key":"ref206","article-title":"RALL-E: Robust codec language modelling with chain-of-thought prompting for text-to-speech synthesis","author":"Xin","year":"2024"},{"key":"ref207","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890943"},{"key":"ref208","doi-asserted-by":"publisher","DOI":"10.1109\/SLT61566.2024.10832301"},{"key":"ref209","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-1531"},{"key":"ref210","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888194"},{"key":"ref211","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.40"},{"key":"ref212","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890588"},{"key":"ref213","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890141"},{"key":"ref214","article-title":"SpeechGen: Unlocking the generative power of speech language models with prompts","author":"Wu","year":"2023"},{"key":"ref215","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3419418"},{"key":"ref216","first-page":"56422","article-title":"UniAudio: Towards universal audio generation with large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"235","author":"Yang","year":"2024"},{"key":"ref217","doi-asserted-by":"publisher","DOI":"10.3362\/0262-8104.2002.009"},{"key":"ref218","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890230"},{"key":"ref219","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.313"},{"key":"ref220","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.65"},{"key":"ref221","article-title":"Continuous speech synthesis using per-token latent diffusion","author":"Turetzky","year":"2024"},{"key":"ref222","first-page":"27255","article-title":"DiTAR: Diffusion transformer autoregressive modeling for speech generation","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"267","author":"Jia","year":"2025"},{"key":"ref223","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.997"},{"key":"ref224","first-page":"30","article-title":"SpiRit-LM: Interleaved spoken and written language model","volume":"13","author":"Nguyen","year":"2025","journal-title":"Trans. Assoc. Comput. Linguistics"},{"key":"ref225","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.21"},{"key":"ref226","article-title":"TaDiCodec: Text-aware diffusion speech tokenizer for speech language modeling","author":"Wang","year":"2025"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11424231\/11298521.pdf?arnumber=11298521","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T01:34:56Z","timestamp":1773106496000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11298521\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":226,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2025.3643619","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}