{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:37:40Z","timestamp":1776886660224,"version":"3.51.2"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2025,1,25]],"date-time":"2025-01-25T00:00:00Z","timestamp":1737763200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,25]],"date-time":"2025-01-25T00:00:00Z","timestamp":1737763200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62471012"],"award-info":[{"award-number":["62471012"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s00034-025-02995-0","type":"journal-article","created":{"date-parts":[[2025,1,25]],"date-time":"2025-01-25T17:02:08Z","timestamp":1737824528000},"page":"4013-4032","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["TFF-Codec: A High Fidelity End-to-End Neural Audio Codec"],"prefix":"10.1007","volume":"44","author":[{"given":"Yuhao","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3452-3913","authenticated-orcid":false,"given":"Maoshen","family":"Jia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiawei","family":"Ru","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lizhong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"Wen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,25]]},"reference":[{"issue":"8","key":"2995_CR1","doi-asserted-by":"publisher","first-page":"1973","DOI":"10.1002\/j.1538-7305.1970.tb04297.x","volume":"49","author":"B Atal","year":"1970","unstructured":"B. Atal, M. Schroeder, Adaptive predictive coding of speech signals. Bell Syst. Tech. J. 49(8), 1973\u20131986 (1970)","journal-title":"Bell Syst. Tech. J."},{"key":"2995_CR2","unstructured":"A. Baevski, S. Schneider, M. Auli, vq-wav2vec: Self-supervised learning of discrete speech representations. (2019)"},{"key":"2995_CR3","doi-asserted-by":"crossref","unstructured":"M. Chinen, F. Lim, J. Skoglund, N. Gureev, F. O'Gorman, A. Hines, ViSQOL v3: An open source production ready objective speech and audio metric. 2020 Twelfth international conference on quality of multimedia experience (QoMEX), 1\u20136. (2020)","DOI":"10.1109\/QoMEX48832.2020.9123150"},{"key":"2995_CR4","unstructured":"A. D\u00e9fossez, J. Copet, G. Synnaeve, Y. Adi, High fidelity neural audio compression. (2022)"},{"key":"2995_CR5","doi-asserted-by":"crossref","unstructured":"C. Garbacea, A. Den Oord, Y. Li, F. Lim, A. Luebs, O. Vinyals, T. Walters, Low Bit-rate speech coding with VQ-VAE and a WaveNet Decoder. 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP), 735\u2013739. (2019)","DOI":"10.1109\/ICASSP.2019.8683277"},{"key":"2995_CR6","doi-asserted-by":"crossref","unstructured":"X. Jiang, X. Peng, C. Zheng, H. Xue, Y. Zhang, Y. Lu, End-to-end neural speech coding for real-time communications. 2022 IEEE international conference on acoustics, speech and signal processing (ICASSP), 866\u2013870. (2022)","DOI":"10.1109\/ICASSP43922.2022.9746296"},{"key":"2995_CR7","doi-asserted-by":"publisher","first-page":"4222","DOI":"10.21437\/Interspeech.2022-10084","volume":"2022","author":"X Jiang","year":"2022","unstructured":"X. Jiang, X. Peng, H. Xue, Y. Zhang, Y. Lu, Cross-scale vector quantization for scalable neural speech coding. Interspeech 2022, 4222\u20134226 (2022)","journal-title":"Interspeech"},{"key":"2995_CR8","doi-asserted-by":"publisher","first-page":"2111","DOI":"10.1109\/TASLP.2023.3277693","volume":"31","author":"X Jiang","year":"2023","unstructured":"X. Jiang, X. Peng, H. Xue, Y. Zhang, Y. Lu, Latent-domain predictive neural speech coding. IEEE\/ACM Trans. Audio Speech Language Process. 31, 2111\u20132123 (2023)","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"key":"2995_CR9","unstructured":"D. P. Kingma, J. Ba, Adam: A method for stochastic optimization. (2014)"},{"key":"2995_CR10","doi-asserted-by":"crossref","unstructured":"W. Kleijn, F. Lim, A. Luebs, J. Skoglund, F. Stimberg, Q. Wang, T. Walters, Wavenet based low rate speech coding. 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), 676\u2013680. (2018)","DOI":"10.1109\/ICASSP.2018.8462529"},{"key":"2995_CR11","doi-asserted-by":"crossref","unstructured":"W. Kleijn, A. Storus, M. Chinen, T. Denton, F. Lim, A. Luebs, H. Yeh, Generative speech coding with predictive variance regularization. 2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), 6478\u20136482. (2021)","DOI":"10.1109\/ICASSP39728.2021.9415120"},{"key":"2995_CR12","doi-asserted-by":"crossref","unstructured":"J. Klejsa, P. Hedelin, C. Zhou, R. Fejgin, L. Villemoes, High-quality speech coding with sample RNN. 2019 IEEE International conference on acoustics, speech and signal processing (ICASSP), 7155\u20137159. (2019)","DOI":"10.1109\/ICASSP.2019.8682435"},{"key":"2995_CR13","first-page":"17022","volume":"33","author":"J Kong","year":"2020","unstructured":"J. Kong, J. Kim, J. Bae, Hifi-gan: generative adversarial networks for efficient and high fidelity speech synthesis. Adv. Neural. Inf. Process. Syst. 33, 17022\u201317033 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2995_CR14","doi-asserted-by":"crossref","unstructured":"S. Korse, N. Pia, K. Gupta, G. Fuchs, PostGAN: A GAN-based post-processor to enhance the quality of coded speech. 2022 IEEE international conference on acoustics, speech and signal processing (ICASSP), 831\u2013835. (2022)","DOI":"10.1109\/ICASSP43922.2022.9747733"},{"key":"2995_CR15","unstructured":"R. Kumar, P. Seetharaman, A. Luebs, I. Kumar, K. Kumar, High-fidelity audio compression with improved RVQGAN. (2023)"},{"key":"2995_CR16","unstructured":"J. Lei, Jamie Ryan Kiros, G. Hinton, Layer Normalization. (2016)"},{"issue":"1","key":"2995_CR17","doi-asserted-by":"publisher","DOI":"10.1121\/10.0003321","volume":"1","author":"A Li","year":"2021","unstructured":"A. Li, C. Zheng, R. Peng, X. Li, On the importance of power compression and phase estimation in monaural speech dereverberation. JASA Exp. Lett. 1(1), 014802 (2021)","journal-title":"JASA Exp. Lett."},{"key":"2995_CR18","doi-asserted-by":"crossref","unstructured":"M. Maruschke, O. Jokisch, M. Meszaros, V. Iaroshenko, Review of the Opus codec in a WebRTC scenario for audio and speech communication. In Speech and Computer: 17th International Conference, SPECOM 2015, Athens, Greece, September 20\u201324, 2015, pp. 348\u2013355. (2015)","DOI":"10.1007\/978-3-319-23132-7_43"},{"issue":"1","key":"2995_CR19","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1109\/45.1890","volume":"7","author":"D O'Shaughnessy","year":"1988","unstructured":"D. O\u2019Shaughnessy, Linear predictive coding. IEEE Potentials 7(1), 29\u201332 (1988)","journal-title":"IEEE Potentials"},{"key":"2995_CR20","doi-asserted-by":"crossref","unstructured":"D. Petermann, S. Beack, M. Kim, Harp-Net: Hyper-autoencoded reconstruction propagation for scalable neural audio coding. 2021 IEEE Workshop on applications of signal processing to audio and acoustics (WASPAA), 316\u2013320. (2021)","DOI":"10.1109\/WASPAA52581.2021.9632723"},{"key":"2995_CR21","doi-asserted-by":"crossref","unstructured":"D. Petermann, I. Jang, M. Kim, Native multi-band audio coding within hyper-autoencoded reconstruction propagation networks. 2023 IEEE international conference on acoustics, speech and signal processing (ICASSP), 1\u20135. (2023)","DOI":"10.1109\/ICASSP49357.2023.10094593"},{"key":"2995_CR22","unstructured":"Z. Rafii, A. Liutkus, F. St\u00f6ter, S. Mimilakis, R. Bittner, MUSDB18 - a Corpus for Music Separation. (2017)"},{"key":"2995_CR23","doi-asserted-by":"crossref","unstructured":"J. Roux, S. Wisdom, H. Erdogan, J. Hershey, SDR - Half-baked or Well Done? ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 626\u2013630. (2019)","DOI":"10.1109\/ICASSP.2019.8683855"},{"key":"2995_CR24","unstructured":"D. Rowe, Codec 2-open source speech coding at 2400 bits\/s and below. In TAPR and ARRL 30th Digital Communications Conference, 80\u201384. (2011)"},{"issue":"1","key":"2995_CR25","doi-asserted-by":"publisher","first-page":"12005","DOI":"10.1088\/1742-6596\/2759\/1\/012005","volume":"2759","author":"J Ru","year":"2024","unstructured":"J. Ru, L. Wang, M. Jia, L. Wen, C. Wang, Y. Zhao, J. Wang, Neural audio coding with deep complex networks. J. Phys. Conf. Ser. 2759(1), 12005 (2024)","journal-title":"J. Phys. Conf. Ser."},{"key":"2995_CR26","doi-asserted-by":"crossref","unstructured":"M. Schroeder, B. Atal, Code-excited linear prediction (CELP): High-quality speech at very low bit rates. ICASSP '85. IEEE International Conference on Acoustics, Speech, and Signal Processing, 10, 937\u2013940. (1985)","DOI":"10.1109\/ICASSP.1985.1168147"},{"key":"2995_CR27","unstructured":"B. Series, Method for the subjective assessment of intermediate quality level of audio systems. International Telecommunication Union Radiocommunication Assembly. (2014)"},{"key":"2995_CR28","unstructured":"J. Valin, G. Maxwell, T. Terriberry, K. Vos, High-quality, low-delay music coding in the opus codec. (2016)"},{"key":"2995_CR29","doi-asserted-by":"crossref","unstructured":"J. Valin, J. Skoglund, LPCNET: Improving neural speech synthesis through linear prediction. ICASSP 2019 - 2019 IEEE international conference on acoustics, speech and signal processing (ICASSP), 5891\u20135895. (2019)","DOI":"10.1109\/ICASSP.2019.8682804"},{"key":"2995_CR30","doi-asserted-by":"crossref","unstructured":"J.-M. Valin, J. Skoglund, A Real-Time Wideband Neural Vocoder at 1.6\u00a0kb\/s Using LPCNet. Interspeech 2019, 3406\u20133410. (2019)","DOI":"10.21437\/Interspeech.2019-1255"},{"key":"2995_CR31","unstructured":"A. Van den Oord, S. Dieleman, H. Zen, K. Simonyan, O. Vinyals, A. Graves, K. Kavukcuoglu, WaveNet: A Generative Model for Raw Audio. (2016)"},{"key":"2995_CR32","unstructured":"A. Van den Oord, O. Vinyals, K. Kavukcuoglu, Neural discrete representation learning. (2018)"},{"key":"2995_CR33","unstructured":"K. Vos, K. V. S\u00f8rensen, S. S. Jensen, J. M. Valin, Voice coding with Opus. In Audio Engineering Society Convention 135. Audio Engineering Society. (2013)"},{"key":"2995_CR34","doi-asserted-by":"publisher","first-page":"1778","DOI":"10.1109\/TASLP.2020.2998279","volume":"28","author":"Z Wang","year":"2020","unstructured":"Z. Wang, P. Wang, D. Wang, Complex spectral mapping for single- and multi-channel speech enhancement and robust ASR. IEEE\/ACM Trans. Audio Speech Language Process. 28, 1778\u20131787 (2020)","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"key":"2995_CR35","doi-asserted-by":"crossref","unstructured":"Y. Wu, I. Gebru, D. Markovic, A. Richard, Audiodec: an open-source streaming high-fidelity neural audio codec. 2023 IEEE international conference on acoustics, speech and signal processing (ICASSP), 1\u20135. (2023)","DOI":"10.1109\/ICASSP49357.2023.10096509"},{"key":"2995_CR36","doi-asserted-by":"crossref","unstructured":"R. Yamamoto, E. Song, J. Kim, Parallel wavegan: a fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram. 2020 IEEE international conference on acoustics, speech and signal processing (ICASSP), 6199\u20136203. (2020)","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"2995_CR37","doi-asserted-by":"publisher","first-page":"495","DOI":"10.1109\/TASLP.2021.3129994","volume":"30","author":"N Zeghidour","year":"2022","unstructured":"N. Zeghidour, A. Luebs, A. Omran, J. Skoglund, M. Tagliasacchi, SoundStream: an end-to-end neural audio codec. IEEE\/ACM Trans. Audio Speech Language Process. 30, 495\u2013507 (2022)","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"issue":"4","key":"2995_CR38","doi-asserted-by":"publisher","first-page":"663","DOI":"10.1109\/TASLP.2018.2887337","volume":"27","author":"Z Zhao","year":"2019","unstructured":"Z. Zhao, H. Liu, T. Fingscheidt, Convolutional neural networks to enhance coded speech. IEEE\/ACM Trans. Audio Speech Language Process. 27(4), 663\u2013678 (2019)","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"key":"2995_CR39","doi-asserted-by":"crossref","unstructured":"E. Zwicker, H. Fastl, Psychoacoustics (2nd ed., Vol. 22, Springer Series in Information Sciences). Berlin, Heidelberg: Springer Berlin\/Heidelberg. (1999)","DOI":"10.1007\/978-3-662-09562-1"}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-02995-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-025-02995-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-02995-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T20:08:45Z","timestamp":1747253325000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-025-02995-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,25]]},"references-count":39,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["2995"],"URL":"https:\/\/doi.org\/10.1007\/s00034-025-02995-0","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"value":"0278-081X","type":"print"},{"value":"1531-5878","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,25]]},"assertion":[{"value":"9 August 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 January 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 January 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 January 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}