{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T06:15:40Z","timestamp":1783923340000,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":24,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234370","type":"print"},{"value":"9789819234387","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T00:00:00Z","timestamp":1783987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3438-7_28","type":"book-chapter","created":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T05:46:27Z","timestamp":1783921587000},"page":"335-346","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["FLWCodec: A Low-Bitrate Neural Speech Codec with Feature-Weighted RVQ and Local Window Attention"],"prefix":"10.1007","author":[{"given":"Tianyu","family":"Cai","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ye","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingxiang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuxian","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,14]]},"reference":[{"key":"28_CR1","unstructured":"Chae, Y., et al.: VRVQ: variable bitrate residual vector quantization for audio compression. arXiv Preprint arXiv:2410.06016. (2024)"},{"key":"28_CR2","unstructured":"D\u00e9fossez, A., Copet, J., Synnaeve, G., Adi, Y.: High fidelity neural audio compression. arXiv Preprint arXiv:2210.13438. (2022)"},{"key":"28_CR3","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1109\/ICASSP.1990.115972","volume-title":"International Conference on Acoustics, Speech, and Signal Processing","author":"A Erell","year":"1990","unstructured":"Erell, A., Weintraub, M.: Estimation using log-spectral-distance criterion for noise-robust speech recognition. In: International Conference on Acoustics, Speech, and Signal Processing, pp. 853\u2013856. IEEE (1990)"},{"key":"28_CR4","first-page":"735","volume-title":"ICASSP 2019--2019 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"C G\u00e2rbacea","year":"2019","unstructured":"G\u00e2rbacea, C., et al.: Low bit-rate speech coding with VQ-VAE and a WaveNet decoder. In: ICASSP 2019--2019 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 735\u2013739. IEEE (2019)"},{"key":"28_CR5","unstructured":"Ito, K., Johnson, L.: The LJ speech dataset. https:\/\/keithito.com\/LJ-Speech-Dataset\/. (2017)"},{"key":"28_CR6","unstructured":"Ji, S., et al.: Language-codec: reducing the gaps between discrete codec representation and speech language models. arXiv Preprint arXiv:2402.12208. (2024)"},{"key":"28_CR7","first-page":"17022","volume":"33","author":"J Kong","year":"2020","unstructured":"Kong, J., Kim, J., Bae, J.: HiFi-GAN: generative adversarial networks for efficient and high Fidelity speech synthesis. Adv. Neural Inf. Proces. Syst. 33, 17022\u201317033 (2020)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"28_CR8","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv Preprint arXiv:1711.05101. (2017)"},{"key":"28_CR9","first-page":"281","volume-title":"Proceedings of the Fifth Berkeley Symposium on Mathematical Statistics and Probability, Volume 1: Statistics","author":"J MacQueen","year":"1967","unstructured":"MacQueen, J.: Some methods for classification and analysis of multivariate observations. In: Proceedings of the Fifth Berkeley Symposium on Mathematical Statistics and Probability, Volume 1: Statistics, vol. 5, pp. 281\u2013298. University of California Press (1967)"},{"key":"28_CR10","first-page":"5206","volume-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"V Panayotov","year":"2015","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: LibriSpeech: an ASR corpus based on public domain audio books. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 5206\u20135210. IEEE (2015)"},{"key":"28_CR11","first-page":"862","volume-title":"Recommendation ITU-T P","author":"IT Recommendation","year":"2001","unstructured":"Recommendation, I.T.: Perceptual evaluation of speech quality (PESQ): an objective method for end-to-end speech quality assessment of narrow-band telephone networks and speech codecs. In: Recommendation ITU-T P, p. 862 (2001)"},{"key":"28_CR12","first-page":"80","volume-title":"TAPR and ARRL 30th Digital Communications Conference","author":"D Rowe","year":"2011","unstructured":"Rowe, D.: Codec 2: open source speech coding at 2400 bits\/s and below. In: TAPR and ARRL 30th Digital Communications Conference, pp. 80\u201384 (2011)"},{"key":"28_CR13","unstructured":"Siuzdak, H., Gr\u00f6tschla, F., Lanzend\u00f6rfer, L.A.: SNAC: multi-scale neural audio codec. arXiv Preprint arXiv:2410.14411. (2024)"},{"key":"28_CR14","doi-asserted-by":"publisher","first-page":"1591","DOI":"10.1109\/ICASSP.1997.596257","volume-title":"1997 IEEE International Conference on Acoustics, Speech, and Signal Processing","author":"LM Supplee","year":"1997","unstructured":"Supplee, L.M., Cohn, R.P., Collura, J.S., McCree, A.V.: MELP: the new federal standard at 2400 bps. In: 1997 IEEE International Conference on Acoustics, Speech, and Signal Processing, vol. 2, pp. 1591\u20131594. IEEE (1997)"},{"key":"28_CR15","doi-asserted-by":"publisher","first-page":"4214","DOI":"10.1109\/ICASSP.2010.5495701","volume-title":"2010 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"CH Taal","year":"2010","unstructured":"Taal, C.H., Hendriks, R.C., Heusdens, R., Jensen, J.: A short-time objective intelligibility measure for time-frequency weighted noisy speech. In: 2010 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 4214\u20134217. IEEE (2010)"},{"key":"28_CR16","volume-title":"Definition of the Opus Audio Codec","author":"JM Valin","year":"2012","unstructured":"Valin, J.M., Vos, K., Terriberry, T.: Definition of the Opus Audio Codec. Tech. Rep (2012)"},{"key":"28_CR17","unstructured":"van den Oord, A., et al.: WaveNet: a generative model for raw audio. arXiv Preprint arXiv:1609.03499.12. (2016)"},{"key":"28_CR18","volume-title":"Neural Information Processing Systems (NIPS)","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Neural Information Processing Systems (NIPS) (2017)"},{"key":"28_CR19","unstructured":"Wang, D., Zhang, X.: THCHS-30: a free Chinese speech corpus. arXiv Preprint arXiv:1512.01882. (2015)"},{"key":"28_CR20","doi-asserted-by":"crossref","unstructured":"Xu, L., et al.: An intra-BRNN and GB-RVQ based end-to-end neural audio codec. arXiv Preprint arXiv:2402.01271. (2024)","DOI":"10.21437\/Interspeech.2023-537"},{"key":"28_CR21","unstructured":"Yang, D., Liu, S., Huang, R., Tian, J., Weng, C., Zou, Y.: HiFi-codec: group-residual vector quantization for high fidelity audio codec. arXiv Preprint arXiv:2305.02765. (2023)"},{"key":"28_CR22","first-page":"346","volume-title":"Pacific Rim International Conference on Artificial Intelligence","author":"X Yu","year":"2024","unstructured":"Yu, X., Li, Y., Zhang, P., Lin, L., Cai, T.: MSCAcodec: a low-rate neural speech codec with multi-scale residual channel attention. In: Pacific Rim International Conference on Artificial Intelligence, pp. 346\u2013358. Springer (2024)"},{"key":"28_CR23","doi-asserted-by":"publisher","first-page":"495","DOI":"10.1109\/TASLP.2021.3129994","volume":"30","author":"N Zeghidour","year":"2021","unstructured":"Zeghidour, N., Luebs, A., Omran, A., Skoglund, J., Tagliasacchi, M.: SoundStream: an end-to-end neural audio codec. IEEE\/ACM Trans. Audio Speech Lang. Process. 30, 495\u2013507 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"28_CR24","doi-asserted-by":"crossref","unstructured":"Zheng, R.C., Du, H.P., Jiang, X.H., Ai, Y., Ling, Z.H.: ERVQ: enhanced residual vector quantization with intra-and-inter-codebook optimization for neural audio codecs. arXiv Preprint arXiv:2410.12359. (2024)","DOI":"10.1109\/TASLPRO.2025.3579310"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3438-7_28","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T05:46:29Z","timestamp":1783921589000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3438-7_28"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,14]]},"ISBN":["9789819234370","9789819234387"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3438-7_28","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,14]]},"assertion":[{"value":"14 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}