{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T04:10:36Z","timestamp":1748059836918,"version":"3.41.0"},"publisher-location":"Singapore","reference-count":21,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819609932","type":"print"},{"value":"9789819609949","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-0994-9_25","type":"book-chapter","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T13:23:29Z","timestamp":1748006609000},"page":"267-278","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Two-Stage Neural Network for Speech Signal Reconstruction from Mel Spectrograms"],"prefix":"10.1007","author":[{"given":"Filippo","family":"Villani","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michele","family":"Scarpiniti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aurelio","family":"Uncini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,24]]},"reference":[{"key":"25_CR1","doi-asserted-by":"publisher","unstructured":"Boucheron, L.E., De\u00a0Leon, P.L.: On the inversion of mel-frequency cepstral coefficients for speech enhancement applications. In: 2008 International Conference on Signals and Electronic Systems, pp. 485\u2013488 (2008). https:\/\/doi.org\/10.1109\/ICSES.2008.4673475","DOI":"10.1109\/ICSES.2008.4673475"},{"issue":"11","key":"25_CR2","doi-asserted-by":"publisher","first-page":"5225","DOI":"10.1109\/TSP.2011.2162508","volume":"59","author":"J Chen","year":"2011","unstructured":"Chen, J., Richard, C., Bermudez, J.C.M., Honeine, P.: Nonnegative least-mean-square algorithm. IEEE Trans. Signal Proc. 59(11), 5225\u20135235 (2011). https:\/\/doi.org\/10.1109\/TSP.2011.2162508","journal-title":"IEEE Trans. Signal Proc."},{"issue":"5","key":"25_CR3","doi-asserted-by":"publisher","first-page":"3348","DOI":"10.1121\/10.0002702","volume":"148","author":"X Dong","year":"2020","unstructured":"Dong, X., Williamson, D.S.: Towards real-world objective speech quality and intelligibility assessment using speech-enhancement residuals and convolutional long short-term memory networks. J. Acoust. Soc. Amer. 148(5), 3348\u20133359 (2020). https:\/\/doi.org\/10.1121\/10.0002702","journal-title":"J. Acoust. Soc. Amer."},{"key":"25_CR4","unstructured":"Engel, J., Agrawal, K.K., Chen, S., Gulrajani, I., Donahue, C., Roberts, A.: GANSynth: adversarial neural audio synthesis. In: Proceedings of International Conference on Learning Representations (ICLR 2019), pp. 1\u201317 (2019)"},{"key":"25_CR5","unstructured":"Giorgi, B.D., Levy, M., Sharp, R.: Mel spectrogram inversion with stable pitch. In: Proceedings of the 23st International Society for Music Information Retrieval Conference (ISMIR 2022), pp. 233\u2013239 (2022)"},{"issue":"2","key":"25_CR6","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin, D., Lim, J.: Signal estimation from modified short-time Fourier transform. IEEE Trans. Acoust. Speech Signal Proc. 32(2), 236\u2013243 (1984). https:\/\/doi.org\/10.1109\/TASSP.1984.1164317","journal-title":"IEEE Trans. Acoust. Speech Signal Proc."},{"key":"25_CR7","doi-asserted-by":"publisher","unstructured":"Kathania, H.K., Shahnawazuddin, S., Ahmad, W., Adiga, N.: On the role of linear, mel and inverse-mel filterbank in the context of automatic speech recognition. In: 2019 National Conference on Communications (NCC 2019), pp. 1\u20135 (2019). https:\/\/doi.org\/10.1109\/NCC.2019.8732232","DOI":"10.1109\/NCC.2019.8732232"},{"key":"25_CR8","unstructured":"Kong, J., Kim, J., Bae, J.: HiFi-GAN: generative adversarial networks for efficient and high fidelity speech synthesis. In: 34th Conference on Neural Information Processing Systems (NeurIPS 2020), pp. 17022\u201317033 (2020)"},{"key":"25_CR9","unstructured":"Kong, Z., Ping, W., Huang, J., Zhao, K., Catanzaro, B.: DiffWave: A versatile diffusion model for audio synthesis. In: Proceedings of International Conference on Learning Representations (ICLR 2021), pp. 1\u201317 (2021)"},{"key":"25_CR10","unstructured":"Kumar, K., Kumar, R., De\u00a0Boissiere, T., Gestin, L., Teoh, W.Z., Sotelo, J., de\u00a0Br\u00e9bisson, A., Bengio, Y., Courville, A.C.: MelGAN: Generative adversarial networks for conditional waveform synthesis. In: 33rd Conference on Neural Information Processing Systems (NeurIPS 2019), vol. 32, pp. 14910\u201314921 (2019)"},{"key":"25_CR11","unstructured":"Le\u00a0Roux, J., Kameoka, H., Ono, N., Sagayama, S.: Fast signal reconstruction from magnitude STFT spectrogram based on spectrogram consistency. In: Proceedings of the 13th International Conference on Digital Audio Effects (DAFx-10), vol. 10, pp. 397\u2013403 (2010)"},{"key":"25_CR12","doi-asserted-by":"publisher","unstructured":"Lu, X., Tsao, Y., Matsuda, S., Hori, C.: Speech enhancement based on deep denoising autoencoder. In: Interspeech, vol. 2013, pp. 436\u2013440 (2013). https:\/\/doi.org\/10.21437\/Interspeech.2013-130","DOI":"10.21437\/Interspeech.2013-130"},{"issue":"8","key":"25_CR13","doi-asserted-by":"publisher","first-page":"1256","DOI":"10.1109\/TASLP.2019.2915167","volume":"27","author":"Y Luo","year":"2019","unstructured":"Luo, Y., Mesgarani, N.: Conv-TasNet: Surpassing ideal time-frequency magnitude masking for speech separation. IEEE\/ACM Trans. Audio Speech Lang. Proc. 27(8), 1256\u20131266 (2019). https:\/\/doi.org\/10.1109\/TASLP.2019.2915167","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Proc."},{"issue":"1","key":"25_CR14","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1109\/JSTSP.2020.3034486","volume":"15","author":"Y Masuyama","year":"2020","unstructured":"Masuyama, Y., Yatabe, K., Koizumi, Y., Oikawa, Y., Harada, N.: Deep Griffin-Lim iteration: Trainable iterative phase reconstruction using neural network. IEEE J. Select. Top. Signal Proc. 15(1), 37\u201350 (2020). https:\/\/doi.org\/10.1109\/JSTSP.2020.3034486","journal-title":"IEEE J. Select. Top. Signal Proc."},{"key":"25_CR15","unstructured":"Morrison, M., Kumar, R., Kumar, K., Seetharaman, P., Courville, A., Bengio, Y.: Chunked autoregressive GAN for conditional waveform synthesis. In: Proceedings of International Conference on Learning Representations (ICLR 2022), pp. 1\u201322 (2022)"},{"key":"25_CR16","doi-asserted-by":"publisher","unstructured":"Nustede, E.J., Anem\u00fcller, J.: Towards speech enhancement using a variational U-Net architecture. In: 2021 29th European Signal Processing Conference (EUSIPCO), pp. 481\u2013485 (2021). https:\/\/doi.org\/10.23919\/EUSIPCO54536.2021.9616114","DOI":"10.23919\/EUSIPCO54536.2021.9616114"},{"key":"25_CR17","doi-asserted-by":"publisher","unstructured":"Perraudin, N., Balazs, P., S\u00f8ndergaard, P.L.: A fast Griffin-Lim algorithm. In: 2013 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA 2013), pp. 1\u20134 (2013). https:\/\/doi.org\/10.1109\/WASPAA.2013.6701851","DOI":"10.1109\/WASPAA.2013.6701851"},{"key":"25_CR18","unstructured":"Ping, W., Peng, K., Zhao, K., Song, Z.: WaveFlow: A compact flow-based model for raw audio. In: International Conference on Learning Representations (ICLR 2020) (2020)"},{"key":"25_CR19","doi-asserted-by":"publisher","unstructured":"Prenger, R., Valle, R., Catanzaro, B.: WaveGlow: A flow-based generative network for speech synthesis. In: 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP 2019), pp. 3617\u20133621 (2019). https:\/\/doi.org\/10.1109\/ICASSP.2019.8683143","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"25_CR20","doi-asserted-by":"publisher","unstructured":"Shen, J., Pang, R., Weiss, R.J., Schuster, M., Jaitly, N., Yang, Z., Chen, Z., Zhang, Y., Wang, Y., Skerrv-Ryan, R., Saurous, R.A., Agiomvrgiannakis, Y., Wu, Y.: Natural TTS synthesis by conditioning WaveNet on mel spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783 (2018). https:\/\/doi.org\/10.1109\/ICASSP.2018.8461368","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"25_CR21","unstructured":"van den Oord, A., Dieleman, S., Zen, H., Simonyan, K., Vinyals, O., Graves, A., Kalchbrenner, N., Senior, A., Kavukcuoglu, K.: WaveNet: A generative model for raw audio. In: Proceedings of the 9th ISCA Workshop on Speech Synthesis Workshop (SSW 9), p. 125 (2016)"}],"container-title":["Smart Innovation, Systems and Technologies","Advanced Neural Artificial Intelligence: Theories and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-0994-9_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T13:23:36Z","timestamp":1748006616000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-0994-9_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819609932","9789819609949"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-0994-9_25","relation":{},"ISSN":["2190-3018","2190-3026"],"issn-type":[{"value":"2190-3018","type":"print"},{"value":"2190-3026","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"24 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}