{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T21:24:56Z","timestamp":1781731496858,"version":"3.54.5"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,7,6]],"date-time":"2024-07-06T00:00:00Z","timestamp":1720224000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,6]],"date-time":"2024-07-06T00:00:00Z","timestamp":1720224000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s10772-024-10119-3","type":"journal-article","created":{"date-parts":[[2024,7,6]],"date-time":"2024-07-06T03:37:02Z","timestamp":1720237022000},"page":"539-549","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Noise robust speech encoding system in challenging acoustic conditions"],"prefix":"10.1007","volume":"27","author":[{"given":"B. G.","family":"Nagaraja","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3266-9732","authenticated-orcid":false,"given":"G. Thimmaraja","family":"Yadava","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"K.","family":"Harshitha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,7,6]]},"reference":[{"issue":"7","key":"10119_CR1","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi, J., Hazan, E., & Singer, Y. (2011). Adaptive subgradient methods for online learning and stochastic optimization. Journal of Machine Learning Research, 12(7), 2121\u20132159.","journal-title":"Journal of Machine Learning Research"},{"key":"10119_CR2","doi-asserted-by":"crossref","unstructured":"Erdogan, H., Hershey, J. R., Watanabe, S. & Roux, J. L. (2015). Phase-sensitive and recognition-boosted speech separation using deep recurrent neural networks. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 708\u2013712).","DOI":"10.1109\/ICASSP.2015.7178061"},{"key":"10119_CR3","doi-asserted-by":"publisher","first-page":"1121","DOI":"10.1109\/ICASSP.1985.1168461","volume":"10","author":"D Friedman","year":"1985","unstructured":"Friedman, D. (1985). An interpretation of the phase structure of speech. Instantaneous-frequency distribution vs. time. In\u00a0IEEE international conference on acoustics, speech, and signal processing (ICASSP) (pp.1121\u20131124).","journal-title":"IEEE International Conference on Acoustics, Speech, and Signal Processing"},{"key":"10119_CR4","doi-asserted-by":"crossref","unstructured":"Gaich, A. & Mowlaee, P. (2015). On speech quality estimation of phase-aware single-channel speech enhancement. In Proceedings of IEEE on acoustics, speech and signal processing (ICASSP) (pp. 216\u2013220).","DOI":"10.1109\/ICASSP.2015.7177963"},{"key":"10119_CR5","doi-asserted-by":"crossref","unstructured":"Gajecki, T. & Nogueira, W. (2022). An end-to-end deep learning speech coding and denoising strategy for cochlear implants. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 3109\u20133113).","DOI":"10.1101\/2021.11.04.467324"},{"issue":"2","key":"10119_CR6","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1109\/MSP.2014.2369251","volume":"32","author":"T Gerkmann","year":"2015","unstructured":"Gerkmann, T., Krawczyk-Becker, M., & Le Roux, J. (2015). Phase processing for single-channel speech enhancement: History and recent advances. IEEE Signal Processing Magazine, 32(2), 55\u201366.","journal-title":"IEEE Signal Processing Magazine"},{"key":"10119_CR7","unstructured":"Glorot, X., Bordes, A. & Bengio, Y. (2011). Deep sparse rectifier neural networks. In Proceedings of the fourteenth international conference on artificial intelligence and statistics (pp. 315\u2013323)."},{"key":"10119_CR8","unstructured":"He, F., Chu, S. H. C., Kjartansson, O., Rivera, C. E., Katanova, A., Gutkin, A., Demirsahin, I., Johny, C., Jansche, M., Sarin, S. & Pipatsrisawat, K. (2020). Open-source multi-speaker speech corpora for building Gujarati, Kannada, Malayalam, Marathi, Tamil and Telugu speech synthesis systems. In Proceedings of the twelfth language resources and evaluation conference (ELRA) (pp. 6494\u20136503)."},{"key":"10119_CR9","unstructured":"Hinton, G. E., Srivastava, N., Krizhevsky, A., Sutskever, I. & Salakhutdinov, R. (2012). Improving neural networks by preventing co-adaptation of feature detectors. arXiv:1207.0580"},{"key":"10119_CR10","doi-asserted-by":"crossref","unstructured":"Hirsch, H. & Pearce, D. (2000). The Aurora experimental framework for the performance evaluation of speech recognition systems under noisy conditions. In ISCA ITRW ASR2000, (pp. 181\u2013188).","DOI":"10.21437\/ICSLP.2000-743"},{"issue":"7\u20138","key":"10119_CR11","doi-asserted-by":"publisher","first-page":"588","DOI":"10.1016\/j.specom.2006.12.006","volume":"49","author":"Y Hu","year":"2007","unstructured":"Hu, Y., & Loizou, P. C. (2007). Subjective comparison and evaluation of speech enhancement algorithms. Speech Communication, 49(7\u20138), 588\u2013601.","journal-title":"Speech Communication"},{"issue":"1","key":"10119_CR12","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TASL.2007.911054","volume":"16","author":"Y Hu","year":"2008","unstructured":"Hu, Y., & Loizou, P. C. (2008). Evaluation of objective quality measures for speech enhancement. IEEE Transactions on Speech and Audio Processing, 16(1), 229\u2013238.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"10119_CR13","doi-asserted-by":"crossref","unstructured":"Huang, P. S., Kim, M., Hasegawa-Johnson, M. & Smaragdis, P. (2014). Deep learning for monaural speech separation. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 1562\u20131566).","DOI":"10.1109\/ICASSP.2014.6853860"},{"key":"10119_CR14","doi-asserted-by":"crossref","unstructured":"Jiang, X., Peng, X., Zheng, C., Xue, H., Zhang, Y. & Lu, Y. (2022). End-to-end neural speech coding for real-time communications. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 866\u2013870).","DOI":"10.1109\/ICASSP43922.2022.9746296"},{"key":"10119_CR15","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1016\/j.tcs.2022.08.017","volume":"941","author":"S Kantamaneni","year":"2023","unstructured":"Kantamaneni, S., Charles, A., & Babu, T. R. (2023). Speech enhancement with noise estimation and filtration using deep learning models. Theoretical Computer Science, 941, 14\u201328.","journal-title":"Theoretical Computer Science"},{"issue":"12","key":"10119_CR16","doi-asserted-by":"publisher","first-page":"1987","DOI":"10.1109\/29.45547","volume":"37","author":"S Kay","year":"1989","unstructured":"Kay, S. (1989). A fast and accurate single frequency estimator. IEEE Transactions on Acoustics, Speech, and Signal Processing, 37(12), 1987\u20131990.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"issue":"12","key":"10119_CR17","doi-asserted-by":"publisher","first-page":"1931","DOI":"10.1109\/TASLP.2014.2354236","volume":"22","author":"M Krawczyk","year":"2014","unstructured":"Krawczyk, M., & Gerkmann, T. (2014). STFT phase reconstruction in voiced speech for an improved single-channel speech enhancement. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 22(12), 1931\u20131940.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10119_CR18","doi-asserted-by":"crossref","unstructured":"Lin, J., Kalgaonkar, K., He, Q. & Lei, X. (2022, May). Speech enhancement for low bit rate speech codec. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 7777\u20137781).","DOI":"10.1109\/ICASSP43922.2022.9746670"},{"key":"10119_CR19","doi-asserted-by":"crossref","unstructured":"Luo, Y., Chen, Z., Hershey, J. R., Roux, J. L. & Mesgarani, N. (2017). Deep clustering and conventional networks for music separation: Stronger together. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 61\u201365).","DOI":"10.1109\/ICASSP.2017.7952118"},{"issue":"5","key":"10119_CR20","doi-asserted-by":"publisher","first-page":"3387","DOI":"10.1121\/1.3097493","volume":"125","author":"J Ma","year":"2009","unstructured":"Ma, J., Hu, Y., & Loizou, P. C. (2009). Objective measures for predicting speech intelligibility in noisy conditions based on new band-importance functions. The Journal of the Acoustical Society of America, 125(5), 3387\u20133405.","journal-title":"The Journal of the Acoustical Society of America"},{"issue":"5","key":"10119_CR21","doi-asserted-by":"publisher","first-page":"3387","DOI":"10.1121\/1.3097493","volume":"125","author":"J Ma","year":"2009","unstructured":"Ma, J., Yi, H., & Loizou, P. C. (2009). Objective measures for predicting speech intelligibility in noisy conditions based on new band-importance functions. The Journal of the Acoustical Society of America, 125(5), 3387\u20133405.","journal-title":"The Journal of the Acoustical Society of America"},{"key":"10119_CR22","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-023-00274-x","author":"D O\u2019Shaughnessy","year":"2023","unstructured":"O\u2019Shaughnessy, D. (2023). Review of methods for coding of speech signals. EURASIP Journal on Audio, Speech, and Music Processing. https:\/\/doi.org\/10.1186\/s13636-023-00274-x","journal-title":"EURASIP Journal on Audio, Speech, and Music Processing"},{"issue":"1","key":"10119_CR23","first-page":"8881007","volume":"2024","author":"M Pashaian","year":"2024","unstructured":"Pashaian, M., & Seyedin, S. (2024). Speech enhancement using joint DNN-NMF model learned with multi-objective frequency differential spectrum loss function. IET Signal Processing, 2024(1), 8881007.","journal-title":"IET Signal Processing"},{"key":"10119_CR24","doi-asserted-by":"publisher","first-page":"425","DOI":"10.1007\/s10772-021-09809-z","volume":"24","author":"S Raj","year":"2021","unstructured":"Raj, S., Prakasam, P., & Gupta, S. (2021). Audio signal quality enhancement using multi-layered convolutional neural network based auto encoder-decoder. International Journal of Speech Technology, 24, 425\u2013437.","journal-title":"International Journal of Speech Technology"},{"issue":"9","key":"10119_CR25","doi-asserted-by":"publisher","first-page":"4394","DOI":"10.3390\/s23094394","volume":"23","author":"C Rascon","year":"2023","unstructured":"Rascon, C. (2023). Characterization of deep learning-based speech-enhancement techniques in online audio processing applications. Sensors, 23(9), 4394.","journal-title":"Sensors"},{"key":"10119_CR26","doi-asserted-by":"crossref","unstructured":"Rix, A. W., Beerends, J. G., Hollier, M. P., & Hekstra, A. P. (2001). Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs. In Proceedings of the IEEE conference on acoustics, speech, and signal processing (ICASSP) (Vol. 2, pp. 749\u2013752).","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"10119_CR27","doi-asserted-by":"crossref","unstructured":"Shimauchi, S., Kudo, S., Koizumi, Y. & Furuya, K. I. (2017). On relationships between amplitude and phase of short-time Fourier transform. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 676\u2013680).","DOI":"10.1109\/ICASSP.2017.7952241"},{"key":"10119_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2024.105991","volume":"92","author":"NK Shukla","year":"2024","unstructured":"Shukla, N. K., Shajin, F. H., & Rajendran, R. (2024). Speech enhancement system using deep neural network optimized with Battle Royale Optimization. Biomedical Signal Processing and Control, 92, 105991.","journal-title":"Biomedical Signal Processing and Control"},{"issue":"11","key":"10119_CR29","doi-asserted-by":"publisher","first-page":"1486","DOI":"10.1016\/j.specom.2006.09.003","volume":"48","author":"S Srinivasan","year":"2006","unstructured":"Srinivasan, S., Roman, N., & Wang, D. (2006). Binary and ratio time-frequency masks for robust speech recognition. Speech Communication, 48(11), 1486\u20131501.","journal-title":"Speech Communication"},{"issue":"8","key":"10119_CR30","doi-asserted-by":"publisher","first-page":"2434","DOI":"10.1109\/JSAC.2021.3087240","volume":"39","author":"Z Weng","year":"2021","unstructured":"Weng, Z., & Qin, Z. (2021). Semantic communication systems for speech transmission. IEEE Journal on Selected Areas in Communications, 39(8), 2434\u20132444.","journal-title":"IEEE Journal on Selected Areas in Communications"},{"key":"10119_CR31","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1007\/s10772-020-09786-9","volume":"24","author":"TG Yadava","year":"2021","unstructured":"Yadava, T. G., Nagaraja, B. G., & Jayanna, H. S. (2021). Speech enhancement and encoding by combining SS-VAD and LPC. International Journal of Speech Technology, 24, 165\u2013172.","journal-title":"International Journal of Speech Technology"},{"key":"10119_CR32","doi-asserted-by":"crossref","unstructured":"Yadava, T. G., Nagaraja, B. G. & Jayanna, H. S. (2022). Performance evaluation of spectral subtraction with VAD and time-frequency filtering for speech enhancement. In Proceedings of ERCICA (pp. 407\u2013414). Springer Nature.","DOI":"10.1007\/978-981-19-5482-5_35"},{"key":"10119_CR33","doi-asserted-by":"publisher","DOI":"10.1016\/j.iswa.2023.200273","author":"TG Yadava","year":"2023","unstructured":"Yadava, T. G., Nagaraja, B. G., & Jayanna, H. S. (2023). Enhancements in encoded noisy speech data by background noise reduction. Intelligent Systems with Applications. https:\/\/doi.org\/10.1016\/j.iswa.2023.200273","journal-title":"Intelligent Systems with Applications"},{"key":"10119_CR34","doi-asserted-by":"crossref","unstructured":"Yang, K., Markovic, D., Krenn, S., Agrawal, V. & Richard, A. (2022). Audio-visual speech codecs: Rethinking audio-visual speech enhancement by re-synthesis. In IEEE\/CVF conference on computer vision and pattern recognition (pp. 8227\u20138237).","DOI":"10.1109\/CVPR52688.2022.00805"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10119-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10119-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10119-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,12]],"date-time":"2024-09-12T12:09:32Z","timestamp":1726142972000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10119-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,6]]},"references-count":34,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["10119"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10119-3","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,6]]},"assertion":[{"value":"11 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 June 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 July 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest on the manuscript.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}