{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,26]],"date-time":"2026-04-26T06:18:06Z","timestamp":1777184286097,"version":"3.51.4"},"reference-count":64,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,7,1]],"date-time":"2024-07-01T00:00:00Z","timestamp":1719792000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,1]],"date-time":"2024-07-01T00:00:00Z","timestamp":1719792000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. Comput. Sci. Technol."],"published-print":{"date-parts":[[2024,7]]},"DOI":"10.1007\/s11390-024-2934-x","type":"journal-article","created":{"date-parts":[[2024,9,20]],"date-time":"2024-09-20T06:01:55Z","timestamp":1726812115000},"page":"895-911","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Audio Enhancement for Computer Audition\u2014An Iterative Training Paradigm Using Sample Importance"],"prefix":"10.1007","volume":"39","author":[{"given":"Manuel","family":"Milling","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuo","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andreas","family":"Triantafyllopoulos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ilhan","family":"Aslan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bj\u00f6rn W.","family":"Schuller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,20]]},"reference":[{"key":"2934_CR1","volume-title":"A neural attention model for speech command recognition","author":"D C De Andrade","year":"2024","unstructured":"De Andrade D C, Leo S, Da Silva Viana M L, Bernkopf C. A neural attention model for speech command recognition. arXiv: 1808.08929, 2018. https:\/\/arxiv.org\/abs\/1808.08929, Jul. 2024."},{"key":"2934_CR2","first-page":"12449","volume-title":"Proc. the 34th Conference on Neural Information Processing Systems","author":"A Baevski","year":"2020","unstructured":"Baevski A, Zhou Y, Mohamed A, Auli M. wav2vec 2.0: A framework for self-supervised learning of speech representations. In Proc. the 34th Conference on Neural Information Processing Systems, Dec. 2020, pp.12449\u201312460."},{"issue":"9","key":"2934_CR3","doi-asserted-by":"publisher","first-page":"10745","DOI":"10.1109\/TPAMI.2023.3263585","volume":"45","author":"J Wagner","year":"2023","unstructured":"Wagner J, Triantafyllopoulos A, Wierstorf H et al. Dawn of the transformer era in speech emotion recognition: Closing the valence gap. IEEE Trans. Pattern Analysis and Machine Intelligence, 2023, 45(9): 10745\u201310759. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2023.3263585.","journal-title":"IEEE Trans. Pattern Analysis and Machine Intelligence"},{"key":"2934_CR4","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1109\/ICASSP.2019.8683434","volume-title":"Proc. the 2019 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"Z Ren","year":"2019","unstructured":"Ren Z, Kong Q, Han J, Plumbley M D, Schuller B W. Attention-based atrous convolutional neural networks: Visualisation and understanding perspectives of acoustic scenes. In Proc. the 2019 IEEE International Conference on Acoustics, Speech and Signal Processing, May 2019, pp.56\u201360. DOI: https:\/\/doi.org\/10.1109\/ICASSP.2019.8683434."},{"issue":"18","key":"2934_CR5","doi-asserted-by":"publisher","first-page":"28365","DOI":"10.1007\/s11042-021-11080-y","volume":"80","author":"S Liu","year":"2021","unstructured":"Liu S, Keren G, Parada-Cabaleiro E, Schuller B. NHANS: A neural network-based toolkit for in-the-wild audio enhancement. Multimedia Tools and Applications, 2021, 80(18): 28365\u201328389. DOI: https:\/\/doi.org\/10.1007\/s11042-021-11080-y.","journal-title":"Multimedia Tools and Applications"},{"key":"2934_CR6","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1016\/j.csl.2018.04.003","volume":"52","author":"C Spille","year":"2018","unstructured":"Spille C, Kollmeier B, Meyer B T. Comparing human and automatic speech recognition in simple and complex acoustic scenes. Computer Speech & Language, 2018, 52: 123\u2013140. DOI: https:\/\/doi.org\/10.1016\/j.csl.2018.04.003.","journal-title":"Computer Speech & Language"},{"key":"2934_CR7","first-page":"1691","volume-title":"Proc. the 20th Annual Conf. International Speech Communication Association","author":"A Triantafyllopoulos","year":"2019","unstructured":"Triantafyllopoulos A, Keren G, Wagner J et al. Towards robust speech emotion recognition using deep residual networks for speech enhancement. In Proc. the 20th Annual Conf. International Speech Communication Association, Sept. 2019, pp.1691\u20131695."},{"key":"2934_CR8","first-page":"3087","volume-title":"Proc. the 21st Annual Conference of the International Speech Communication Association","author":"S Liu","year":"2020","unstructured":"Liu S, Triantafyllopoulos A, Ren Z et al. Towards speech robustness for acoustic scene classification. In Proc. the 21st Annual Conference of the International Speech Communication Association, Oct. 2020, pp.3087\u20133091."},{"key":"2934_CR9","first-page":"2613","volume-title":"Proc. the 20th Annual Conference of the International Speech Communication Association","author":"D S Park","year":"2019","unstructured":"Park D S, Chan W, Zhang Y, Chiu C C, Zoph B, Cubuk E D, Le Q V. SpecAugment: A simple data augmentation method for automatic speech recognition. In Proc. the 20th Annual Conference of the International Speech Communication Association, Sept. 2019, pp.2613\u20132617."},{"key":"2934_CR10","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1007\/978-3-319-22482-4_11","volume-title":"Proc. the 12th Int. Conf. Latent Variable Analysis and Signal Separation","author":"F Weninger","year":"2015","unstructured":"Weninger F, Erdogan H, Watanabe S et al. Speech enhancement with LSTM recurrent neural networks and its application to noise-robust ASR. In Proc. the 12th Int. Conf. Latent Variable Analysis and Signal Separation, Aug. 2015, pp.91\u201399. DOI: https:\/\/doi.org\/10.1007\/978-3-319-22482-4_11."},{"key":"2934_CR11","doi-asserted-by":"publisher","first-page":"7009","DOI":"10.1109\/ICASSP40776.2020.9053266","volume-title":"Proc. the 2020 IEEE Int. Conf. Acoustics, Speech and Signal Processing","author":"K Kinoshita","year":"2020","unstructured":"Kinoshita K, Ochiai T, Delcroix M, Nakatani T. Improving noise robust automatic speech recognition with singlechannel time-domain enhancement network. In Proc. the 2020 IEEE Int. Conf. Acoustics, Speech and Signal Processing, May 2020, pp.7009\u20137013. DOI: https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9053266."},{"key":"2934_CR12","doi-asserted-by":"publisher","first-page":"482","DOI":"10.1109\/ASRU.2015.7404834","volume-title":"Proc. the 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU)","author":"S Sivasankaran","year":"2015","unstructured":"Sivasankaran S, Nugraha A A, Vincent E, Morales-Cordovilla J A, Dalmia S, Illina I, Liutkus A. Robust ASR using neural network based speech enhancement and feature simulation. In Proc. the 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), Dec. 2015, pp.482\u2013489. DOI: https:\/\/doi.org\/10.1109\/ASRU.2015.7404834."},{"key":"2934_CR13","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1109\/ASRU46091.2019.9003785","volume-title":"Proc. the 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","author":"C Zoril\u0103","year":"2019","unstructured":"Zoril\u0103 C, Boeddeker C, Doddipatla R, Haeb-Umbach R. An investigation into the effectiveness of enhancement in ASR training and test for chime-5 dinner party transcription. In Proc. the 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), Dec. 2019, pp.47\u201353. DOI: https:\/\/doi.org\/10.1109\/ASRU46091.2019.9003785."},{"key":"2934_CR14","first-page":"5418","volume-title":"Proc. the 23rd Annual Conference of the International Speech Communication Association","author":"K Iwamoto","year":"2022","unstructured":"Iwamoto K, Ochiai T, Delcroix M et al. How bad are artifacts?: Analyzing the impact of speech enhancement errors on ASR. In Proc. the 23rd Annual Conference of the International Speech Communication Association, Sept. 2022, pp.5418\u20135422."},{"issue":"4","key":"2934_CR15","doi-asserted-by":"publisher","first-page":"796","DOI":"10.1109\/TASLP.2016.2528171","volume":"24","author":"Z Q Wang","year":"2016","unstructured":"Wang Z Q, Wang D L. A joint training framework for robust automatic speech recognition. IEEE\/ACM Trans. Audio, Speech, and Language Processing, 2016, 24(4): 796\u2013806. DOI: https:\/\/doi.org\/10.1109\/TASLP.2016.2528171.","journal-title":"IEEE\/ACM Trans. Audio, Speech, and Language Processing"},{"key":"2934_CR16","first-page":"3571","volume-title":"Proc. the 16th Annual Conference of the International Speech Communication Association","author":"A Narayanan","year":"2015","unstructured":"Narayanan A, Misra A, Chin K K. Large-scale, sequence-discriminative, joint adaptive training for masking-based robust ASR. In Proc. the 16th Annual Conference of the International Speech Communication Association, Sept. 2015, pp.3571\u20133575."},{"key":"2934_CR17","first-page":"497","volume-title":"Proc. the 2021 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC)","author":"D Ma","year":"2021","unstructured":"Ma D, Hou N N, Pham V T et al. Multitask-based joint learning approach to robust ASR for radio communication speech. In Proc. the 2021 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC), Dec. 2021, pp.497\u2013502."},{"key":"2934_CR18","first-page":"3274","volume-title":"Proc. the 16th Annual Conference of the International Speech Communication Association","author":"Z Chen","year":"2015","unstructured":"Chen Z, Watanabe S, Erdogan H et al. Speech enhancement and recognition using multi-task learning of long short-term memory recurrent neural networks. In Proc. the 16th Annual Conference of the International Speech Communication Association, Sept. 2015, pp.3274\u20133278."},{"key":"2934_CR19","first-page":"491","volume-title":"Proc. the 20th Annual Conference of the International Speech Communication Association","author":"B Liu","year":"2019","unstructured":"Liu B, Nie S, Liang S, Liu W J, Yu M, Chen L W, Peng S Y, Li C L. Jointly adversarial enhancement training for robust end-to-end speech recognition. In Proc. the 20th Annual Conference of the International Speech Communication Association, Sept. 2019, pp.491\u2013495."},{"issue":"1","key":"2934_CR20","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1186\/s13636-021-00215-6","volume":"2021","author":"L J Li","year":"2021","unstructured":"Li L J, Kang Y K, Shi Y C, K\u00fcrzinger L, Watzel T, Rigoll G. Adversarial joint training with self-attention mechanism for robust end-to-end speech recognition. EURASIP Journal on Audio, Speech, and Music Processing, 2021, 2021(1): 26. DOI: https:\/\/doi.org\/10.1186\/S13636-021-00215-6.","journal-title":"EURASIP Journal on Audio, Speech, and Music Processing"},{"key":"2934_CR21","volume-title":"Joint training of speech enhancement and self-supervised model for noise-robust ASR","author":"Q S Zhu","year":"2024","unstructured":"Zhu Q S, Zhang J, Zhang Z Q, Dai L R. Joint training of speech enhancement and self-supervised model for noise-robust ASR. arXiv: 2205.13293, 2022. https:\/\/arxiv.org\/abs\/2205.13293, Jul. 2024."},{"key":"2934_CR22","doi-asserted-by":"publisher","first-page":"6773","DOI":"10.1109\/ICASSP39728.2021.9414117","volume-title":"Proc. the 2021 IEEE Int. Conf. Acoustics, Speech and Signal Processing","author":"C Kim","year":"2021","unstructured":"Kim C, Garg A, Gowda D, Mun S, Han C. Streaming end-to-end speech recognition with jointly trained neural feature enhancement. In Proc. the 2021 IEEE Int. Conf. Acoustics, Speech and Signal Processing, Jun. 2021, pp.6773\u20136777. DOI: https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9414117."},{"key":"2934_CR23","doi-asserted-by":"publisher","unstructured":"C\u00e1mbara G, L\u00f3pez F, Bonet D et al. TASE: Task-aware speech enhancement for wake-up word detection in voice assistants. Applied Sciences, 2022, 12 (4): Article No. 1974. DOI: https:\/\/doi.org\/10.3390\/app12041974.","DOI":"10.3390\/app12041974"},{"key":"2934_CR24","volume-title":"A monaural speech enhancement method for robust small-footprint keyword spotting","author":"Y Gu","year":"2024","unstructured":"Gu Y, Du Z H, Zhang H, Zhang X. A monaural speech enhancement method for robust small-footprint keyword spotting. arXiv: 1906.08415, 2019. https:\/\/arxiv.org\/abs\/1906.08415, Jul. 2024."},{"key":"2934_CR25","first-page":"4098","volume-title":"Proc. the 21st Annual Conference of the International Speech Communication Association","author":"H Zhou","year":"2020","unstructured":"Zhou H, Du J, Tu Y H, Lee C H. Using speech enhancement preprocessing for speech emotion recognition in realistic noisy conditions. In Proc. the 21st Annual Conference of the International Speech Communication Association, Oct. 2020, pp.4098\u20134102."},{"key":"2934_CR26","first-page":"201","volume-title":"Proc. the 22nd Annual Conference of the International Speech Communication Association","author":"S W Fu","year":"2021","unstructured":"Fu S W, Yu C, Hsieh T A, Plantinga P, Ravanelli M, Lu X, Tsao Y. MetricGAN+: An improved version of metric-GAN for speech enhancement. In Proc. the 22nd Annual Conference of the International Speech Communication Association, Aug. 30 -Sept. 3 2021, pp.201\u2013205."},{"key":"2934_CR27","first-page":"2008","volume-title":"Proc. the 24th Annual Conference of the International Speech Communication Association","author":"H Schr\u00f6ter","year":"2023","unstructured":"Schr\u00f6ter H, Rosenkranz T, Escalante-B A N, Maier A. DeepFilterNet: Perceptually motivated real-time speech enhancement. In Proc. the 24th Annual Conference of the International Speech Communication Association, Aug. 2023, pp.2008\u20132009."},{"key":"2934_CR28","first-page":"146","volume-title":"Proc. the 9th ISCA Speech Synthesis Workshop","author":"C Valentini-Botinhao","year":"2016","unstructured":"Valentini-Botinhao C, Wang X, Takaki S, Yamagishi J. Investigating RNN-based speech enhancement methods for noise-robust Text-to-Speech. In Proc. the 9th ISCA Speech Synthesis Workshop, Sept. 2016, pp.146\u2013152."},{"key":"2934_CR29","doi-asserted-by":"publisher","first-page":"9271","DOI":"10.1109\/ICASSP43922.2022.9747230","volume-title":"Proc. the 2022 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"H Dubey","year":"2022","unstructured":"Dubey H, Gopal V, Cutler R, Aazami A, Matusevych S, Braun S, Eskimez S E, Thakker M, Yoshioka T, Gamper H, Aichner R. ICASSP 2022 deep noise suppression challenge. In Proc. the 2022 IEEE International Conference on Acoustics, Speech and Signal Processing, May 2022, pp.9271\u20139275. DOI: https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9747230."},{"key":"2934_CR30","first-page":"107","volume-title":"Proc. the 32nd Conference on Neural Information Processing Systems","author":"L Le","year":"2018","unstructured":"Le L, Patterson A, White M. Supervised autoencoders: Improving generalization performance with unsupervised regularizers. In Proc. the 32nd Conference on Neural Information Processing Systems, Dec. 2018, pp.107\u2013117."},{"key":"2934_CR31","first-page":"137","volume-title":"Proc. the 20th Annual Conference on Neural Information Processing Systems","author":"S Ben-David","year":"2006","unstructured":"Ben-David S, Blitzer J, Crammer K, Pereira F. Analysis of representations for domain adaptation. In Proc. the 20th Annual Conference on Neural Information Processing Systems, Dec. 2006, pp.137\u2013144."},{"key":"2934_CR32","doi-asserted-by":"publisher","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","volume-title":"Proc. the 18th International Conference on Medical Image Computing and Computer-Assisted Intervention","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger O, Fischer P, Brox T. U-net: Convolutional networks for biomedical image segmentation. In Proc. the 18th International Conference on Medical Image Computing and Computer-Assisted Intervention, Oct. 2015, pp.234\u2013241. DOI: https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28."},{"key":"2934_CR33","volume-title":"Proc. the 7th International Conference on Learning Representations","author":"H S Choi","year":"2018","unstructured":"Choi H S, Kim J H, Huh J, Kim A, Ha J W, Lee K. Phase-aware speech enhancement with deep complex u-net. In Proc. the 7th International Conference on Learning Representations, May 2018."},{"key":"2934_CR34","first-page":"334","volume-title":"Proc. the 19th International Society for Music Information Retrieval Conference","author":"D Stoller","year":"2018","unstructured":"Stoller D, Ewert S, Dixon S. Wave-U-Net: A multi-scale neural network for end-to-end audio source separation. In Proc. the 19th International Society for Music Information Retrieval Conference, Sept. 2018, pp.334\u2013340."},{"key":"2934_CR35","volume-title":"Speech commands: A dataset for limited-vocabulary speech recognition","author":"P Warden","year":"2024","unstructured":"Warden P. Speech commands: A dataset for limited-vocabulary speech recognition. arXiv: 1804.03209, 2018. https:\/\/arxiv.org\/abs\/1804.03209, Jul. 2024."},{"key":"2934_CR36","doi-asserted-by":"publisher","first-page":"421","DOI":"10.1109\/ICASSP.2017.7952190","volume-title":"Proc. the 2017 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"W Dai","year":"2017","unstructured":"Dai W, Dai C, Qu S H, Li J C, Das S. Very deep convolutional neural networks for raw waveforms. In Proc. the 2017 IEEE International Conference on Acoustics, Speech and Signal Processing, Mar. 2017, pp.421\u2013425. DOI: https:\/\/doi.org\/10.1109\/ICASSP.2017.7952190."},{"issue":"8","key":"2934_CR37","doi-asserted-by":"publisher","first-page":"1018","DOI":"10.3390\/sym11081018","volume":"11","author":"D Wang","year":"2019","unstructured":"Wang D, Wang X, Lv S. An overview of end-to-end automatic speech recognition. Symmetry, 2019, 11(8): 1018. DOI: https:\/\/doi.org\/10.3390\/sym11081018.","journal-title":"Symmetry"},{"key":"2934_CR38","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"W N Hsu","year":"2021","unstructured":"Hsu W N, Bolte B, Tsai Y H H, Lakhotia K, Salakhutdinov R, Mohamed A. HuBERT: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio, Speech, and Language Processing, 2021, 29: 3451\u20133460. DOI: https:\/\/doi.org\/10.1109\/TASLP.2021.3122291.","journal-title":"IEEE\/ACM Trans. Audio, Speech, and Language Processing"},{"key":"2934_CR39","volume-title":"XLS-R: Self-supervised cross-lingual speech representation learning at scale","author":"A Babu","year":"2024","unstructured":"Babu A, Wang C H, Tjandra A et al. XLS-R: Self-supervised cross-lingual speech representation learning at scale. arXiv: 2111.09296, 2021. https:\/\/arxiv.org\/abs\/2111.09296, Jul. 2024."},{"issue":"11","key":"2934_CR40","doi-asserted-by":"publisher","first-page":"4037","DOI":"10.1109\/TPAMI.2020.2992393","volume":"43","author":"L Jing","year":"2021","unstructured":"Jing L, Tian Y. Self-supervised visual feature learning with deep neural networks: A survey. IEEE Trans. Pattern Analysis and Machine Intelligence, 2021, 43(11): 4037\u20134058. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2020.2992393.","journal-title":"IEEE Trans. Pattern Analysis and Machine Intelligence"},{"issue":"1","key":"2934_CR41","doi-asserted-by":"publisher","first-page":"857","DOI":"10.1109\/TKDE.2021.3090866","volume":"35","author":"X Liu","year":"2023","unstructured":"Liu X, Zhang F, Hou Z Y, Mian L, Wang Z, Zhang J, Tang J. Self-supervised learning: Generative or contrastive. IEEE Trans. Knowledge and Data Engineering, 2023, 35(1): 857\u2013876. DOI: https:\/\/doi.org\/10.1109\/TKDE.2021.3090866.","journal-title":"IEEE Trans. Knowledge and Data Engineering"},{"key":"2934_CR42","first-page":"173","volume-title":"Proc. the 33rd International Conference on Machine Learning","author":"A Amodei","year":"2016","unstructured":"Amodei A, Ananthanarayanan S, Anubhai R et al. Deep speech 2: End-to-end speech recognition in English and mandarin. In Proc. the 33rd International Conference on Machine Learning, Jun. 2016, pp.173\u2013182."},{"key":"2934_CR43","first-page":"6391","volume-title":"Proc. the 32nd International Conference on Neural Information Processing Systems","author":"H Li","year":"2018","unstructured":"Li H, Xu Z, Taylor G, Studer C, Goldstein T. Visualizing the loss landscape of neural nets. In Proc. the 32nd International Conference on Neural Information Processing Systems, Dec. 2018, pp.6391\u20136401."},{"issue":"8","key":"2934_CR44","doi-asserted-by":"publisher","first-page":"875","DOI":"10.1007\/s11265-020-01518-1","volume":"92","author":"N H Zheng","year":"2020","unstructured":"Zheng N H, Shi Y P, Rong W C, Kang Y Y. Effects of skip connections in CNN-based architectures for speech enhancement. Journal of Signal Processing Systems, 2020, 92(8): 875\u2013884. DOI: https:\/\/doi.org\/10.1007\/s11265-020-01518-1.","journal-title":"Journal of Signal Processing Systems"},{"key":"2934_CR45","volume-title":"Deep speech: Scaling up end-to-end speech recognition","author":"A Hannun","year":"2024","unstructured":"Hannun A, Case C, Casper J et al. Deep speech: Scaling up end-to-end speech recognition. arXiv: 1412.5567, 2014. https:\/\/arxiv.org\/abs\/1412.5567, Jul. 2024."},{"issue":"1","key":"2934_CR46","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1186\/s13636-014-0047-0","volume":"2015","author":"S Yin","year":"2015","unstructured":"Yin S, Liu C, Zhang Z, Lin Y, Wang D, Tejedor J, Zheng F, Li Y. Noisy training for deep neural networks in speech recognition. EURASIP Journal on Audio, Speech, and Music Processing, 2015, 2015(1): 2. DOI: https:\/\/doi.org\/10.1186\/s13636-014-0047-0.","journal-title":"EURASIP Journal on Audio, Speech, and Music Processing"},{"key":"2934_CR47","doi-asserted-by":"publisher","first-page":"5719","DOI":"10.1109\/ICASSP.2018.8462137","volume-title":"Proc. the 2018 IEEE Int. Conf. Acoustics, Speech and Signal Processing","author":"J Kim","year":"2018","unstructured":"Kim J, El-Khamy M, Lee J. Bridgenets: Student-teacher transfer learning based on recursive neural networks and its application to distant speech recognition. In Proc. the 2018 IEEE Int. Conf. Acoustics, Speech and Signal Processing, Apr. 2018, pp.5719\u20135723. DOI: https:\/\/doi.org\/10.1109\/ICASSP.2018.8462137."},{"key":"2934_CR48","doi-asserted-by":"publisher","first-page":"268","DOI":"10.1109\/ASRU46091.2019.9003776","volume-title":"Proc. the 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","author":"Z Meng","year":"2019","unstructured":"Meng Z, Li J, Gaur Y, Gong Y. Domain adaptation via teacher-student learning for end-to-end speech recognition. In Proc. the 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), Dec. 2019, pp.268\u2013275. DOI: https:\/\/doi.org\/10.1109\/ASRU46091.2019.9003776."},{"issue":"5","key":"2934_CR49","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1145\/3129340","volume":"61","author":"B W Schuller","year":"2018","unstructured":"Schuller B W. Speech emotion recognition: Two decades in a nutshell, benchmarks, and ongoing trends. Communications of the ACM, 2018, 61(5): 90\u201399. DOI: https:\/\/doi.org\/10.1145\/3129340.","journal-title":"Communications of the ACM"},{"issue":"4","key":"2934_CR50","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","volume":"42","author":"C Busso","year":"2008","unstructured":"Busso C, Bulut M, Lee C C et al. IEMOCAP: Interactive emotional dyadic motion capture database. Language Resources and Evaluation, 2008, 42(4): 335\u2013359. DOI: https:\/\/doi.org\/10.1007\/s10579-008-9076-6.","journal-title":"Language Resources and Evaluation"},{"key":"2934_CR51","doi-asserted-by":"publisher","first-page":"397","DOI":"10.1109\/SLT48900.2021.9383542","volume-title":"Proc. the 2021 IEEE Spoken Language Technology Workshop (SLT)","author":"A Baird","year":"2021","unstructured":"Baird A, Amiriparian S, Milling M, Schuller B W. Emotion recognition in public speaking scenarios utilising an LSTM-RNN approach with attention. In Proc. the 2021 IEEE Spoken Language Technology Workshop (SLT), Jan. 2021, pp.397\u2013402. DOI: https:\/\/doi.org\/10.1109\/SLT48900.2021.9383542."},{"key":"2934_CR52","doi-asserted-by":"publisher","first-page":"837269","DOI":"10.3389\/fcomp.2022.837269","volume":"4","author":"M Milling","year":"2022","unstructured":"Milling M, Baird A, Bartl-Pokorny K D, Liu S, Alcorn A M, Shen J, Tavassoli T, Ainger E, Pellicano E, Pantic M, Cummins N, Schuller B W. Evaluating the impact of voice activity detection on speech emotion recognition for autistic children. Frontiers in Computer Science, 2022, 4: 837269. DOI: https:\/\/doi.org\/10.3389\/fcomp.2022.837269.","journal-title":"Frontiers in Computer Science"},{"key":"2934_CR53","first-page":"3935","volume-title":"Proc. the 20th Annual Conference of the International Speech Communication Association","author":"C Oates","year":"2019","unstructured":"Oates C, Triantafyllopoulos A, Steiner I, Schuller B W. Robust speech emotion recognition under different encoding conditions. In Proc. the 20th Annual Conference of the International Speech Communication Association, Sept. 2019, pp.3935\u20133939."},{"key":"2934_CR54","volume-title":"ConcealNet: An end-to-end neural network for packet loss concealment in deep speech emotion recognition","author":"M M Mohamed","year":"2024","unstructured":"Mohamed M M, Schuller B W. ConcealNet: An end-to-end neural network for packet loss concealment in deep speech emotion recognition. arXiv: 2005.07777, 2020. https:\/\/arxiv.org\/abs\/2005.07777, Jul. 2024."},{"key":"2934_CR55","doi-asserted-by":"publisher","first-page":"1072479","DOI":"10.3389\/fcomp.2023.1072479","volume":"5","author":"A Triantafyllopoulos","year":"2023","unstructured":"Triantafyllopoulos A, Reichel U, Liu S, Huber S, Eyben F, Schuller B W. Multistage linguistic conditioning of convolutional layers for speech emotion recognition. Frontiers in Computer Science, 2023, 5: 1072479. DOI: https:\/\/doi.org\/10.3389\/fcomp.2023.1072479.","journal-title":"Frontiers in Computer Science"},{"key":"2934_CR56","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1109\/BalkanCom53780.2021.9593258","volume-title":"Proc. the 2021 In. Balkan Conf. Communications and Networking (BalkanCom)","author":"D Bajovic","year":"2021","unstructured":"Bajovic D, Bakhtiarnia A, Bravos G et al. MARVEL: Multimodal extreme scale data analytics for smart cities environments. In Proc. the 2021 In. Balkan Conf. Communications and Networking (BalkanCom), Sept. 2021, pp.143\u2013147. DOI: https:\/\/doi.org\/10.1109\/BalkanCom53780.2021.9593258."},{"key":"2934_CR57","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1109\/ICASSP40776.2020.9053274","volume-title":"Proc. the 2020 IEEE Int. Conf. Acoustics, Speech and Signal Processing","author":"M D McDonnell","year":"2020","unstructured":"McDonnell M D, Gao W. Acoustic scene classification using deep residual networks with late fusion of separated high and low frequency paths. In Proc. the 2020 IEEE Int. Conf. Acoustics, Speech and Signal Processing, May 2020, pp.141\u2013145. DOI: https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9053274."},{"key":"2934_CR58","first-page":"56","volume-title":"Proc. the 5th Workshop on Detection and Classification of Acoustic Scenes and Events 2020 (DCASE2020)","author":"T Heittola","year":"2020","unstructured":"Heittola T, Mesaros A, Virtanen T. Acoustic scene classification in DCASE 2020 challenge: Generalization across devices and low complexity solutions. In Proc. the 5th Workshop on Detection and Classification of Acoustic Scenes and Events 2020 (DCASE2020), Nov. 2020, pp.56\u201360."},{"key":"2934_CR59","doi-asserted-by":"publisher","first-page":"369","DOI":"10.1145\/1143844.1143891","volume-title":"Proc. the 23rd International Conference on Machine Learning","author":"A Graves","year":"2006","unstructured":"Graves A, Fern\u00e1ndez S, Gomez F J, Schmidhuber J. Connectionist temporal classification: Labelling unsegmented sequence data with recurrent neural networks. In Proc. the 23rd International Conference on Machine Learning, Jun. 2006, pp.369\u2013376."},{"key":"2934_CR60","doi-asserted-by":"publisher","first-page":"5206","DOI":"10.1109\/ICASSP.2015.7178964","volume-title":"Proc. the 2015 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"V Panayotov","year":"2015","unstructured":"Panayotov V, Chen G G, Povey D, Khudanpur S. Librispeech: An ASR corpus based on public domain audio books. In Proc. the 2015 IEEE International Conference on Acoustics, Speech and Signal Processing, Apr. 2015, pp.5206\u20135210. DOI: https:\/\/doi.org\/10.1109\/ICASSP.2015.7178964."},{"key":"2934_CR61","volume-title":"Towards selection of text-to-speech data to augment ASR training","author":"S Liu","year":"2024","unstructured":"Liu S, Sar\u0131 L, Wu C Y, Keren G, Shangguan Y, Mahadeokar J, Kalinli O. Towards selection of text-to-speech data to augment ASR training. arXiv: 2306.00998, 2023. https:\/\/arxiv.org\/abs\/2306.00998, Jul. 2024."},{"issue":"2","key":"2934_CR62","doi-asserted-by":"publisher","first-page":"341","DOI":"10.1007\/s10579-019-09450-y","volume":"54","author":"E Parada-Cabaleiro","year":"2020","unstructured":"Parada-Cabaleiro E, Costantini G, Batliner A, Schmitt M, Schuller B W. DEMoS: An Italian emotional speech corpus: Elicitation methods, machine learning, and perception. Language Resources and Evaluation, 2020, 54(2): 341\u2013383. DOI: https:\/\/doi.org\/10.1007\/s10579-019-09450-y.","journal-title":"Language Resources and Evaluation"},{"key":"2934_CR63","doi-asserted-by":"publisher","first-page":"7184","DOI":"10.1109\/ICASSP40776.2020.9054087","volume-title":"Proc. the 2020 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"Z Ren","year":"2020","unstructured":"Ren Z, Baird A, Han J, Zhang Z, Schuller B. Generating and protecting against adversarial attacks for deep speech-based emotion recognition models. In Proc. the 2020 IEEE International Conference on Acoustics, Speech and Signal Processing, May 2020, pp.7184\u20137188. DOI: https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9054087."},{"key":"2934_CR64","doi-asserted-by":"publisher","first-page":"626","DOI":"10.1109\/ICASSP39728.2021.9415085","volume-title":"Proc. the 2021 IEEE International Conference on Acoustics, Speech and Signal Processing","author":"S S Wang","year":"2021","unstructured":"Wang S S, Mesaros A, Heittola T, Virtanen T. A curated dataset of urban scenes for audio-visual scene analysis. In Proc. the 2021 IEEE International Conference on Acoustics, Speech and Signal Processing, Jun. 2021, pp.626\u2013630. DOI: https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9415085."}],"container-title":["Journal of Computer Science and Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-024-2934-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11390-024-2934-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11390-024-2934-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,20]],"date-time":"2024-09-20T06:12:10Z","timestamp":1726812730000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11390-024-2934-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7]]},"references-count":64,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,7]]}},"alternative-id":["2934"],"URL":"https:\/\/doi.org\/10.1007\/s11390-024-2934-x","relation":{},"ISSN":["1000-9000","1860-4749"],"issn-type":[{"value":"1000-9000","type":"print"},{"value":"1860-4749","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7]]},"assertion":[{"value":"26 October 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 June 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 September 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"<b>Conflict of Interest<\/b> The authors declare that they have no conflict of interest.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics"}}]}}