{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,20]],"date-time":"2026-01-20T14:20:06Z","timestamp":1768918806308,"version":"3.49.0"},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T00:00:00Z","timestamp":1738713600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T00:00:00Z","timestamp":1738713600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62471249"],"award-info":[{"award-number":["62471249"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62071242"],"award-info":[{"award-number":["62071242"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2022M711693"],"award-info":[{"award-number":["2022M711693"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Humanities and Social Science Foundation of China Ministry of Education","award":["24YJC740076"],"award-info":[{"award-number":["24YJC740076"]}]},{"name":"Jiangsu Government Scholarship for Overseas Studies","award":["NJUPT-2023-002"],"award-info":[{"award-number":["NJUPT-2023-002"]}]},{"name":"DFG (German Research Foundation) Reinhart Koselleck-Project AUDI0NOMOUS","award":["442218748"],"award-info":[{"award-number":["442218748"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1007\/s10772-025-10171-7","type":"journal-article","created":{"date-parts":[[2025,2,5]],"date-time":"2025-02-05T14:50:08Z","timestamp":1738767008000},"page":"129-139","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Request and complaint recognition in call-center speech using a pointwise-convolution recurrent network"],"prefix":"10.1007","volume":"28","author":[{"given":"Zhipeng","family":"Yin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4017-5919","authenticated-orcid":false,"given":"Xinzhou","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bj\u00f6rn","family":"Schuller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,2,5]]},"reference":[{"key":"10171_CR1","unstructured":"Ardila, R., Branson, M., Davis, K., Henretty, M., Kohler, M., Meyer, J., Morais, R., Saunders, L., Tyers, F. M., & Weber, G. (2020). Common voice: A massively-multilingual speech corpus. In Proceedings of the conference on language resources and evaluation (LREC) (pp. 4211\u20134215)."},{"key":"10171_CR2","doi-asserted-by":"crossref","unstructured":"Baby, A., Joseph, G., & Singh, S. (2024). Robust speaker personalisation using generalized low-rank adaptation for automatic speech recognition. In Proceedings IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 11381\u201311385). IEEE.","DOI":"10.1109\/ICASSP48485.2024.10446630"},{"key":"10171_CR3","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., & Auli, M. (2020). wav2vec 2.0: A framework for self-supervised learning of speech representations. Proceedings of the Advances in Neural Information Processing Systems, 33, 12449\u201312460.","journal-title":"Proceedings of the Advances in Neural Information Processing Systems"},{"issue":"1","key":"10171_CR4","first-page":"9667377","volume":"2024","author":"JF Bauer","year":"2024","unstructured":"Bauer, J. F., Gerczuk, M., Schindler-Gmelch, L., Amiriparian, S., Ebert, D. D., Krajewski, J., Schuller, B., Berking, M., et al. (2024). Validation of machine learning-based assessment of major depressive disorder from paralinguistic speech characteristics in routine care. Depression and Anxiety, 2024(1), 9667377.","journal-title":"Depression and Anxiety"},{"key":"10171_CR5","unstructured":"Bommasani, R., Hudson, D.A., Adeli, E., Altman, R., Arora, S., Arx, S., Bernstein, M.S., Bohg, J., Bosselut, A., Brunskill, E., et al. (2021). On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258."},{"key":"10171_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Z.-C., Fu, C.-L., Liu, C.-Y., Li, S.-W. D., & Lee, H.-Y. (2023). Exploring efficient-tuning methods in self-supervised speech models. In Proceedings of the IEEE spoken language technology workshop (SLT) (pp. 1120\u20131127). IEEE.","DOI":"10.1109\/SLT54892.2023.10023274"},{"issue":"6","key":"10171_CR7","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"S Chen","year":"2022","unstructured":"Chen, S., Wang, C., Chen, Z., Wu, Y., Liu, S., Chen, Z., Li, J., Kanda, N., Yoshioka, T., Xiao, X., et al. (2022). WavLM: Large-scale self-supervised pre-training for full stack speech processing. IEEE Journal of Selected Topics in Signal Processing, 16(6), 1505\u20131518.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"10171_CR8","doi-asserted-by":"crossref","unstructured":"Deschamps-Berger, T., Lamel, L., & Devillers, L. (2021). End-to-end speech emotion recognition: Challenges of real-life emergency call centers data recordings. In Proceedings international conference on affective computing and intelligent interaction (ACII) (pp. 1\u20138). IEEE.","DOI":"10.1109\/ACII52823.2021.9597419"},{"key":"10171_CR9","doi-asserted-by":"crossref","unstructured":"Deschamps-Berger, T., Lamel, L., & Devillers, L. (2022). Investigating transformer encoders and fusion strategies for speech emotion recognition in emergency call center conversations. In Proceedings of the international conference on multimodal interaction (pp. 144\u2013153).","DOI":"10.1145\/3536220.3558038"},{"key":"10171_CR10","doi-asserted-by":"crossref","unstructured":"Deschamps-Berger, T., Lamel, L., & Devillers, L. (2023). Exploring attention mechanisms for multimodal emotion recognition in an emergency call center corpus. In Proceedings IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE.","DOI":"10.1109\/ICASSP49357.2023.10096112"},{"key":"10171_CR11","doi-asserted-by":"crossref","unstructured":"Deschamps-Berger, T., Lamel, L., & Devillers, L. (2023). Multiscale contextual learning for speech emotion recognition in emergency call center conversations. In Proceedings of the international conference on multimodal interaction (pp. 337\u2013343).","DOI":"10.1145\/3610661.3616189"},{"key":"10171_CR12","unstructured":"Deshpande, G., & Schuller, B. W. (2020). Audio, speech, language, & signal processing for COVID-19: A comprehensive overview. arXiv preprint arXiv:2011.14445."},{"key":"10171_CR13","doi-asserted-by":"crossref","unstructured":"Dong, Z., Zhang, Z., Xu, W., Han, J., Ou, J., & Schuller, B. W. (2024). HAFFormer: A hierarchical attention-free framework for Alzheimer\u2019s disease detection from spontaneous speech. In Proceedings IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 11246\u201311250). IEEE.","DOI":"10.1109\/ICASSP48485.2024.10446795"},{"key":"10171_CR14","doi-asserted-by":"crossref","unstructured":"English, P. C., Shams, E. A., Kelleher, J. D., & Carson-Berndsen, J. (2024). Following the embedding: Identifying transition phenomena in wav2vec 2.0 representations of speech audio. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 6685\u20136689). IEEE.","DOI":"10.1109\/ICASSP48485.2024.10446494"},{"issue":"1","key":"10171_CR15","doi-asserted-by":"publisher","first-page":"78","DOI":"10.1159\/000515346","volume":"5","author":"G Fagherazzi","year":"2021","unstructured":"Fagherazzi, G., Fischer, A., Ismael, M., & Despotovic, V. (2021). Voice for health: The use of vocal biomarkers from research to clinical practice. Digital Biomarkers, 5(1), 78\u201388.","journal-title":"Digital Biomarkers"},{"key":"10171_CR17","doi-asserted-by":"crossref","unstructured":"Feng, T., & Narayanan, S. (2024). Foundation model assisted automatic speech emotion recognition: Transcribing, annotating, and augmenting. In Proceedings IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 12116\u201312120). IEEE.","DOI":"10.1109\/ICASSP48485.2024.10448130"},{"key":"10171_CR16","doi-asserted-by":"crossref","unstructured":"Feng, Y., & Devillers, L. (2023). End-to-end continuous speech emotion recognition in real-life customer service call center conversations. In Proceedings of the international conference on affective computing and intelligent interaction workshops and demos (pp. 1\u20138). IEEE.","DOI":"10.1109\/ACIIW59127.2023.10388120"},{"key":"10171_CR18","doi-asserted-by":"crossref","unstructured":"Gao, Y., Shi, H., Chu, C., & Kawahara, T. (2024). Enhancing two-stage finetuning for speech emotion recognition using adapters. In Proceedings IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 11316\u201311320). IEEE.","DOI":"10.1109\/ICASSP48485.2024.10446645"},{"key":"10171_CR19","doi-asserted-by":"crossref","unstructured":"Gong, Y., Chung, Y.-A., & Glass, J. (2021). AST: Audio spectrogram transformer. In Proceedings of the annual conference of the international speech communication association (INTERSPEECH) (pp. 571\u2013575).","DOI":"10.21437\/Interspeech.2021-698"},{"key":"10171_CR20","doi-asserted-by":"crossref","unstructured":"Gr\u00f3sz, T., Porjazovski, D., Getman, Y., Kadiri, S., & Kurimo, M. (2022). Wav2vec2-based paralinguistic systems to recognise vocalised emotions and stuttering. In Proceedings of the ACM international conference on multimedia (pp. 7026\u20137029).","DOI":"10.1145\/3503161.3551572"},{"key":"10171_CR21","doi-asserted-by":"crossref","unstructured":"Guo, Y., Huang, H., Chen, X., Zhao, H., & Wang, Y. (2024). Audio deepfake detection with self-supervised WavLM and multi-fusion attentive classifier. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 12702\u201312706). IEEE.","DOI":"10.1109\/ICASSP48485.2024.10447923"},{"key":"10171_CR22","doi-asserted-by":"crossref","unstructured":"Han, W., Jiang, T., Li, Y., Schuller, B., & Ruan, H. (2020). Ordinal learning for emotion recognition in customer service calls. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 6494\u20136498). IEEE.","DOI":"10.1109\/ICASSP40776.2020.9053648"},{"key":"10171_CR23","doi-asserted-by":"crossref","unstructured":"He, Y., Minematsu, N., & Saito, D. (2023). Multiple acoustic features speech emotion recognition using cross-attention transformer. In Procedings IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE.","DOI":"10.1109\/ICASSP49357.2023.10095777"},{"key":"10171_CR24","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"W-N Hsu","year":"2021","unstructured":"Hsu, W.-N., Bolte, B., Tsai, Y.-H.H., Lakhotia, K., Salakhutdinov, R., & Mohamed, A. (2021). HuBERT: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 29, 3451\u20133460.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10171_CR25","doi-asserted-by":"crossref","unstructured":"Huo, Z., Sim, K. C., Li, B., Hwang, D., Sainath, T. N., & Strohman, T. (2023). Resource-efficient transfer learning from speech foundation model using hierarchical feature fusion. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE.","DOI":"10.1109\/ICASSP49357.2023.10096799"},{"key":"10171_CR26","unstructured":"Lackovic, N., Montaci\u00e9, C., Lalande, G., & Caraty, M.-J. (2022). Prediction of user request and complaint in spoken customer-agent conversations. arXiv preprint arXiv:2208.10249."},{"key":"10171_CR27","doi-asserted-by":"crossref","unstructured":"Lackovic, N., Montaci\u00e9, C., Lequilliec, C., & Caraty, M.-J. (2023). Healthcall corpus and Transformer embeddings from healthcare customer-agent conversations. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE.","DOI":"10.1109\/ICASSP49357.2023.10096196"},{"key":"10171_CR28","doi-asserted-by":"crossref","unstructured":"Lashkarashvili, N., Wu, W., Sun, G., & Woodland, P. C. (2024). Parameter efficient finetuning for speech emotion recognition and domain adaptation. In International conference on acoustics, speech, and signal processing (ICASSP), (pp. 10986\u201310990). IEEE.","DOI":"10.1109\/ICASSP48485.2024.10446272"},{"key":"10171_CR29","unstructured":"Latif, S., Rana, R., Khalifa, S., Jurdak, R., Qadir, J., & Schuller, B. W. (2020). Deep representation learning in speech processing: Challenges, recent advances, and future trends. arXiv preprint arXiv:2001.00378."},{"key":"10171_CR30","doi-asserted-by":"crossref","unstructured":"Li, B., Dimitriadis, D., & Stolcke, A. (2019). Acoustic and lexical sentiment analysis for customer service calls. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 5876\u20135880). IEEE.","DOI":"10.1109\/ICASSP.2019.8683679"},{"key":"10171_CR31","doi-asserted-by":"crossref","unstructured":"Li, B., Hwang, D., Huo, Z., Bai, J., Prakash, G., Sainath, T. N., Sim, K.C., Zhang, Y., Han, W., Strohman, T., et al. (2023). Efficient domain adaptation for speech foundation models. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE.","DOI":"10.1109\/ICASSP49357.2023.10096330"},{"key":"10171_CR32","doi-asserted-by":"crossref","unstructured":"Makiuchi, M. R., Uto, K., & Shinoda, K. (2021). Multimodal emotion recognition with high-level speech and text features. In Proceedings of the IEEE automatic speech recognition and understanding workshop (ASRU) (pp. 350\u2013357). IEEE.","DOI":"10.1109\/ASRU51503.2021.9688036"},{"key":"10171_CR33","doi-asserted-by":"publisher","first-page":"622","DOI":"10.1109\/TASLP.2022.3140482","volume":"30","author":"Q Mao","year":"2022","unstructured":"Mao, Q., Li, J., Lin, C., Chen, C., Peng, H., Wang, L., & Philip, S. Y. (2022). Adaptive pre-training and collaborative fine-tuning: A win-win strategy to improve review analysis tasks. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 30, 622\u2013634.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10171_CR34","doi-asserted-by":"crossref","unstructured":"Mazumder, P., Singh, P., & Namboodiri, V. (2020). CPWC: Contextual point wise convolution for object recognition. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 4152\u20134156). IEEE.","DOI":"10.1109\/ICASSP40776.2020.9054205"},{"issue":"1","key":"10171_CR35","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1109\/TAFFC.2023.3272553","volume":"15","author":"M Niu","year":"2023","unstructured":"Niu, M., Tao, J., Li, Y., Qin, Y., & Li, Y. (2023). WavDepressionNet: Automatic depression level prediction via raw speech signals. IEEE Transactions on Affective Computing, 15(1), 285\u2013296.","journal-title":"IEEE Transactions on Affective Computing"},{"issue":"2","key":"10171_CR36","doi-asserted-by":"publisher","first-page":"1533","DOI":"10.1109\/TAFFC.2021.3112543","volume":"14","author":"PA P\u00e9rez-Toro","year":"2021","unstructured":"P\u00e9rez-Toro, P. A., V\u00e1squez-Correa, J. C., Bocklet, T., N\u00f6th, E., & Orozco-Arroyave, J. R. (2021). User state modeling based on the arousal-valence plane: Applications in customer satisfaction and health-care. IEEE Transactions on Affective Computing, 14(2), 1533\u20131546.","journal-title":"IEEE Transactions on Affective Computing"},{"key":"10171_CR37","doi-asserted-by":"crossref","unstructured":"Porjazovski, D., Getman, Y., Gr\u00f3sz, T., & Kurimo, M. (2023). Advancing audio emotion and intent recognition with large pre-trained models and Bayesian inference. In Proceedings of the ACM international conference on multimedia (pp. 9477\u20139481).","DOI":"10.1145\/3581783.3612848"},{"key":"10171_CR38","unstructured":"Radford, A., Kim, J. W., Xu, T., Brockman, G., McLeavey, C., & Sutskever, I. (2023). Robust speech recognition via large-scale weak supervision. In Proceedings of the international conference on machine learning (pp. 28492\u201328518). PMLR."},{"key":"10171_CR39","doi-asserted-by":"crossref","unstructured":"Ristea, N.-C., & Ionescu, R. T. (2023). Cascaded cross-modal transformer for request and complaint detection. In Proceedings ACM international conference on multimedia (pp. 9467\u20139471).","DOI":"10.1145\/3581783.3612846"},{"key":"10171_CR40","doi-asserted-by":"crossref","unstructured":"Ristea, N. C., Ionescu, R. T., & Khan, F. S. (2022). SepTr: Separable Transformer for audio spectrogram processing. In Proceedings of the annual conference of the international speech communication association (INTERSPEECH) (Vol. 2022, pp. 4103\u20134107).","DOI":"10.21437\/Interspeech.2022-249"},{"key":"10171_CR41","doi-asserted-by":"crossref","unstructured":"Schuller, B. W., Batliner, A., Amiriparian, S., Barnhill, A., Gerczuk, M., Triantafyllopoulos, A., Baird, A. E., Tzirakis, P., Gagne, C., Cowen, A. S., et al. (2023). The ACM multimedia 2023 computational paralinguistics challenge: Emotion share & requests. In Proceedings of the ACM international conference on multimedia (pp 9635\u20139639).","DOI":"10.1145\/3581783.3612835"},{"key":"10171_CR42","doi-asserted-by":"crossref","unstructured":"Schuller, B., Batliner, A., Amiriparian, S., Bergler, C., Gerczuk, M., Holz, N., Larrouy-Maestri, P., Bayerl, S., Riedhammer, K., Mallol-Ragolta, A., et al. (2022). The ACM multimedia 2022 computational paralinguistics challenge: Vocalisations, stuttering, activity, & mosquitoes. In Proceedings of the ACM international conference on multimedia (pp. 7120\u20137124).","DOI":"10.1145\/3503161.3551591"},{"key":"10171_CR43","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1016\/j.csl.2018.02.004","volume":"53","author":"B Schuller","year":"2019","unstructured":"Schuller, B., Weninger, F., Zhang, Y., Ringeval, F., Batliner, A., Steidl, S., Eyben, F., Marchi, E., Vinciarelli, A., Scherer, K., et al. (2019). Affective and behavioural computing: Lessons learnt from the first computational paralinguistics challenge. Computer Speech & Language, 53, 156\u2013180.","journal-title":"Computer Speech & Language"},{"key":"10171_CR44","doi-asserted-by":"crossref","unstructured":"Sun, Y., Xu, K., Liu, C., Dou, Y., & Qian, K. (2023). Automatic audio augmentation for requests sub-challenge. In Proceedings of the ACM international conference on multimedia (pp. 9482\u20139486).","DOI":"10.1145\/3581783.3612849"},{"key":"10171_CR45","doi-asserted-by":"publisher","first-page":"105642","DOI":"10.1016\/j.bspc.2023.105642","volume":"88","author":"A Triantafyllopoulos","year":"2024","unstructured":"Triantafyllopoulos, A., Semertzidou, A., Song, M., Pokorny, F. B., & Schuller, B. W. (2024). Introducing the COVID-19 YouTube (COVYT) speech dataset featuring the same speakers with and without infection. Biomedical Signal Processing and Control, 88, 105642.","journal-title":"Biomedical Signal Processing and Control"},{"key":"10171_CR46","doi-asserted-by":"crossref","unstructured":"Vetr\u00e1b, M., & Gosztolya, G. (2023). Aggregation strategies of wav2vec 2.0 embeddings for computational paralinguistic tasks. In Proceedings international conference on speech and computer (pp. 79\u201393). Springer.","DOI":"10.1007\/978-3-031-48309-7_7"},{"key":"10171_CR47","doi-asserted-by":"crossref","unstructured":"Viksit, S. R., & Abrol, V. (2023). Multi-layer acoustic & linguistic feature fusion for ComParE-23 emotion and requests challenge. In Proceedings of the ACM international conference on multimedia (pp. 9492\u20139495).","DOI":"10.1145\/3581783.3612851"},{"key":"10171_CR48","doi-asserted-by":"crossref","unstructured":"Wagner, J., Triantafyllopoulos, A., Wierstorf, H., Schmitt, M., Burkhardt, F., Eyben, F., & Schuller, B. W. (2023). Dawn of the transformer era in speech emotion recognition: Closing the valence gap. IEEE Transactions on Pattern Analysis and Machine Intelligence.","DOI":"10.1109\/TPAMI.2023.3263585"},{"key":"10171_CR49","unstructured":"Wang, C., Yi, J., Zhang, X., Tao, J., Xu, L., & Fu, R. (2023). Low-rank adaptation method for wav2vec2-based fake audio detection. arXiv preprint arXiv:2306.05617."},{"key":"10171_CR50","doi-asserted-by":"crossref","unstructured":"Wright, S. P. (1992). Adjusted p-values for simultaneous inference. Biometrics, 48, 1005\u20131013.","DOI":"10.2307\/2532694"},{"key":"10171_CR51","doi-asserted-by":"crossref","unstructured":"Wu, W., Zhang, C., & Woodland, P. C. (2023). Self-supervised representations in speech-based depression detection. In Proceedings IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE.","DOI":"10.1109\/ICASSP49357.2023.10094910"},{"key":"10171_CR52","doi-asserted-by":"crossref","unstructured":"Xu, X., Deng, J., Zhang, Z., Wu, C., & Schuller, B. (2021). Identifying surgical-mask speech using deep neural networks on low-level aggregation. In Proceedings of the annual ACM symposium on applied computing (pp. 580\u2013585).","DOI":"10.1145\/3412841.3441938"},{"key":"10171_CR53","doi-asserted-by":"crossref","unstructured":"Xu, X., Deng, J., Zhang, Z., Yang, Z., & Schuller, B. W. (2023). Zero-shot speech emotion recognition using generative learning with reconstructed prototypes. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE.","DOI":"10.1109\/ICASSP49357.2023.10094888"},{"issue":"3","key":"10171_CR54","doi-asserted-by":"publisher","first-page":"795","DOI":"10.1109\/TMM.2018.2865834","volume":"21","author":"X Xu","year":"2019","unstructured":"Xu, X., Deng, J., Coutinho, E., Wu, C., Zhao, L., & Schuller, B. W. (2019). Connecting subspace learning and extreme learning machine in speech emotion recognition. IEEE Transactions on Multimedia, 21(3), 795\u2013808.","journal-title":"IEEE Transactions on Multimedia"},{"key":"10171_CR55","doi-asserted-by":"publisher","first-page":"2752","DOI":"10.1109\/TMM.2021.3087098","volume":"24","author":"X Xu","year":"2022","unstructured":"Xu, X., Deng, J., Cummins, N., Zhang, Z., Zhao, L., & Schuller, B. W. (2022). Exploring zero-shot emotion recognition in speech using semantic-embedding prototypes. IEEE Transactions on Multimedia, 24, 2752\u20132765.","journal-title":"IEEE Transactions on Multimedia"},{"key":"10171_CR56","doi-asserted-by":"publisher","first-page":"2884","DOI":"10.1109\/TASLP.2024.3389631","volume":"32","author":"S-W Yang","year":"2024","unstructured":"Yang, S.-W., Chang, H.-J., Huang, Z., Liu, A. T., Lai, C.-I., Wu, H., Shi, J., Chang, X., Tsai, H.-S., Huang, W.-C., et al. (2024). A large-scale evaluation of speech foundation models. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 32, 2884\u20132899.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10171_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, T., Chang, D., Ma, Z., & Guo, J. (2021). Progressive co-attention network for fine-grained visual classification. In Proceedings of the international conference on visual communications and image processing (VCIP). IEEE.","DOI":"10.1109\/VCIP53242.2021.9675376"},{"issue":"6","key":"10171_CR58","doi-asserted-by":"publisher","first-page":"1227","DOI":"10.1109\/JSTSP.2022.3184480","volume":"16","author":"J Zhao","year":"2022","unstructured":"Zhao, J., & Zhang, W.-Q. (2022). Improving automatic speech recognition performance for low-resource languages with self-supervised models. IEEE Journal of Selected Topics in Signal Processing, 16(6), 1227\u20131241.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"10171_CR59","doi-asserted-by":"crossref","unstructured":"Zhou, C., Li, Q., Li, C., Yu, J., Liu, Y., Wang, G., Zhang, K., Ji, C., Yan, Q., He, L., et al. (2023). A comprehensive survey on pretrained foundation models: A history from BERT to ChatGPT. arXiv preprint arXiv:2302.09419.","DOI":"10.1007\/s13042-024-02443-6"},{"key":"10171_CR60","doi-asserted-by":"crossref","unstructured":"Zhu, Z., & Sato, Y. (2023). Deep investigation of intermediate representations in self-supervised learning models for speech emotion recognition. In Proceedings of the IEEE international conference on acoustics, speech, and signal processing workshops (ICASSPW). IEEE.","DOI":"10.1109\/ICASSPW59220.2023.10193018"},{"key":"10171_CR61","doi-asserted-by":"publisher","first-page":"1927","DOI":"10.1109\/TASLP.2023.3275033","volume":"31","author":"Q-S Zhu","year":"2023","unstructured":"Zhu, Q.-S., Zhang, J., Zhang, Z.-Q., & Dai, L.-R. (2023). A joint speech enhancement and self-supervised representation learning framework for noise-robust speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 31, 1927\u20131939.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10171_CR62","doi-asserted-by":"crossref","unstructured":"Zou, H., Si, Y., Chen, C., Rajan, D., & Chng, E. S. (2022). Speech emotion recognition with co-attention based multi-level acoustic information. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp 7367\u20137371). IEEE.","DOI":"10.1109\/ICASSP43922.2022.9747095"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-025-10171-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-025-10171-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-025-10171-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,10]],"date-time":"2025-04-10T08:53:01Z","timestamp":1744275181000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-025-10171-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,5]]},"references-count":62,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,3]]}},"alternative-id":["10171"],"URL":"https:\/\/doi.org\/10.1007\/s10772-025-10171-7","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2,5]]},"assertion":[{"value":"22 August 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 January 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 February 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}