{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,26]],"date-time":"2026-06-26T20:37:26Z","timestamp":1782506246681,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100002341","name":"Academy of Finland","doi-asserted-by":"publisher","award":["37073,345790"],"award-info":[{"award-number":["37073,345790"]}],"id":[{"id":"10.13039\/501100002341","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Foundation for Aalto University Science and Technology"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612848","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"9477-9481","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Advancing Audio Emotion and Intent Recognition with Large Pre-Trained Models and Bayesian Inference"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7219-9042","authenticated-orcid":false,"given":"Dejan","family":"Porjazovski","sequence":"first","affiliation":[{"name":"Aalto University, Espoo, Finland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4680-8294","authenticated-orcid":false,"given":"Yaroslav","family":"Getman","sequence":"additional","affiliation":[{"name":"Aalto University, Espoo, Finland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7918-9579","authenticated-orcid":false,"given":"Tam\u00e1s","family":"Gr\u00f3sz","sequence":"additional","affiliation":[{"name":"Aalto University, Espoo, Finland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5278-7974","authenticated-orcid":false,"given":"Mikko","family":"Kurimo","sequence":"additional","affiliation":[{"name":"Aalto University, Espoo, Finland"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Workshop on Detection and Classification of Acoustic Scenes and Events.","author":"Amiriparian Shahin","year":"2017","unstructured":"Shahin Amiriparian, Michael J. Freitag, Nicholas Cummins, and Bj\u00f6rn Schuller. 2017a. Sequence to Sequence Autoencoders for Unsupervised Representation Learning from Audio. In Workshop on Detection and Classification of Acoustic Scenes and Events."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-434"},{"key":"e_1_3_2_2_3_1","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems, Vol. 33 (2020), 12449--12460."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021--329"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1037\/amp0000399"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/SPED.2019.8906584"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.5555\/3122009.3242030"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Yaroslav Getman Ragheb Al-Ghezi Katja Voskoboinik Tam\u00e1s Gr\u00f3sz Mikko Kurimo Giampiero Salvi Torbj\u00f8rn Svendsen and Sofia Str\u00f6mbergsson. 2022. Wav2vec2-based speech rating system for children with speech sound disorder. In Interspeech.","DOI":"10.21437\/Interspeech.2022-10103"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-42553-1_3"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3551572"},{"key":"e_1_3_2_2_12_1","volume-title":"Stochastic variational inference. Journal of Machine Learning Research","author":"Hoffman Matthew D","year":"2013","unstructured":"Matthew D Hoffman, David M Blei, Chong Wang, and John Paisley. 2013. Stochastic variational inference. Journal of Machine Learning Research (2013)."},{"key":"e_1_3_2_2_13_1","volume-title":"Prediction of User Request and Complaint in Spoken Customer-Agent Conversations. arXiv preprint arXiv:2208.10249","author":"Lackovic Nikola","year":"2022","unstructured":"Nikola Lackovic, Claude Montaci\u00e9, Gauthier Lalande, and Marie-Jos\u00e9 Caraty. 2022. Prediction of User Request and Complaint in Spoken Customer-Agent Conversations. arXiv preprint arXiv:2208.10249 (2022)."},{"key":"e_1_3_2_2_14_1","volume-title":"Graddiv: Adversarial robustness of randomized neural networks via gradient diversity regularization","author":"Lee Sungyoon","year":"2022","unstructured":"Sungyoon Lee, Hoki Kim, and Jaewook Lee. 2022. Graddiv: Adversarial robustness of randomized neural networks via gradient diversity regularization. IEEE Transactions on Pattern Analysis and Machine Intelligence (2022)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.645"},{"key":"e_1_3_2_2_16_1","volume-title":"Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, et al. 2019. Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-703"},{"key":"e_1_3_2_2_18_1","volume-title":"Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever.","author":"Radford Alec","year":"2022","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2022. Robust speech recognition via large-scale weak supervision. arXiv preprint arXiv:2212.04356 (2022)."},{"key":"e_1_3_2_2_19_1","unstructured":"Mirco Ravanelli Titouan Parcollet Peter Plantinga Aku Rouhe Samuele Cornell Loren Lugosch Cem Subakan Nauman Dawalatabad Abdelwahab Heba Jianyuan Zhong et al. 2021. SpeechBrain: A general-purpose speech toolkit. arXiv preprint arXiv:2106.04624 (2021)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612835"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461785"},{"key":"e_1_3_2_2_22_1","volume-title":"Introducing ECAPA-TDNN and Wav2Vec2. 0 embeddings to stuttering detection. arXiv preprint arXiv:2204.01564","author":"Sheikh Shakeel Ahmad","year":"2022","unstructured":"Shakeel Ahmad Sheikh, Md Sahidullah, Fabrice Hirsch, and Slim Ouni. 2022. Introducing ECAPA-TDNN and Wav2Vec2. 0 embeddings to stuttering detection. arXiv preprint arXiv:2204.01564 (2022)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-118"},{"key":"e_1_3_2_2_24_1","volume-title":"FUNCTIONAL VARIATIONAL BAYESIAN NEURAL NETWORKS. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkxacs0qY7","author":"Sun Shengyang","year":"2019","unstructured":"Shengyang Sun, Guodong Zhang, Jiaxin Shi, and Roger Grosse. 2019. FUNCTIONAL VARIATIONAL BAYESIAN NEURAL NETWORKS. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkxacs0qY7"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746952"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612848","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612848","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:14:21Z","timestamp":1755821661000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612848"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":25,"alternative-id":["10.1145\/3581783.3612848","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612848","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}