{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,24]],"date-time":"2025-10-24T08:29:33Z","timestamp":1761294573544,"version":"3.28.0"},"reference-count":47,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T00:00:00Z","timestamp":1673222400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T00:00:00Z","timestamp":1673222400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,1,9]]},"DOI":"10.1109\/slt54892.2023.10022557","type":"proceedings-article","created":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T18:54:03Z","timestamp":1674845643000},"page":"806-813","source":"Crossref","is-referenced-by-count":1,"title":["Multilingual Speech Emotion Recognition with Multi-Gating Mechanism and Neural Architecture Search"],"prefix":"10.1109","author":[{"given":"Zihan","family":"Wang","sequence":"first","affiliation":[{"name":"Columbia University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Meng","sequence":"additional","affiliation":[{"name":"Columbia University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"HaiFeng","family":"Lan","sequence":"additional","affiliation":[{"name":"Columbia University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"XinRui","family":"Zhang","sequence":"additional","affiliation":[{"name":"Columbia University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"KeHao","family":"Guo","sequence":"additional","affiliation":[{"name":"Columbia University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Akshat","family":"Gupta","sequence":"additional","affiliation":[{"name":"JP Morgan AI Research,New York,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Techniques and ap-plications of emotion recognition in speech","author":"Duner","year":"07 2016","journal-title":"2016 39th International Convention on Information and Commu-nication Technology, Electronics and Microelectronics (MIPRO)"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1109\/34.895976"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1109\/ICASSP43922.2022.9746289"},{"doi-asserted-by":"publisher","key":"ref4","DOI":"10.1109\/ICASSP.2015.7178964"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.1109\/icassp40776.2020.9054362"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1109\/ICASSP39728.2021.9415112"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.25080\/majora-7b98e3ed-003"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.21437\/eurospeech.2003-80"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.21437\/smm.2018-5"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/access.2020.2990405"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1109\/ACCESS.2020.2967791"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/APSIPAASC47483.2019.9023098"},{"year":"10 2018","author":"Devlin","journal-title":"Bert: Pre-training of deep bidi-rectional transformers for language understanding","key":"ref13"},{"year":"01 2019","author":"Lample","journal-title":"Cross-lingual language model pretraining","key":"ref14"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.18653\/v1\/p19-1139"},{"doi-asserted-by":"publisher","key":"ref16","DOI":"10.1109\/ICASSP.2018.8462162"},{"volume-title":"New York Academy of Science Machine Learning Symposium","author":"Goel","article-title":"Cross-lingual cross-corpus speech emotion recognition","key":"ref17"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/icassp.2018.8462162"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1007\/s10579-008-9076-6"},{"doi-asserted-by":"publisher","key":"ref20","DOI":"10.21437\/interspeech.2005-446"},{"volume-title":"ACM Multimedia Systems Conference (MMSys 2018) (MMSys18)","author":"Olivier","article-title":"A canadian french emotional speech dataset (1. 1) [data set]","key":"ref21"},{"key":"ref22","article-title":"Wav2vec: Unsupervised pre-training for speech recognition","volume-title":"Interspeech","author":"Schneider","year":"2019"},{"doi-asserted-by":"publisher","key":"ref23","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"ref24","first-page":"21271","article-title":"Boot-strap your own latent-a new approach to self-supervised learning","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Grill","year":"2020"},{"year":"06 2017","author":"Vaswani","journal-title":"Attention is all you need","key":"ref25"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1007\/978-1-4615-5529-2_5"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1109\/iccv.2015.169"},{"doi-asserted-by":"publisher","key":"ref28","DOI":"10.1109\/tpami.2016.2577031"},{"doi-asserted-by":"publisher","key":"ref29","DOI":"10.1145\/3219819.3220007"},{"doi-asserted-by":"publisher","key":"ref30","DOI":"10.1007\/s10994-009-5152-4"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.1145\/1273496.1273507"},{"year":"05 2020","author":"Kyriakides","journal-title":"An introduction to neural architecture search for convolutional networks","key":"ref32"},{"year":"06 2018","author":"Liu","journal-title":"Darts: Differentiable architecture search","key":"ref33"},{"year":"03 2017","author":"Real","journal-title":"Large-scale evolution of image classifiers","key":"ref34"},{"key":"ref35","article-title":"Neural architecture search with reinforcement learning","author":"Zoph","year":"2016","journal-title":"11"},{"doi-asserted-by":"publisher","key":"ref36","DOI":"10.1609\/aaai.v33i01.3301216"},{"year":"12 2017","author":"Louizos","journal-title":"Learning sparse neural networks through l0regularization","key":"ref37"},{"year":"08 2019","author":"Misra","journal-title":"Mish: A self regularized non-monotonic neural activation function","key":"ref38"},{"year":"2016","author":"Hendrycks","journal-title":"Gaussian Error Linear Units (GELUs)","key":"ref39"},{"doi-asserted-by":"publisher","key":"ref40","DOI":"10.21437\/interspeech.2020-1705"},{"doi-asserted-by":"publisher","key":"ref41","DOI":"10.21437\/interspeech.2021-1852"},{"doi-asserted-by":"publisher","key":"ref42","DOI":"10.1007\/978-3-031-05936-0_31"},{"doi-asserted-by":"publisher","key":"ref43","DOI":"10.1109\/ACCESS.2019.2938007"},{"doi-asserted-by":"publisher","key":"ref44","DOI":"10.1016\/j.bspc.2018.08.035"},{"doi-asserted-by":"publisher","key":"ref45","DOI":"10.1155\/2021\/9916915"},{"doi-asserted-by":"publisher","key":"ref46","DOI":"10.21437\/interspeech.2021-2217"},{"doi-asserted-by":"publisher","key":"ref47","DOI":"10.1109\/ICASSP43922.2022.9747348"}],"event":{"name":"2022 IEEE Spoken Language Technology Workshop (SLT)","start":{"date-parts":[[2023,1,9]]},"location":"Doha, Qatar","end":{"date-parts":[[2023,1,12]]}},"container-title":["2022 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10022052\/10022330\/10022557.pdf?arnumber=10022557","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,13]],"date-time":"2024-02-13T08:33:46Z","timestamp":1707813226000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10022557\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,1,9]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/slt54892.2023.10022557","relation":{},"subject":[],"published":{"date-parts":[[2023,1,9]]}}}