{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T07:40:16Z","timestamp":1735026016153,"version":"3.32.0"},"reference-count":31,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,11,7]],"date-time":"2024-11-07T00:00:00Z","timestamp":1730937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,11,7]],"date-time":"2024-11-07T00:00:00Z","timestamp":1730937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62076144"],"award-info":[{"award-number":["62076144"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,11,7]]},"DOI":"10.1109\/iscslp63861.2024.10800636","type":"proceedings-article","created":{"date-parts":[[2024,12,23]],"date-time":"2024-12-23T19:11:17Z","timestamp":1734981077000},"page":"456-460","source":"Crossref","is-referenced-by-count":0,"title":["ERVQ: Leverage Residual Vector Quantization for Speech Emotion Recognition"],"prefix":"10.1109","author":[{"given":"Jingran","family":"Xie","sequence":"first","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University,Shenzhen"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Xiang","sequence":"additional","affiliation":[{"name":"Pengcheng Laboratory,Shenzhen"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hui","family":"Wang","sequence":"additional","affiliation":[{"name":"Pengcheng Laboratory,Shenzhen"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xixin","family":"Wu","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong,Hong Kong SAR"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiyong","family":"Wu","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University,Shenzhen"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Helen","family":"Meng","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong,Hong Kong SAR"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/3129340"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/IAdCC.2013.6514336"},{"key":"ref3","first-page":"1","article-title":"Tempo-ral modeling matters: A novel temporal emotional modeling approach for speech emotion recognition","volume-title":"ICASSP 2023\u20132023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), Rhodes Island, Greece, June 4\u201310, 2023","author":"Ye","year":"2023"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096966"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1170"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2024.3369726"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.2478\/v10122-012-0013-1"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1515\/9783112414989"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/MASSP.1984.1162229"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/MP.2006.1664069"},{"key":"ref11","article-title":"Soundstorm: Efficient parallel audio generation","author":"Borsos","year":"2023","journal-title":"arXiv preprint"},{"key":"ref12","article-title":"High fidelity neural audio compression","author":"D\u00e9fossez","year":"2022","journal-title":"arXiv preprint"},{"key":"ref13","article-title":"Hifi-codec: Group-residual vector quantization for high fidelity audio codec","author":"Yang","year":"2023","journal-title":"arXiv preprint"},{"key":"ref14","article-title":"Fast and accurate deep network learning by exponential linear units (elus)","author":"Clevert","year":"2015","journal-title":"arXiv preprint"},{"key":"ref15","article-title":"Weight normalization: A sim-ple reparameterization to accelerate training of deep neural networks","volume":"29","author":"Salimans","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref16","article-title":"Estimating or propa-gating gradients through stochastic neurons for conditional computation","author":"Bengio","year":"2013","journal-title":"arXiv preprint"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1139"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3207050"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref23","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017","journal-title":"arXiv preprint"},{"key":"ref24","first-page":"1298","article-title":"Data2vec: A general framework for self-supervised learning in speech, vision and language","volume-title":"International Conference on Ma-chine Learning","author":"Baevski"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1775"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096808"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.391"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3204949.3208121"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2005-446"},{"key":"ref31","first-page":"3501","article-title":"Emovo corpus: an italian emotional speech database","volume-title":"Proceedings of the ninth international conference on language resources and evaluation (LREC\u201914)","author":"Costantini","year":"2014"}],"event":{"name":"2024 IEEE 14th International Symposium on Chinese Spoken Language Processing (ISCSLP)","start":{"date-parts":[[2024,11,7]]},"location":"Beijing, China","end":{"date-parts":[[2024,11,10]]}},"container-title":["2024 IEEE 14th International Symposium on Chinese Spoken Language Processing (ISCSLP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10799944\/10799969\/10800636.pdf?arnumber=10800636","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,24]],"date-time":"2024-12-24T06:29:38Z","timestamp":1735021778000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10800636\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,7]]},"references-count":31,"URL":"https:\/\/doi.org\/10.1109\/iscslp63861.2024.10800636","relation":{},"subject":[],"published":{"date-parts":[[2024,11,7]]}}}