{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:24:04Z","timestamp":1784737444543,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Special Fund for Military Healthcare Committee","award":["22BJZ28"],"award-info":[{"award-number":["22BJZ28"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62271083"],"award-info":[{"award-number":["62271083"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"BUPT Excellent Ph.D. Students Foundation"},{"name":"Fundamental Research Funds for the Central Universities","award":["2023RC13"],"award-info":[{"award-number":["2023RC13"]}]},{"name":"open research fund of The State Key Laboratory of Multimodal Artificial Intelligence Systems","award":["202200042,202200012"],"award-info":[{"award-number":["202200042,202200012"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612862","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"9546-9550","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Mining High-quality Samples from Raw Data and Majority Voting Method for Multimodal Emotion Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5016-0094","authenticated-orcid":false,"given":"Qifei","family":"Li","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5881-3723","authenticated-orcid":false,"given":"Yingming","family":"Gao","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6284-5039","authenticated-orcid":false,"given":"Ya","family":"Li","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-434"},{"key":"e_1_3_2_2_2_1","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in Neural Information Processing Systems 33 (2020), 12449--12460.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_3_1","volume-title":"Openface: an open source facial behavior analysis toolkit. In 2016 IEEE winter conference on applications of computer vision (WACV)","author":"Tadas","unstructured":"Tadas Baltru?aitis, Peter Robinson, and Louis-Philippe Morency. 2016. Openface: an open source facial behavior analysis toolkit. In 2016 IEEE winter conference on applications of computer vision (WACV). IEEE, Lake Placid, USA, 1--10."},{"key":"e_1_3_2_2_4_1","first-page":"120","article-title":"The openCV library","volume":"25","author":"Bradski Gary","year":"2000","unstructured":"Gary Bradski. 2000. The openCV library. Dr. Dobb's Journal: Software Tools for the Professional Programmer 25, 3 (2000), 120--123.","journal-title":"Dr. Dobb's Journal: Software Tools for the Professional Programmer"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","first-page":"110","DOI":"10.1093\/acprof:oso\/9780195387643.003.0008","article-title":"Toward effective automatic recognition systems of emotion in speech","volume":"7","author":"Busso Carlos","year":"2013","unstructured":"Carlos Busso, Murtaza Bulut, Shrikanth Narayanan, J Gratch, and S Marsella. 2013. Toward effective automatic recognition systems of emotion in speech. Social Emotions in Nature and Artifact: Emotions in Human and Human-Computer Interaction 7, 17 (2013), 110--127.","journal-title":"Social Emotions in Nature and Artifact: Emotions in Human and Human-Computer Interaction"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095036"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746598"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/79.911197"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3124365"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Florian Eyben Klaus R Scherer Bj\u00f6rn W Schuller Johan Sundberg Elisabeth Andr\u00e9 Carlos Busso Laurence Y Devillers Julien Epps Petri Laukka Shrikanth S Narayanan et al. 2015. The Geneva minimalistic acoustic parameter set (GeMAPS) for voice research and affective computing. IEEE transactions on affective computing 7 2 (2015) 190--202.","DOI":"10.1109\/TAFFC.2015.2457417"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1874246"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Zhifu Gao Zerui Li Jiaming Wang Haoneng Luo Xian Shi Mengzhe Chen Yabin Li Lingyun Zuo Zhihao Du Zhangyu Xiao and Shiliang Zhang. 2023. FunASR: A Fundamental End-to-End Speech Recognition Toolkit. arXiv:arXiv:2305.11013","DOI":"10.21437\/Interspeech.2023-1428"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2015.12"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","unstructured":"DN Krishna and Ankita Patil. 2020. Multimodal Emotion Recognition Using Cross-Modal Attention and 1D Convolutional Neural Networks.. In Interspeech. ISCA Shanghai China 4243--4247. https:\/\/doi.org\/10.21437\/Interspeech.2020-1190","DOI":"10.21437\/Interspeech.2020-1190"},{"key":"e_1_3_2_2_20_1","volume-title":"MER 2023: Multi-label Learning, Modality Robustness, and Semi-Supervised Learning. arXiv:arXiv:2304","author":"Lian Zheng","year":"2023","unstructured":"Zheng Lian, Haiyang Sun, Licai Sun, Jinming Zhao, Ye Liu, Bin Liu, Jiangyan Yi, Meng Wang, Erik Cambria, Guoying Zhao, Bj\u00f6rn W. Schuller, and Jianhua Tao. 2023. MER 2023: Multi-label Learning, Modality Robustness, and Semi-Supervised Learning. arXiv:arXiv:2304.08981"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2009-103"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3026823"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Shamane Siriwardhana Andrew Reis Rivindu Weerasekera and Suranga Nanayakkara. 2020. Jointly Fine-Tuning \"BERT-like\" Self Supervised Models to Improve Multimodal Speech Emotion Recognition. arXiv:arXiv:2008.06682","DOI":"10.21437\/Interspeech.2020-1212"},{"key":"e_1_3_2_2_24_1","volume-title":"MUSAN: A Music, Speech, and Noise Corpus. arXiv:arXiv:1510.08484","author":"Snyder David","year":"2015","unstructured":"David Snyder, Guoguo Chen, and Daniel Povey. 2015. MUSAN: A Music, Speech, and Noise Corpus. arXiv:arXiv:1510.08484"},{"key":"e_1_3_2_2_25_1","first-page":"16857","article-title":"Mpnet: Masked and permuted pre-training for language understanding","volume":"33","author":"Song Kaitao","year":"2020","unstructured":"Kaitao Song, Xu Tan, Tao Qin, Jianfeng Lu, and Tie-Yan Liu. 2020. Mpnet: Masked and permuted pre-training for language understanding. Advances in Neural Information Processing Systems 33 (2020), 16857--16867.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414654"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2022.03.009"},{"key":"e_1_3_2_2_28_1","unstructured":"Jiaxing Zhang Ruyi Gan Junjie Wang Yuxiang Zhang Lin Zhang Ping Yang Xinyu Gao Ziwei Wu Xiaoqun Dong Junqing He Jianheng Zhuo Qi Yang Yongfeng Huang Xiayu Li Yanghan Wu Junyu Lu Xinyu Zhu Weifeng Chen Ting Han Kunhao Pan Rui Wang Hao Wang Xiaojun Wu Zhongshen Zeng and Chongpei Chen. 2022. Fengshenbang 1.0: Being the Foundation of Chinese Cognitive Intelligence. arXiv:arXiv:2209.02970"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.391"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612862","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612862","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:15:09Z","timestamp":1755821709000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612862"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":29,"alternative-id":["10.1145\/3581783.3612862","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612862","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}