{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T21:51:44Z","timestamp":1775253104696,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,10,15]],"date-time":"2018-10-15T00:00:00Z","timestamp":1539561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100011002","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61772535"],"award-info":[{"award-number":["61772535"]}],"id":[{"id":"10.13039\/501100011002","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development","award":["2016YFB1001202"],"award-info":[{"award-number":["2016YFB1001202"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,10,15]]},"DOI":"10.1145\/3266302.3266313","type":"proceedings-article","created":{"date-parts":[[2018,10,18]],"date-time":"2018-10-18T10:19:29Z","timestamp":1539857969000},"page":"65-72","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":33,"title":["Multi-modal Multi-cultural Dimensional Continues Emotion Recognition in Dyadic Interactions"],"prefix":"10.1145","author":[{"given":"Jinming","family":"Zhao","sequence":"first","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruichen","family":"Li","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shizhe","family":"Chen","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qin","family":"Jin","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Yusuf Aytar Carl Vondrick and Antonio Torralba. 2016. SoundNet: Learning Sound Representations from Unlabeled Video. (2016). Yusuf Aytar Carl Vondrick and Antonio Torralba. 2016. SoundNet: Learning Sound Representations from Unlabeled Video. (2016).","DOI":"10.1109\/CVPR.2016.18"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2993148.2993165"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2988257.2988264"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2808196.2811634"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2663204.2666277"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2808196.2811638"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967286"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3133944.3133949"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1080\/08839510290030390"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3266302.3266316"},{"key":"e_1_3_2_1_12_1","first-page":"32","article-title":"Emotion recognition in human-computer interaction","volume":"18","author":"Fragopanagos N","year":"2002","journal-title":"IEEE Signal Processing Magazine"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Shawn Hershey Sourish Chaudhuri Daniel P. W. Ellis Jort F. Gemmeke Aren Jansen R. Channing Moore Manoj Plakal Devin Platt Rif A. Saurous and Bryan Seybold. 2016. CNN architectures for large-scale audio classification. (2016) 131--135. Shawn Hershey Sourish Chaudhuri Daniel P. W. Ellis Jort F. Gemmeke Aren Jansen R. Channing Moore Manoj Plakal Devin Platt Rif A. Saurous and Bryan Seybold. 2016. CNN architectures for large-scale audio classification. (2016) 131--135.","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"e_1_3_2_1_14_1","volume-title":"Densely Connected Convolutional Networks. In IEEE Conference on Computer Vision and Pattern Recognition. 2261--2269","author":"Huang Gao"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2808196.2811640"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Chi Chun Lee Carlos Busso Sungbok Lee and et.al. 2009. Modeling mutual influence of interlocutor emotion states in dyadic spoken interactions. In INTERSPEECH. 1983--1986. Chi Chun Lee Carlos Busso Sungbok Lee and et.al. 2009. Modeling mutual influence of interlocutor emotion states in dyadic spoken interactions. In INTERSPEECH. 1983--1986.","DOI":"10.21437\/Interspeech.2009-480"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"L. I. Lin. 1989. A concordance correlation coefficient to evaluate reproducibility. Biometrics (1989). L. I. Lin. 1989. A concordance correlation coefficient to evaluate reproducibility. Biometrics (1989).","DOI":"10.2307\/2532051"},{"key":"e_1_3_2_1_18_1","unstructured":"Mingsheng Long Yue Cao Jianmin Wang and Michael I. Jordan. 2015. Learning transferable features with deep adaptation networks. (2015) 97--105. Mingsheng Long Yue Cao Jianmin Wang and Michael I. Jordan. 2015. Learning transferable features with deep adaptation networks. (2015) 97--105."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/T-AFFC.2013.11"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2631912"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Angeliki Metallinou Athanasios Katsamanis and Shrikanth Narayanan. 2012. A hierarchical framework for modeling multimodality and emotional evolution in affective dialogs. In ICASSP. 2401--2404. Angeliki Metallinou Athanasios Katsamanis and Shrikanth Narayanan. 2012. A hierarchical framework for modeling multimodality and emotional evolution in affective dialogs. In ICASSP. 2401--2404.","DOI":"10.1109\/ICASSP.2012.6288399"},{"key":"e_1_3_2_1_22_1","first-page":"3111","article-title":"Distributed Representations of Words and Phrases and their Compositionality","volume":"26","author":"Mikolov Tomas","year":"2013","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Ha?im Sak Andrew Senior and Fran?oise Beaufays. 2014. Long Short-Term Memory Based Recurrent Neural Network Architectures for Large Vocabulary Speech Recognition. Computer Science (2014) 338--342. Ha?im Sak Andrew Senior and Fran?oise Beaufays. 2014. Long Short-Term Memory Based Recurrent Neural Network Architectures for Large Vocabulary Speech Recognition. Computer Science (2014) 338--342.","DOI":"10.21437\/Interspeech.2014-80"},{"key":"e_1_3_2_1_24_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. Computer Science (2014). Karen Simonyan and Andrew Zisserman. 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. Computer Science (2014)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Eric Tzeng Judy Hoffman Kate Saenko and Trevor Darrell. 2017. Adversarial Discriminative Domain Adaptation. (2017). Eric Tzeng Judy Hoffman Kate Saenko and Trevor Darrell. 2017. Adversarial Discriminative Domain Adaptation. (2017).","DOI":"10.1109\/CVPR.2017.316"},{"key":"e_1_3_2_1_26_1","unstructured":"Eric Tzeng Judy Hoffman Ning Zhang Kate Saenko and Trevor Darrell. 2014. Deep Domain Confusion: Maximizing for Domain Invariance. Computer Science (2014). Eric Tzeng Judy Hoffman Ning Zhang Kate Saenko and Trevor Darrell. 2014. Deep Domain Confusion: Maximizing for Domain Invariance. Computer Science (2014)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1017\/ATSIP.2014.11"},{"key":"e_1_3_2_1_28_1","volume-title":"INTERSPEECH 2010, Conference of the International Speech Communication Association","author":"W\u00f6llmer Martin"}],"event":{"name":"MM '18: ACM Multimedia Conference","location":"Seoul Republic of Korea","acronym":"MM '18","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2018 on Audio\/Visual Emotion Challenge and Workshop"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3266302.3266313","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3266302.3266313","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T20:37:11Z","timestamp":1775248631000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3266302.3266313"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,10,15]]},"references-count":28,"alternative-id":["10.1145\/3266302.3266313","10.1145\/3266302"],"URL":"https:\/\/doi.org\/10.1145\/3266302.3266313","relation":{},"subject":[],"published":{"date-parts":[[2018,10,15]]},"assertion":[{"value":"2018-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}