{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T11:48:16Z","timestamp":1763466496502,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","license":[{"start":{"date-parts":[[2014,11,7]],"date-time":"2014-11-07T00:00:00Z","timestamp":1415318400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["13&ZD189","61233009","61332017","61305003","61273288","61203258","61375027"],"award-info":[{"award-number":["13&ZD189","61233009","61332017","61305003","61273288","61203258","61375027"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2014,11,7]]},"DOI":"10.1145\/2661806.2661811","type":"proceedings-article","created":{"date-parts":[[2014,11,3]],"date-time":"2014-11-03T14:41:51Z","timestamp":1415025711000},"page":"11-18","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":31,"title":["Multi-scale Temporal Modeling for Dimensional Emotion Recognition in Video"],"prefix":"10.1145","author":[{"given":"Linlin","family":"Chao","sequence":"first","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianhua","family":"Tao","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minghao","family":"Yang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ya","family":"Li","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengqi","family":"Wen","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2014,11,7]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/11573548_125"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2522848.2531739"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.4018\/jse.2010101605"},{"volume-title":"An approach to environmental psychology","author":"Mehrabian A.","key":"e_1_3_2_1_4_1","unstructured":"A. Mehrabian , and J. Russell , An approach to environmental psychology . Cambridge, MA : MIT Press , A. Mehrabian, and J. Russell, An approach to environmental psychology. Cambridge, MA: MIT Press,"},{"key":"e_1_3_2_1_5_1","volume-title":"The communication of emotional meaning (pp. 101--112).New York","author":"Davitz J.","year":"1964","unstructured":"J. Davitz , Auditory correlates of vocal expression of emotional feeling . In J. Davitz (Ed.), The communication of emotional meaning (pp. 101--112).New York : McGraw-Hill , 1964 . J. Davitz, Auditory correlates of vocal expression of emotional feeling. In J. Davitz (Ed.), The communication of emotional meaning (pp. 101--112).New York: McGraw-Hill, 1964."},{"key":"e_1_3_2_1_6_1","volume-title":"What the Face Reveals : Basic and Applied Studies of Spontaneous Expression Using the Facial Action Coding System, seconded","author":"Ekman P.","year":"2005","unstructured":"P. Ekman and E. L. Rosenberg , What the Face Reveals : Basic and Applied Studies of Spontaneous Expression Using the Facial Action Coding System, seconded . Oxford Univ. Press , 2005 . P. Ekman and E. L. Rosenberg, What the Face Reveals : Basic and Applied Studies of Spontaneous Expression Using the Facial Action Coding System, seconded. Oxford Univ. Press, 2005."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","first-page":"637","DOI":"10.1002\/0470013494.ch30","volume-title":"Handbook of Cognition and Emotion","author":"Scherer K. R.","year":"1999","unstructured":"K. R. Scherer , Appraisal Theory , Handbook of Cognition and Emotion , T. Dalgleish and M. J. Power, eds., pp. 637 -- 663 , Wiley , 1999 . K. R. Scherer, Appraisal Theory, Handbook of Cognition and Emotion, T. Dalgleish and M. J. Power, eds., pp. 637--663, Wiley,1999."},{"key":"e_1_3_2_1_8_1","volume-title":"Society of Photo-Optical Instrumentation Engineers (SPIE) Conference Series","volume":"5670","author":"Sebe N.","year":"2004","unstructured":"N. Sebe , I. Cohen , T. Gevers , and T. S. Huang . Multimodal approaches for emotion recognition: a survey. In S. Santini, R. Schettini, and T. Gevers, editors , Society of Photo-Optical Instrumentation Engineers (SPIE) Conference Series , volume 5670 of Society of Photo-Optical Instrumentation Engineers (SPIE) Conference Series, pages56-- 67, Dec. 2004 . N. Sebe, I. Cohen, T. Gevers, and T. S. Huang. Multimodal approaches for emotion recognition: a survey. In S. Santini, R. Schettini, and T. Gevers, editors, Society of Photo-Optical Instrumentation Engineers (SPIE) Conference Series, volume 5670 of Society of Photo-Optical Instrumentation Engineers (SPIE) Conference Series, pages56--67, Dec.2004."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30568-2_27"},{"key":"e_1_3_2_1_10_1","volume-title":"Cognitive Neuroscience of Emotion","author":"Lane R.","year":"2000","unstructured":"R. Lane and L. Nadel , Cognitive Neuroscience of Emotion . Oxford Univ. Press , 2000 . R. Lane and L. Nadel, Cognitive Neuroscience of Emotion. Oxford Univ. Press, 2000."},{"issue":"3","key":"e_1_3_2_1_11_1","first-page":"742","article-title":"Neural correlates of processing valence and arousal in affective words","volume":"17","author":"Lewisetal P. A.","year":"2007","unstructured":"P. A. Lewisetal , Neural correlates of processing valence and arousal in affective words , CerebralCortex ,vol. 17 ,no. 3 , pp. 742 -- 748 , Mar 2007 . P. A. Lewisetal , Neural correlates of processing valence and arousal in affective words, CerebralCortex,vol.17,no.3, pp. 742--748, Mar 2007.","journal-title":"CerebralCortex"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2661806.2661807"},{"key":"e_1_3_2_1_13_1","volume-title":"Affective Computing and Intelligent Interaction (pp. 378--387)","author":"Meng H.","year":"2011","unstructured":"H. Meng , N. Bianchi-Berthouze , Naturalistic Affective Expression Classification by a Multi-stage Approach Based on Hidden Markov Models , In Affective Computing and Intelligent Interaction (pp. 378--387) . Springer Berlin Heidelberg , 2011 . H. Meng, N. Bianchi-Berthouze, Naturalistic Affective Expression Classification by a Multi-stage Approach Based on Hidden Markov Models, In Affective Computing and Intelligent Interaction (pp. 378--387). Springer Berlin Heidelberg, 2011."},{"key":"e_1_3_2_1_14_1","volume-title":"Affective Computing and Intelligent Interaction (pp. 378--387)","author":"Glodek M.","year":"2011","unstructured":"M. Glodek , S. Tschechne , G. Layher , M. Schels , T. Brosch , S. Scherer , F. Schwenker , Multiple Classifier Systems for the Classification of Audio-Visual Emotion States , In Affective Computing and Intelligent Interaction (pp. 378--387) . Springer Berlin Heidelberg , 2011 . M. Glodek, S. Tschechne, G. Layher, M. Schels, T. Brosch, S. Scherer, F. Schwenker, Multiple Classifier Systems for the Classification of Audio-Visual Emotion States, In Affective Computing and Intelligent Interaction (pp. 378--387). Springer Berlin Heidelberg, 2011."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.5555\/2062850.2062905"},{"key":"e_1_3_2_1_16_1","volume-title":"Image and Vision Computing","author":"W\u00f6llmer M.","year":"2012","unstructured":"M. W\u00f6llmer , M. Kaiser , F. Eyben , B. Schuller, LSTM-Modeling of continuous emotions in an audiovisual affect recognition framework , Image and Vision Computing , 2012 . M. W\u00f6llmer, M. Kaiser, F. Eyben, B. Schuller, LSTM-Modeling of continuous emotions in an audiovisual affect recognition framework, Image and Vision Computing, 2012."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","first-page":"597","DOI":"10.21437\/Interspeech.2008-192","volume-title":"Proc. Interspeech","author":"W\u00f6llmer M.","year":"2008","unstructured":"M. W\u00f6llmer , F. Eyben , S. Reiter , B. Schuller , C. Cox , E. Douglas-Cowie , R. Cowie, Abandoning emotion classes -- towards continuous emotion recognition with modeling of long-range dependences , In Proc. Interspeech , pp. 597 -- 600 , 2008 . M. W\u00f6llmer, F. Eyben, S. Reiter, B. Schuller, C. Cox, E. Douglas-Cowie, R. Cowie, Abandoning emotion classes -- towards continuous emotion recognition with modeling of long-range dependences, In Proc. Interspeech, pp. 597--600, 2008."},{"key":"e_1_3_2_1_18_1","volume-title":"Combining Long Short-Term Memory and Dynamic Bayesian Networks for Incremental Emotion-Sensitive Artificial Listening","author":"W\u00f6llmer M.","year":"2010","unstructured":"M. W\u00f6llmer , B. Schuller , F. Eyben , G. Rigoll , Combining Long Short-Term Memory and Dynamic Bayesian Networks for Incremental Emotion-Sensitive Artificial Listening , IEEE Journal of Selected Topics in Signal Processing (J-STSP), Special Issue on Speech Processing for Natural Interaction with Intelligent Environments, Vol 4, Issue 5, 867--881, 2010 M. W\u00f6llmer, B. Schuller, F. Eyben, G. Rigoll, Combining Long Short-Term Memory and Dynamic Bayesian Networks for Incremental Emotion-Sensitive Artificial Listening, IEEE Journal of Selected Topics in Signal Processing (J-STSP), Special Issue on Speech Processing for Natural Interaction with Intelligent Environments, Vol 4, Issue 5, 867--881, 2010"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2388676.2388783"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2512530.2512532"},{"key":"e_1_3_2_1_21_1","first-page":"185","volume-title":"From the Lab to the Real World: Affect Recognition Usng","author":"Gunes H.","year":"2008","unstructured":"H. Gunes , M. Piccardi , and M. Pantic , From the Lab to the Real World: Affect Recognition Usng , Affective Computing : Focus on Emotion Expression, Synthesis, and Recognition. I-Tech Education and Publishing , Vienna, Austria, pp. 185 - 218 , 2008 . H. Gunes, M. Piccardi, and M. Pantic, From the Lab to the Real World: Affect Recognition Usng, Affective Computing: Focus on Emotion Expression, Synthesis, and Recognition. I-Tech Education and Publishing, Vienna, Austria, pp. 185 - 218, 2008."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2011.12.005"},{"volume-title":"Speech and Signal Processing (ICASSP), 2011 IEEE International Conference on (pp. 5688--5691)","author":"Stuhlsatz A.","key":"e_1_3_2_1_23_1","unstructured":"A. Stuhlsatz , C. Meyer , F. Eyben , T. ZieIke , G. Meier , and B. Schuller , Deep neural networks for acoustic emotion recognition: raising the benchmarks. In Acoustics , Speech and Signal Processing (ICASSP), 2011 IEEE International Conference on (pp. 5688--5691) . IEEE. A. Stuhlsatz, C. Meyer, F. Eyben, T. ZieIke, G. Meier, and B. Schuller, Deep neural networks for acoustic emotion recognition: raising the benchmarks. In Acoustics, Speech and Signal Processing (ICASSP), 2011 IEEE International Conference on (pp. 5688--5691). IEEE."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2522848.2531745"},{"key":"e_1_3_2_1_25_1","volume-title":"Czech Republic.","author":"Le D.","year":"2013","unstructured":"D. Le and E. M. Provost . Emotion Recognition from Spontaneous Speech using Hidden Markov Models with Deep Belief Networks, Automatic Speech Recognition and Understanding (ASRU). Olomouc , Czech Republic. December , 2013 . D. Le and E. M. Provost. Emotion Recognition from Spontaneous Speech using Hidden Markov Models with Deep Belief Networks, Automatic Speech Recognition and Understanding (ASRU). Olomouc, Czech Republic. December, 2013."},{"key":"e_1_3_2_1_26_1","volume-title":"International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Kim Y.","year":"2013","unstructured":"Y. Kim , H. Lee , and E. M. Provost , Deep Learning for Robust Feature Generation in Audio-Visual Emotion Recognition , International Conference on Acoustics, Speech and Signal Processing (ICASSP) . Vancouver, British Columbia, Canada. May , 2013 . Y. Kim, H. Lee, and E. M. Provost, Deep Learning for Robust Feature Generation in Audio-Visual Emotion Recognition, International Conference on Acoustics, Speech and Signal Processing (ICASSP) . Vancouver, British Columbia, Canada. May, 2013."},{"key":"e_1_3_2_1_27_1","volume-title":"ISMIR (pp. 729--734)","author":"Hamel P.","year":"2011","unstructured":"P. Hamel , S. Lemieux , Y. Bengio , and D. Eck , Temporal pooling and multiscale learning for automatic annotation and ranking of music audio . In ISMIR (pp. 729--734) , 2011 . P. Hamel, S. Lemieux, Y. Bengio, and D. Eck, Temporal pooling and multiscale learning for automatic annotation and ranking of music audio. In ISMIR (pp. 729--734), 2011."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2512530.2512533"},{"key":"e_1_3_2_1_29_1","volume-title":"Utrecht","author":"Mathieu B.","year":"2010","unstructured":"B. Mathieu , S. Essid , T. Fillon , J. Prado , G. Richard , YAAFE, an Easy to Use and Efficient Audio Feature Extraction Software , proceedings of the 11th ISMIR conference , Utrecht , Netherlands , 2010 . B. Mathieu, S. Essid, T. Fillon, J. Prado, G. Richard, YAAFE, an Easy to Use and Efficient Audio Feature Extraction Software, proceedings of the 11th ISMIR conference, Utrecht, Netherlands, 2010."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.75"},{"key":"e_1_3_2_1_31_1","volume-title":"International Journal of Computer Vision (IJCV)","author":"Viola P.","year":"2001","unstructured":"P. Viola and M. Jones , Robust real time object detection {J} , International Journal of Computer Vision (IJCV) , 2001 . P. Viola and M. Jones, Robust real time object detection {J}, International Journal of Computer Vision (IJCV), 2001."},{"key":"e_1_3_2_1_32_1","volume-title":"AISTATS","author":"Coates A.","year":"2011","unstructured":"A. Coates , H. Lee , and A. Y. Ng . An Analysis of single-layer networks in unsupervised feature learning . In AISTATS , 2011 . A. Coates, H. Lee, and A. Y. Ng. An Analysis of single-layer networks in unsupervised feature learning. In AISTATS, 2011."},{"key":"e_1_3_2_1_33_1","first-page":"17","article-title":"Deep learning of representations for unsupervised and transfer learning","volume":"2012","author":"Bengio Y.","unstructured":"Y. Bengio , Deep learning of representations for unsupervised and transfer learning , ICML Unsupervised and Transfer Learning , 2012 : 17 -- 36 Y. Bengio, Deep learning of representations for unsupervised and transfer learning, ICML Unsupervised and Transfer Learning, 2012: 17--36","journal-title":"ICML Unsupervised and Transfer Learning"},{"key":"e_1_3_2_1_34_1","volume-title":"Combining Emotional History Through Multimodal Fusion Methods","author":"Chao L.","year":"2013","unstructured":"L. Chao , J. Tao and M. Yang , Combining Emotional History Through Multimodal Fusion Methods . Asia Pacific Signal and Information Processing Association (APSIPA 2013 ), Oct.29-Nov.1 2013 . L. Chao, J. Tao and M. Yang, Combining Emotional History Through Multimodal Fusion Methods. Asia Pacific Signal and Information Processing Association (APSIPA 2013), Oct.29-Nov.1 2013 ."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2388676.2388782"},{"key":"e_1_3_2_1_36_1","volume-title":"Science","volume":"313","author":"Hinton G. E.","unstructured":"G. E. Hinton , and R. R. Salakhutdinov , Reducing the dimensionality of data with neural networks . Science , Vol. 313 . no. 5786, pp. 504 - 507, 28 July 2006. G. E. Hinton, and R. R. Salakhutdinov, Reducing the dimensionality of data with neural networks. Science, Vol. 313. no. 5786, pp. 504 - 507, 28 July 2006."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.5772\/54002"}],"event":{"name":"MM '14: 2014 ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Orlando Florida USA","acronym":"MM '14"},"container-title":["Proceedings of the 4th International Workshop on Audio\/Visual Emotion Challenge"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2661806.2661811","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2661806.2661811","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T06:12:55Z","timestamp":1750227175000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2661806.2661811"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,11,7]]},"references-count":37,"alternative-id":["10.1145\/2661806.2661811","10.1145\/2661806"],"URL":"https:\/\/doi.org\/10.1145\/2661806.2661811","relation":{},"subject":[],"published":{"date-parts":[[2014,11,7]]},"assertion":[{"value":"2014-11-07","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}