{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T13:42:23Z","timestamp":1769694143241,"version":"3.49.0"},"publisher-location":"Cham","reference-count":21,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783319773797","type":"print"},{"value":"9783319773803","type":"electronic"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-319-77380-3_20","type":"book-chapter","created":{"date-parts":[[2018,5,9]],"date-time":"2018-05-09T14:59:02Z","timestamp":1525877942000},"page":"205-214","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Multimodal Fusion of Spatial-Temporal Features for Emotion Recognition in the Wild"],"prefix":"10.1007","author":[{"given":"Zuchen","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7085-8876","authenticated-orcid":false,"given":"Yuchun","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,5,10]]},"reference":[{"key":"20_CR1","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1109\/MMUL.2012.26","volume":"19","author":"A Dhall","year":"2012","unstructured":"Dhall, A., Goecke, R., Lucey, S., Gedeon, T.: Collecting large, richly annotated facial-expression databases from movies. IEEE Multimedia 19, 34\u201341 (2012)","journal-title":"IEEE Multimedia"},{"key":"20_CR2","doi-asserted-by":"crossref","unstructured":"Chen, J., Chen, Z., Chi, Z., Fu, H.: Emotion recognition in the wild with feature fusion and multiple kernel learning. In: Proceedings of the 16th International Conference on Multimodal Interaction, pp. 508\u2013513 (2014)","DOI":"10.1145\/2663204.2666277"},{"key":"20_CR3","doi-asserted-by":"crossref","unstructured":"Sun, B., Li, L., Zuo, T., Chen, Y., Zhou, G., Wu, X.: Combining multimodal features with hierarchical classifier fusion for emotion recognition in the wild. In: Proceedings of the 16th International Conference on Multimodal Interaction, pp. 481\u2013486 (2014)","DOI":"10.1145\/2663204.2666272"},{"key":"20_CR4","doi-asserted-by":"crossref","unstructured":"Liu, M., Wang, R., Li, S., Shan, S., Huang, Z., Chen, X.: Combining multiple kernel methods on riemannian manifold for emotion recognition in the wild. In: Proceedings of the 16th International Conference on Multimodal Interaction, pp. 494\u2013501 (2014)","DOI":"10.1145\/2663204.2666274"},{"key":"20_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"425","DOI":"10.1007\/978-3-319-46475-6_27","volume-title":"Computer Vision \u2013 ECCV 2016","author":"X Zhao","year":"2016","unstructured":"Zhao, X., Liang, X., Liu, L., Li, T., Han, Y., Vasconcelos, N., Yan, S.: Peak-piloted deep network for facial expression recognition. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9906, pp. 425\u2013442. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46475-6_27"},{"key":"20_CR6","unstructured":"Carrier, P.L., Courville, A., Goodfellow, I.J., Mirza, M., Bengio, Y.: FER-2013 face database. Technical report (2013)"},{"key":"20_CR7","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., Paluri, M.: Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4489\u20134497 (2015)","DOI":"10.1109\/ICCV.2015.510"},{"key":"20_CR8","doi-asserted-by":"publisher","first-page":"915","DOI":"10.1109\/TPAMI.2007.1110","volume":"29","author":"G Zhao","year":"2007","unstructured":"Zhao, G., Pietikainen, M.: Dynamic texture recognition using local binary patterns with an application to facial expressions. IEEE Trans. Pattern Anal. Mach. Intell. 29, 915\u2013928 (2007)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"20_CR9","doi-asserted-by":"crossref","unstructured":"Schuller, B., Steidl, S., Batliner, A., et al.: The INTERSPEECH 2010 paralinguistic challenge. In: Conference of the International Speech Communication Association, INTERSPEECH 2010, pp. 2794\u20132797 (2010)","DOI":"10.21437\/Interspeech.2010-739"},{"key":"20_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"415","DOI":"10.1007\/978-3-642-24571-8_53","volume-title":"Affective Computing and Intelligent Interaction","author":"B Schuller","year":"2011","unstructured":"Schuller, B., Valstar, M., Eyben, F., McKeown, G., Cowie, R., Pantic, M.: AVEC 2011\u2013the first international audio\/visual emotion challenge. In: D\u2019Mello, S., Graesser, A., Schuller, B., Martin, J.-C. (eds.) ACII 2011. LNCS, vol. 6975, pp. 415\u2013424. Springer, Heidelberg (2011). https:\/\/doi.org\/10.1007\/978-3-642-24571-8_53"},{"key":"20_CR11","doi-asserted-by":"crossref","unstructured":"Dhall, A., Goecke, R., Joshi, J., Sikka, K., Gedeon, T.: Emotion recognition in the wild challenge 2014: baseline, data and protocol. In: Proceedings of the 16th International Conference on Multimodal Interaction, pp. 461\u2013466 (2014)","DOI":"10.1145\/2663204.2666275"},{"key":"20_CR12","doi-asserted-by":"crossref","unstructured":"Ringeval, F., Amiriparian, S., Eyben, F., Scherer, K., Schuller, B.: Emotion recognition in the wild: incorporating voice and lip activity in multimodal decision-level fusion. In: Proceedings of the 16th International Conference on Multimodal Interaction, pp. 473\u2013480 (2014)","DOI":"10.1145\/2663204.2666271"},{"key":"20_CR13","doi-asserted-by":"crossref","unstructured":"Khorrami, P., Le Paine, T., Brady, K., Dagli, C., Huang, T.: How deep neural networks can improve emotion recognition on video data. In: 2016 IEEE International Conference on Image Processing (ICIP), pp. 619\u2013623 (2016)","DOI":"10.1109\/ICIP.2016.7532431"},{"key":"20_CR14","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"20_CR15","doi-asserted-by":"crossref","unstructured":"Jung, H., Lee, S., Yim, J., Park, S., Kim, J.: Joint fine-tuning in deep neural networks for facial expression recognition. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2983\u20132991 (2015)","DOI":"10.1109\/ICCV.2015.341"},{"key":"20_CR16","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1007\/s12193-015-0175-6","volume":"10","author":"H Kaya","year":"2016","unstructured":"Kaya, H., Salah, A.: Combining modality-specific extreme learning machines for emotion recognition in the wild. J. Multimodal User Interfaces 10, 139\u2013149 (2016)","journal-title":"J. Multimodal User Interfaces"},{"key":"20_CR17","doi-asserted-by":"crossref","unstructured":"Huang, X., He, Q., Hong, X., Zhao, G., Pietikainen, M.: Improved spatiotemporal local monogenic binary pattern for emotion recognition in the wild. In: Proceedings of the 16th International Conference on Multimodal Interaction, pp. 514\u2013520 (2014)","DOI":"10.1145\/2663204.2666278"},{"key":"20_CR18","doi-asserted-by":"publisher","first-page":"1319","DOI":"10.1109\/TMM.2016.2557721","volume":"18","author":"J Yan","year":"2016","unstructured":"Yan, J., Zheng, W., Xu, Q., Lu, G., Li, H., Wang, B.: Sparse kernel reduced-rank regression for bimodal emotion recognition from facial expression and speech. IEEE Trans. Multimedia 18, 1319\u20131329 (2016)","journal-title":"IEEE Trans. Multimedia"},{"key":"20_CR19","doi-asserted-by":"crossref","unstructured":"Fan, Y., Lu, X., Li, D., Liu, Y.: Video-based emotion recognition using CNN-RNN and C3D hybrid networks. In: Proceedings of the 18th ACM International Conference on Multimodal Interaction, pp. 445\u2013450 (2016)","DOI":"10.1145\/2993148.2997632"},{"key":"20_CR20","doi-asserted-by":"crossref","unstructured":"Cho, K., Van Merri\u00ebnboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., Bengio, Y.: Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078 (2014)","DOI":"10.3115\/v1\/D14-1179"},{"key":"20_CR21","doi-asserted-by":"crossref","unstructured":"Eyben, F., W\u00f6llmer, M., Schuller, B.: OpenSmile: the Munich versatile and fast open-source audio feature extractor. In: ACM International Conference on Multimedia, pp. 1459\u20131462 (2010)","DOI":"10.1145\/1873951.1874246"}],"container-title":["Lecture Notes in Computer Science","Advances in Multimedia Information Processing \u2013 PCM 2017"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-77380-3_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,22]],"date-time":"2022-08-22T23:48:56Z","timestamp":1661212136000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-77380-3_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9783319773797","9783319773803"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-77380-3_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018]]}}}