{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T12:01:48Z","timestamp":1778587308185,"version":"3.51.4"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"13","license":[{"start":{"date-parts":[[2021,1,2]],"date-time":"2021-01-02T00:00:00Z","timestamp":1609545600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,2]],"date-time":"2021-01-02T00:00:00Z","timestamp":1609545600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100009033","name":"Center of Innovation Program","doi-asserted-by":"publisher","award":["JPMJCE1314"],"award-info":[{"award-number":["JPMJCE1314"]}],"id":[{"id":"10.13039\/501100009033","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2021,7]]},"DOI":"10.1007\/s00521-020-05557-4","type":"journal-article","created":{"date-parts":[[2021,1,2]],"date-time":"2021-01-02T10:03:09Z","timestamp":1609581789000},"page":"7381-7392","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":25,"title":["Enhanced convolutional LSTM with spatial and temporal skip connections and temporal gates for facial expression recognition from video"],"prefix":"10.1007","volume":"33","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4676-1959","authenticated-orcid":false,"given":"Ryo","family":"Miyoshi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Noriko","family":"Nagata","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manabu","family":"Hashimoto","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,1,2]]},"reference":[{"issue":"2","key":"5557_CR1","doi-asserted-by":"publisher","first-page":"124","DOI":"10.1037\/h0030377","volume":"17","author":"P Ekman","year":"1971","unstructured":"Ekman P, Friesen WV (1971) Constants across cultures in the face and emotion. J Personal Soc Psychol 17(2):124","journal-title":"J Personal Soc Psychol"},{"key":"5557_CR2","doi-asserted-by":"crossref","unstructured":"Bartlett MS, Littlewort G, Fasel I, Movellan JR (2003) Real time face detection and facial expression recognition: development and applications to human computer interaction. In 2003 conference on computer vision and pattern recognition workshop, vol\u00a05. IEEE, pp 53\u201353","DOI":"10.1109\/CVPRW.2003.10057"},{"issue":"2","key":"5557_CR3","doi-asserted-by":"publisher","first-page":"159","DOI":"10.1007\/BF00992253","volume":"10","author":"P Ekman","year":"1986","unstructured":"Ekman P, Friesen WV (1986) A new pan-cultural facial expression of emotion. Motiv Emot 10(2):159\u2013168","journal-title":"Motiv Emot"},{"issue":"5","key":"5557_CR4","doi-asserted-by":"publisher","first-page":"403","DOI":"10.1111\/j.0956-7976.2005.01548.x","volume":"16","author":"Z Ambadar","year":"2005","unstructured":"Ambadar Z, Schooler JW, Cohn JF (2005) Deciphering the enigmatic face: the importance of facial dynamics in interpreting subtle facial expressions. Psychol Sci 16(5):403\u2013410","journal-title":"Psychol Sci"},{"key":"5557_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.sigpro.2015.04.007","volume":"117","author":"W-L Chao","year":"2015","unstructured":"Chao W-L, Ding J-J, Liu J-Z (2015) Facial expression recognition based on improved local binary pattern and class-regularized locality preserving projection. Signal Process 117:1\u201310","journal-title":"Signal Process"},{"key":"5557_CR6","doi-asserted-by":"crossref","unstructured":"Liu P, Han S, Meng Z, Tong Y (2014) Facial expression recognition via a boosted deep belief network. In: 2014 IEEE conference on computer vision and pattern recognition, pp 1805\u20131812","DOI":"10.1109\/CVPR.2014.233"},{"key":"5557_CR7","doi-asserted-by":"crossref","unstructured":"De la\u00a0Torre\u00a0Frade F, Chu W-S, Xiong X, Carrasco F\u00a0V, Ding X, Cohn J (2015) Intraface. In: Automatic face and gesture recognition","DOI":"10.1109\/FG.2015.7163082"},{"key":"5557_CR8","doi-asserted-by":"crossref","unstructured":"Mollahosseini A, Chan D, Mahoor MH (2016) Going deeper in facial expression recognition using deep neural networks. In: 2016 IEEE winter conference on applications of computer vision (WACV). IEEE, pp 1\u201310","DOI":"10.1109\/WACV.2016.7477450"},{"key":"5557_CR9","doi-asserted-by":"publisher","first-page":"610","DOI":"10.1016\/j.patcog.2016.07.026","volume":"61","author":"AT Lopes","year":"2017","unstructured":"Lopes AT, de Aguiar E, De Souza AF, Oliveira-Santos T (2017) Facial expression recognition with convolutional neural networks: coping with few data and the training sample order. Pattern Recognit 61:610\u2013628","journal-title":"Pattern Recognit"},{"key":"5557_CR10","doi-asserted-by":"crossref","unstructured":"Ding H, Zhou SK, Chellappa R (2017) Facenet2expnet: regularizing a deep face recognition net for expression recognition. In: 2017 12th IEEE international conference on automatic face and gesture recognition (FG 2017). IEEE, pp 118\u2013126","DOI":"10.1109\/FG.2017.23"},{"issue":"2","key":"5557_CR11","doi-asserted-by":"publisher","first-page":"277","DOI":"10.1007\/s11042-009-0344-2","volume":"49","author":"M Mansoorizadeh","year":"2010","unstructured":"Mansoorizadeh M, Charkari NM (2010) Multimodal information fusion application to human emotion recognition from face and speech. Multimed Tools Appl 49(2):277\u2013297","journal-title":"Multimed Tools Appl"},{"issue":"2","key":"5557_CR12","doi-asserted-by":"publisher","first-page":"399","DOI":"10.1007\/s00521-012-1228-3","volume":"24","author":"M Bejani","year":"2014","unstructured":"Bejani M, Gharavian D, Charkari NM (2014) Audiovisual emotion recognition using ANOVA feature selection method and multi-classifier neural networks. Neural Comput Appl 24(2):399\u2013412","journal-title":"Neural Comput Appl"},{"issue":"3","key":"5557_CR13","doi-asserted-by":"publisher","first-page":"300","DOI":"10.1109\/TAFFC.2016.2553038","volume":"8","author":"S Zhalehpour","year":"2017","unstructured":"Zhalehpour S, Onder O, Akhtar Z, Erdem CE (2017) Baum-1: a spontaneous audio-visual face database of affective and mental states. IEEE Trans Affect Comput 8(3):300\u2013313","journal-title":"IEEE Trans Affect Comput"},{"key":"5557_CR14","doi-asserted-by":"crossref","unstructured":"Khorrami P, Le Paine T, Brady K, Dagli C, Huang TS (2016) How deep neural networks can improve emotion recognition on video data. In: 2016 IEEE international conference on image processing (ICIP), pp 619\u2013623","DOI":"10.1109\/ICIP.2016.7532431"},{"key":"5557_CR15","doi-asserted-by":"publisher","first-page":"48807","DOI":"10.1109\/ACCESS.2019.2907271","volume":"7","author":"X Pan","year":"2019","unstructured":"Pan X, Ying G, Chen G, Li H, Li W (2019) A deep spatial and temporal aggregation framework for video-based facial expression recognition. IEEE Access 7:48807\u201348815","journal-title":"IEEE Access"},{"key":"5557_CR16","doi-asserted-by":"crossref","unstructured":"Hara K, Kataoka H, Satoh Y (2017) Learning spatio-temporal features with 3d residual networks for action recognition. In: Proceedings of the IEEE international conference on computer vision, pp 3154\u20133160","DOI":"10.1109\/ICCVW.2017.373"},{"key":"5557_CR17","unstructured":"Tran D, Ray J, Shou Z, Chang SF, Paluri M (2017) Convnet architecture search for spatiotemporal feature learning. arXiv:1708.05038"},{"key":"5557_CR18","doi-asserted-by":"crossref","unstructured":"Hara K, Kataoka H, Satoh Y (2018) Can spatiotemporal 3d cnns retrace the history of 2d cnns and imagenet? In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6546\u20136555","DOI":"10.1109\/CVPR.2018.00685"},{"key":"5557_CR19","unstructured":"Shi X, Chen Z, Wang H, Yeung D-Y, Wong W, Woo WC (2015) Convolutional LSTM network: a machine learning approach for precipitation nowcasting. In: Cortes C, Lawrence ND, Lee DD, Sugiyama M, Garnett R (eds) Advances in neural information processing systems, vol 28. Curran Associates, Inc, pp 802\u2013810"},{"key":"5557_CR20","doi-asserted-by":"crossref","unstructured":"Lucey P, Cohn JF, Kanade T, Saragih J, Ambadar Z, Matthews I (2010) The extended cohn-kanade dataset (ck+): a complete dataset for action unit and emotion-specified expression. In: 2010 IEEE computer society conference on computer vision and pattern recognition\u2014workshops, pp 94\u2013101","DOI":"10.1109\/CVPRW.2010.5543262"},{"key":"5557_CR21","doi-asserted-by":"crossref","unstructured":"Martin O, Kotsia I, Macq B, Pitas I (2006) The enterface\u201905 audio-visual emotion database. In: 22nd international conference on data engineering workshops (ICDEW\u201906). IEEE, pp 8\u20138","DOI":"10.1109\/ICDEW.2006.145"},{"key":"5557_CR22","unstructured":"Pantic M, Valstar M, Rademaker R, Maat L (2005) Web-based database for facial expression analysis. In: 2005 IEEE international conference on multimedia and expo, p 5"},{"key":"5557_CR23","doi-asserted-by":"crossref","unstructured":"Caba\u00a0Heilbron F, Escorcia V, Ghanem B, Carlos\u00a0Niebles J (2015) Activitynet: a large-scale video benchmark for human activity understanding. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 961\u2013970","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"5557_CR24","unstructured":"Soomro K, Zamir AR, Shah M (2012) Ucf101: a dataset of 101 human actions classes from videos in the wild. arXiv:1212.0402"},{"key":"5557_CR25","unstructured":"Kay W, Carreira J, Simonyan K, Zhang B, Hillier C, Vijayanarasimhan S, Viola F, Green T, Back T, Natsev P et al (2017) The kinetics human action video dataset. arXiv:1705.06950"},{"key":"5557_CR26","unstructured":"Wang Y, Jiang L, Yang MH, Li LJ, Long M, Fei-Fei L (2019) Eidetic 3d lstm: a model for video prediction and beyond. In: ICLR"},{"key":"5557_CR27","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv:1409.1556"},{"issue":"6","key":"5557_CR28","doi-asserted-by":"publisher","first-page":"1333","DOI":"10.1109\/72.963769","volume":"12","author":"FA Gers","year":"2001","unstructured":"Gers FA, Schmidhuber E (2001) LSTM recurrent networks learn simple context-free and context-sensitive languages. IEEE Trans Neural Netw 12(6):1333\u20131340","journal-title":"IEEE Trans Neural Netw"},{"key":"5557_CR29","doi-asserted-by":"crossref","unstructured":"Wu Y, He K (2018) Group normalization. In: Proceedings of the European conference on computer vision (ECCV), pp 3\u201319","DOI":"10.1007\/978-3-030-01261-8_1"},{"key":"5557_CR30","doi-asserted-by":"crossref","unstructured":"Baltrusaitis T, Zadeh A, Lim YC, Morency LP (2018) Openface 2.0: facial behavior analysis toolkit. In: 2018 13th IEEE international conference on automatic face and gesture recognition (FG 2018). IEEE, pp 59\u201366","DOI":"10.1109\/FG.2018.00019"},{"key":"5557_CR31","doi-asserted-by":"crossref","unstructured":"Farneb\u00e4ck G (2003) Two-frame motion estimation based on polynomial expansion. In: Scandinavian conference on Image analysis. Springer, pp 363\u2013370","DOI":"10.1007\/3-540-45103-X_50"},{"key":"5557_CR32","doi-asserted-by":"crossref","unstructured":"Liu M, Shan S, Wang R, Chen X (2014) Learning expressionlets on spatio-temporal manifold for dynamic facial expression recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1749\u20131756","DOI":"10.1109\/CVPR.2014.226"},{"key":"5557_CR33","doi-asserted-by":"crossref","unstructured":"Kuo CM, Lai SH, Sarkis M (2018) A compact deep learning model for robust facial expression recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition workshops, pp 2121\u20132129","DOI":"10.1109\/CVPRW.2018.00286"},{"key":"5557_CR34","doi-asserted-by":"crossref","unstructured":"Meng D, Peng X, Wang K, Qiao Y (2019) Frame attention networks for facial expression recognition in videos. In: 2019 IEEE international conference on image processing (ICIP). IEEE, pp 3866\u20133870","DOI":"10.1109\/ICIP.2019.8803603"},{"issue":"9","key":"5557_CR35","doi-asserted-by":"publisher","first-page":"4193","DOI":"10.1109\/TIP.2017.2689999","volume":"26","author":"K Zhang","year":"2017","unstructured":"Zhang K, Huang Y, Yong D, Wang L (2017) Facial expression recognition based on deep evolutional spatial-temporal networks. IEEE Trans Image Process 26(9):4193\u20134203","journal-title":"IEEE Trans Image Process"},{"key":"5557_CR36","doi-asserted-by":"crossref","unstructured":"Jung H, Lee S, Yim J, Park S, Kim J (2015) Joint fine-tuning in deep neural networks for facial expression recognition. In: Proceedings of the IEEE international conference on computer vision, pp 2983\u20132991","DOI":"10.1109\/ICCV.2015.341"},{"key":"5557_CR37","doi-asserted-by":"crossref","unstructured":"Cai J, Meng Z, Khan AS, Li Z, O\u2019Reilly J, Tong Y (2018) Island loss for learning discriminative features in facial expression recognition. In: 2018 13th IEEE international conference on automatic face and gesture recognition (FG 2018). IEEE, pp 302\u2013309","DOI":"10.1109\/FG.2018.00051"},{"key":"5557_CR38","doi-asserted-by":"crossref","unstructured":"Sikka K, Sharma G, Bartlett M (2016) Lomo: latent ordinal model for facial analysis in videos. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5580\u20135589","DOI":"10.1109\/CVPR.2016.602"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-020-05557-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-020-05557-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-020-05557-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,6,24]],"date-time":"2021-06-24T06:18:23Z","timestamp":1624515503000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-020-05557-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,2]]},"references-count":38,"journal-issue":{"issue":"13","published-print":{"date-parts":[[2021,7]]}},"alternative-id":["5557"],"URL":"https:\/\/doi.org\/10.1007\/s00521-020-05557-4","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,1,2]]},"assertion":[{"value":"20 March 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 November 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 January 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}