{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T13:07:43Z","timestamp":1775653663267,"version":"3.50.1"},"reference-count":121,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach. Intell. Res."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s11633-025-1629-x","type":"journal-article","created":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T10:35:53Z","timestamp":1775644553000},"page":"308-330","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Micro-gesture Recognition: A Comprehensive Survey of Datasets, Methods, and Challenges"],"prefix":"10.1007","volume":"23","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-3492-4068","authenticated-orcid":false,"given":"Taorui","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xun","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0530-2123","authenticated-orcid":false,"given":"Yong","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qilang","family":"Ye","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dan","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sergio","family":"Escalera","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ghada","family":"Khoriba","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6505-3304","authenticated-orcid":false,"given":"Zitong","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,8]]},"reference":[{"issue":"6111","key":"1629_CR1","doi-asserted-by":"publisher","first-page":"1225","DOI":"10.1126\/science.1224313","volume":"338","author":"H Aviezer","year":"2012","unstructured":"H. Aviezer, Y. Trope, A. Todorov. Body cues, not facial expressions, discriminate between intense positive and negative emotions. Science, vol. 338, no. 6111, pp. 1225\u20131229, 2012. DOI: https:\/\/doi.org\/10.1126\/science.1224313.","journal-title":"Science"},{"key":"1629_CR2","volume-title":"Gestures: The Do\u2019s and Taboos of Body Language Around the World","author":"R E Axtell","year":"1998","unstructured":"R. E. Axtell, M. Fornwald. Gestures: The Do\u2019s and Taboos of Body Language Around the World, New York, USA: Wiley, 1998."},{"key":"1629_CR3","doi-asserted-by":"publisher","DOI":"10.4324\/9780429475283","volume-title":"Gestalt Therapy: The Art of Contact","author":"S Ginger","year":"2007","unstructured":"S. Ginger. Gestalt Therapy: The Art of Contact, London, UK: Routledge, 2007. DOI: https:\/\/doi.org\/10.4324\/9780429475283."},{"key":"1629_CR4","volume-title":"Nonverbal Communication: The Unspoken Dialogue","author":"J K Burgoon","year":"1996","unstructured":"J. K. Burgoon, D. B. Buller, W. G. Woodall. Nonverbal Communication: The Unspoken Dialogue, 2nd ed., New York, USA: McGraw-Hill, 1996.","edition":"2nd ed."},{"issue":"12","key":"1629_CR5","doi-asserted-by":"publisher","first-page":"1743","DOI":"10.1016\/j.imavis.2008.11.007","volume":"27","author":"A Vinciarelli","year":"2009","unstructured":"A. Vinciarelli, M. Pantic, H. Bourlard. Social signal processing: Survey of an emerging domain. Image and Vision Computing, vol. 27, no. 12, pp. 1743\u20131759, 2009. DOI: https:\/\/doi.org\/10.1016\/j.imavis.2008.11.007.","journal-title":"Image and Vision Computing"},{"key":"1629_CR6","volume-title":"The Definitive Book of Body Language: The Hidden Meaning Behind People\u2019s Gestures and Expressions","author":"B Pease","year":"2008","unstructured":"B. Pease, A. Pease. The Definitive Book of Body Language: The Hidden Meaning Behind People\u2019s Gestures and Expressions, New York, USA: Bantam, 2008."},{"issue":"3","key":"1629_CR7","doi-asserted-by":"publisher","first-page":"1195","DOI":"10.1109\/TAFFC.2020.2981446","volume":"13","author":"S Li","year":"2022","unstructured":"S. Li, W. Deng. Deep facial expression recognition: A survey. IEEE Transactions on Affective Computing, vol. 13, no. 3, pp. 1195\u20131215, 2022. DOI: https:\/\/doi.org\/10.1109\/TAFFC.2020.2981446.","journal-title":"IEEE Transactions on Affective Computing"},{"key":"1629_CR8","doi-asserted-by":"publisher","first-page":"248","DOI":"10.1093\/acprof:oso\/9780195333176.003.0015","volume-title":"The Science of Social Vision","author":"M Shiffrar","year":"2011","unstructured":"M. Shiffrar, M. D. Kaiser, A. Chouchourelou. Seeing human movement as inherently social. The Science of Social Vision, R. B. Adams, N. Ambady, K. Nakayama, S. Shimojo, Eds., Oxford, UK: Oxford Academic, pp. 248\u2013263, 2011. DOI: https:\/\/doi.org\/10.1093\/acprof:oso\/9780195333176.003.0015."},{"issue":"3","key":"1629_CR9","doi-asserted-by":"publisher","first-page":"572","DOI":"10.1016\/j.patcog.2010.09.020","volume":"44","author":"M El Ayadi","year":"2011","unstructured":"M. El Ayadi, M. S. Kamel, F. Karray. Survey on speech emotion recognition: Features, classification schemes, and databases. Pattern Recognition, vol. 44, no. 3, pp. 572\u2013587, 2011. DOI: https:\/\/doi.org\/10.1016\/j.patcog.2010.09.020.","journal-title":"Pattern Recognition"},{"issue":"3","key":"1629_CR10","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1207\/s15327965pli0903_1","volume":"9","author":"J Polivy","year":"1998","unstructured":"J. Polivy. The effects of behavioral inhibition: Integrating internal cues, cognition, behavior, and affect. Psychological Inquiry, vol. 9, no. 3, pp. 181\u2013204, 1998. DOI: https:\/\/doi.org\/10.1207\/s15327965pli0903_1.","journal-title":"Psychological Inquiry"},{"key":"1629_CR11","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2019.8756513","volume-title":"Proceedings of the 14th IEEE International Conference on Automatic Face & Gesture Recognition","author":"H Chen","year":"2019","unstructured":"H. Chen, X. Liu, H. Li, H. Shi, G. Zhao. Analyze spontaneous gestures for emotional stress state recognition: A micro-gesture dataset and analysis with deep learning. In Proceedings of the 14th IEEE International Conference on Automatic Face & Gesture Recognition, Lille, France, 2019. DOI: https:\/\/doi.org\/10.1109\/FG.2019.8756513."},{"key":"1629_CR12","doi-asserted-by":"publisher","first-page":"1148","DOI":"10.1109\/ICPR.2006.39","volume-title":"Proceedings of the 18th International Conference on Pattern Recognition","author":"H Gunes","year":"2006","unstructured":"H. Gunes, M. Piccardi. A bimodal face and body gesture database for automatic analysis of human nonverbal affective behavior. In Proceedings of the 18th International Conference on Pattern Recognition, Hong Kong, China, pp. 1148\u20131153, 2006. DOI: https:\/\/doi.org\/10.1109\/ICPR.2006.39."},{"key":"1629_CR13","volume-title":"Definitive Book of Body Language: How to Read Others\u2019 Attitudes by Their Gestures","author":"A Pease","year":"2004","unstructured":"A. Pease, B. Pease. Definitive Book of Body Language: How to Read Others\u2019 Attitudes by Their Gestures. Pease International, Australia, 2004."},{"key":"1629_CR14","doi-asserted-by":"publisher","first-page":"10626","DOI":"10.1109\/CVPR46437.2021.01049","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Liu","year":"2021","unstructured":"X. Liu, H. Shi, H. Chen, Z. Yu, X. Li, G. Zhao. iMiGUE: An identity-free video dataset for micro-gesture understanding and emotion analysis. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Nashville, USA, pp. 10626\u201310637, 2021. DOI: https:\/\/doi.org\/10.1109\/CVPR46437.2021.01049."},{"key":"1629_CR15","volume-title":"Identity-free artificial emotional intelligence via micro-gesture understanding","author":"R Gao","year":"2024","unstructured":"R. Gao, X. Liu, B. Xing, Z. Yu, B. W. Schuller, H. K\u00e4lvi\u00e4inen. Identity-free artificial emotional intelligence via micro-gesture understanding, [Online], Available: https:\/\/arxiv.org\/abs\/2405.13206, 2024."},{"key":"1629_CR16","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2009.5349544","volume-title":"Proceedings of the 3rd International Conference on Affective Computing and Intelligent Interaction and Workshops","author":"M Kipp","year":"2009","unstructured":"M. Kipp, J. C. Martin. Gesture and emotion: Can basic gestural form features discriminate emotions? In Proceedings of the 3rd International Conference on Affective Computing and Intelligent Interaction and Workshops, Amsterdam, Netherlands, 2009. DOI: https:\/\/doi.org\/10.1109\/ACII.2009.5349544."},{"key":"1629_CR17","volume-title":"Telling Lies: Clues to Deceit in the Marketplace, Politics, and Marriage","author":"P Ekman","year":"2009","unstructured":"P. Ekman. Telling Lies: Clues to Deceit in the Marketplace, Politics, and Marriage, New York, USA: W. W. Norton & Company, 2009."},{"key":"1629_CR18","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1007\/3-540-46616-9_10","volume-title":"Proceedings of International Gesture Workshop on Gesture-Based Communication in Human-Computer Interaction","author":"Y Wu","year":"1999","unstructured":"Y. Wu, T. S. Huang. Vision-based gesture recognition: A review. In Proceedings of International Gesture Workshop on Gesture-Based Communication in Human-Computer Interaction, Gif-sur-Yvette, France, pp. 103\u2013115, 1999. DOI: https:\/\/doi.org\/10.1007\/3-540-46616-9_10."},{"issue":"3","key":"1629_CR19","doi-asserted-by":"publisher","first-page":"311","DOI":"10.1109\/TSMCC.2007.893280","volume":"37","author":"S Mitra","year":"2007","unstructured":"S. Mitra, T. Acharya. Gesture recognition: A survey. IEEE Transactions on Systems, Man, and Cybernetics, Part C (Applications and Reviews), vol. 37, no. 3, pp. 311\u2013324, 2007. DOI: https:\/\/doi.org\/10.1109\/TSMCC.2007.893280.","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics, Part C (Applications and Reviews)"},{"key":"1629_CR20","volume-title":"Body Language: How to Read Others\u2019 Thoughts by Their Gestures","author":"A Pease","year":"1984","unstructured":"A. Pease. Body Language: How to Read Others\u2019 Thoughts by Their Gestures, London, UK: Sheldon Press, 1984."},{"issue":"1","key":"1629_CR21","doi-asserted-by":"publisher","first-page":"185","DOI":"10.1075\/pc.8.1.09lis","volume":"8","author":"C L Lisetti","year":"2000","unstructured":"C. L. Lisetti, D. J. Schiano. Automatic facial expression interpretation: Where human-computer interaction, artificial intelligence and cognitive science intersect. Pragmatics & Cognition, vol. 8, no. 1, pp. 185\u2013235, 2000. DOI: https:\/\/doi.org\/10.1075\/pc.8.1.09lis.","journal-title":"Pragmatics & Cognition"},{"issue":"4","key":"1629_CR22","doi-asserted-by":"publisher","first-page":"161","DOI":"10.5121\/ijaia.2012.3412","volume":"3","author":"R Z Khan","year":"2012","unstructured":"R. Z. Khan, N. A. Ibraheem. Hand gesture recognition: A literature review. International Journal of Artificial Intelligence & Applications, vol. 3, no. 4, pp. 161\u2013174, 2012. DOI: https:\/\/doi.org\/10.5121\/ijaia.2012.3412.","journal-title":"International Journal of Artificial Intelligence & Applications"},{"issue":"10","key":"1629_CR23","doi-asserted-by":"publisher","first-page":"2684","DOI":"10.1109\/TPAMI.2019.2916873","volume":"42","author":"J Liu","year":"2020","unstructured":"J. Liu, A. Shahroudy, M. Perez, G. Wang, L. Y. Duan, A. C. Kot. NTU RGB+D 120: A large-scale benchmark for 3D human activity understanding. IEEE Transactions on Pattern Analysis and Machine Intelligence, vol. 42, no. 10, pp. 2684\u20132701, 2020. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2019.2916873.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1629_CR24","doi-asserted-by":"publisher","first-page":"1010","DOI":"10.1109\/CVPR.2016.115","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"A Shahroudy","year":"2016","unstructured":"A. Shahroudy, J. Liu, T. T. Ng, G. Wang. NTU RGB+D: A large scale dataset for 3D human activity analysis. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, USA, pp. 1010\u20131019, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.115."},{"issue":"5","key":"1629_CR25","doi-asserted-by":"publisher","first-page":"1038","DOI":"10.1109\/TMM.2018.2808769","volume":"20","author":"Y Zhang","year":"2018","unstructured":"Y. Zhang, C. Cao, J. Cheng, H. Lu. EgoGesture: A new dataset and benchmark for egocentric hand gesture recognition. IEEE Transactions on Multimedia, vol. 20, no. 5, pp. 1038\u20131050, 2018. DOI: https:\/\/doi.org\/10.1109\/TMM.2018.2808769.","journal-title":"IEEE Transactions on Multimedia"},{"key":"1629_CR26","doi-asserted-by":"publisher","first-page":"445","DOI":"10.1145\/2522848.2532595","volume-title":"Proceedings of the 15th ACM on International Conference on Multimodal Interaction","author":"S Escalera","year":"2013","unstructured":"S. Escalera, J. Gonz\u00e0lez, X. Bar\u00d3, M. Reyes, O. Lopes, I. Guyon, V. Athitsos, H. Escalante. Multi-modal gesture recognition challenge 2013: Dataset and results. In Proceedings of the 15th ACM on International Conference on Multimodal Interaction, Sydney, Australia, pp. 445\u2013452, 2013. DOI: https:\/\/doi.org\/10.1145\/2522848.2532595."},{"key":"1629_CR27","doi-asserted-by":"publisher","first-page":"4560","DOI":"10.1109\/WACV57701.2024.00451","volume-title":"Proceedings of IEEE\/CVF Winter Conference on Applications of Computer Vision","author":"K Alexander","year":"2024","unstructured":"K. Alexander, K. Karina, N. Alexander, K. Roman, M. Andrei. HaGRID-HAnd gesture recognition image dataset. In Proceedings of IEEE\/CVF Winter Conference on Applications of Computer Vision, Waikoloa, USA, pp. 4560\u20134569, 2024. DOI: https:\/\/doi.org\/10.1109\/WACV57701.2024.00451."},{"issue":"2","key":"1629_CR28","doi-asserted-by":"publisher","first-page":"149","DOI":"10.1002\/wcs.1335","volume":"6","author":"B de Gelder","year":"2015","unstructured":"B. de Gelder, A. W. de Borst, R. Watson. The perception of emotion in body expressions. WIREs Cognitive Science, vol. 6, no. 2, pp. 149\u2013158, 2015. DOI: https:\/\/doi.org\/10.1002\/wcs.1335.","journal-title":"WIREs Cognitive Science"},{"key":"1629_CR29","volume-title":"What Every Body is Saying: An Ex-FBI Agent\u2019s Guide to Speed-Reading","author":"J Navarro","year":"2008","unstructured":"J. Navarro, M. Karlins. What Every Body is Saying: An Ex-FBI Agent\u2019s Guide to Speed-Reading, New York, USA: William Morrow Paperbacks, 2008."},{"issue":"7","key":"1629_CR30","doi-asserted-by":"publisher","first-page":"6238","DOI":"10.1109\/TCSVT.2024.3358415","volume":"34","author":"D Guo","year":"2024","unstructured":"D. Guo, K. Li, B. Hu, Y. Zhang, M. Wang. Benchmarking micro-action recognition: Dataset, methods, and applications. IEEE Transactions on Circuits and Systems for Video Technology, vol. 34, no. 7, pp. 6238\u20136252, 2024. DOI: https:\/\/doi.org\/10.1109\/TCSVT.2024.3358415.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"issue":"2","key":"1629_CR31","doi-asserted-by":"publisher","first-page":"124","DOI":"10.1002\/jip.1495","volume":"15","author":"N Palena","year":"2018","unstructured":"N. Palena, L. Caso, A. Vrij, R. Orthey. Detecting deception through small talk and comparable truth baselines. Journal of Investigative Psychology and Offender Profiling, vol. 15, no. 2, pp. 124\u2013132, 2018. DOI: https:\/\/doi.org\/10.1002\/jip.1495.","journal-title":"Journal of Investigative Psychology and Offender Profiling"},{"issue":"4","key":"1629_CR32","doi-asserted-by":"publisher","first-page":"1027","DOI":"10.1109\/TSMCB.2010.2103557","volume":"41","author":"A Kleinsmith","year":"2011","unstructured":"A. Kleinsmith, N. Bianchi-Berthouze, A. Steed. Automatic recognition of non-acted affective postures. IEEE Transactions on Systems, Man, and Cybernetics, Part B (Cybernetics), vol. 41, no. 4, pp. 1027\u20131038, 2011. DOI: https:\/\/doi.org\/10.1109\/TSMCB.2010.2103557.","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics, Part B (Cybernetics)"},{"issue":"1","key":"1629_CR33","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11263-019-01215-y","volume":"128","author":"Y Luo","year":"2020","unstructured":"Y. Luo, J. Ye, R. B. Adams Jr, J. Li, M. G. Newman, J. Z. Wang. ARBEE: Towards automated recognition of bodily expression of emotion in the wild. International Journal of Computer Vision, vol. 128, no. 1, pp. 1\u201325, 2020. DOI: https:\/\/doi.org\/10.1007\/s11263-019-01215-y.","journal-title":"International Journal of Computer Vision"},{"issue":"2","key":"1629_CR34","doi-asserted-by":"publisher","first-page":"505","DOI":"10.1109\/TAFFC.2018.2874986","volume":"12","author":"F Noroozi","year":"2021","unstructured":"F. Noroozi, C. A. Corneanu, D. Kaminska, T. Sapinski, S. Escalera, G. Anbarjafari. Survey on emotional body gesture recognition. IEEE Transactions on Affective Computing, vol. 12, no. 2, pp. 505\u2013523, 2021. DOI: https:\/\/doi.org\/10.1109\/TAFFC.2018.2874986.","journal-title":"IEEE Transactions on Affective Computing"},{"issue":"6","key":"1629_CR35","doi-asserted-by":"publisher","first-page":"879","DOI":"10.1002\/(SICI)1099-0992(1998110)28:6<879::AID-EJSP901>3.0.CO;2-W","volume":"28","author":"H G Wallbott","year":"1998","unstructured":"H. G. Wallbott. Bodily expression of emotion. European Journal of Social Psychology, vol. 28, no. 6, pp. 879\u2013896, 1998. DOI: https:\/\/doi.org\/10.1002\/(SICI)1099-0992(1998110)28:6<879::AID-EJSP901>3.0.CO;2-W.","journal-title":"European Journal of Social Psychology"},{"issue":"45","key":"1629_CR36","doi-asserted-by":"publisher","first-page":"16518","DOI":"10.1073\/pnas.0507650102","volume":"102","author":"H K M Meeren","year":"2005","unstructured":"H. K. M. Meeren, C. C. R. J. van Heijnsbergen, B. de Gelder. Rapid perceptual integration of facial expression and emotional body language. Proceedings of the National Academy of Sciences of the United States of America, vol. 102, no. 45, pp. 16518\u201316523, 2005. DOI: https:\/\/doi.org\/10.1073\/pnas.0507650102.","journal-title":"Proceedings of the National Academy of Sciences of the United States of America"},{"issue":"3","key":"1629_CR37","doi-asserted-by":"publisher","first-page":"242","DOI":"10.1038\/nrn1872","volume":"7","author":"B de Gelder","year":"2006","unstructured":"B. de Gelder. Towards the neurobiology of emotional body language. Nature Reviews Neuroscience, vol. 7, no. 3, pp. 242\u2013249, 2006. DOI: https:\/\/doi.org\/10.1038\/nrn1872.","journal-title":"Nature Reviews Neuroscience"},{"key":"1629_CR38","first-page":"3486","volume-title":"Proceedings of the 9th International Conference on Language Resources and Evaluation","author":"N Fourati","year":"2014","unstructured":"N. Fourati, C. Pelachaud. Emilya: Emotional body expression in daily actions database. In Proceedings of the 9th International Conference on Language Resources and Evaluation, Reykjavik, Iceland, pp. 3486\u20133493, 2014."},{"key":"1629_CR39","doi-asserted-by":"publisher","first-page":"720","DOI":"10.1109\/TELFOR.2015.7377568","volume-title":"Proceedings of the 23rd Telecommunications Forum Telfor","author":"M Gavrilescu","year":"2015","unstructured":"M. Gavrilescu. Recognizing emotions from videos by studying facial expressions, body postures and hand gestures. In Proceedings of the 23rd Telecommunications Forum Telfor, Belgrade, Serbia, pp. 720\u2013723, 2015. DOI: https:\/\/doi.org\/10.1109\/TELFOR.2015.7377568."},{"key":"1629_CR40","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2016.7477679","volume-title":"Proceedings of IEEE Winter Conference on Applications of Computer Vision","author":"H Ranganathan","year":"2016","unstructured":"H. Ranganathan, S. Chakraborty, S. Panchanathan. Multimodal emotion recognition using deep learning architectures. In Proceedings of IEEE Winter Conference on Applications of Computer Vision, Lake Placid, USA 2016. DOI: https:\/\/doi.org\/10.1109\/WACV.2016.7477679."},{"issue":"2","key":"1629_CR41","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1007\/BF00995674","volume":"15","author":"K R Scherer","year":"1991","unstructured":"K. R. Scherer, R. Banse, H. G. Wallbott, T. Goldbeck. Vocal cues in emotion encoding and decoding. Motivation and Emotion, vol. 15, no. 2, pp. 123\u2013148, 1991. DOI: https:\/\/doi.org\/10.1007\/BF00995674.","journal-title":"Motivation and Emotion"},{"key":"1629_CR42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-15184-2","volume-title":"Emotion-Oriented Systems: The Humaine Handbook","author":"R Cowie","year":"2011","unstructured":"R. Cowie, C. Pelachaud, P. Petta. Emotion-Oriented Systems: The Humaine Handbook, Berlin, Germany: Springer, 2011. DOI: https:\/\/doi.org\/10.1007\/978-3-642-15184-2."},{"key":"1629_CR43","volume-title":"Honest Signals: How They Shape Our World","author":"A Pentland","year":"2010","unstructured":"A. Pentland. Honest Signals: How They Shape Our World, Cambridge, UK: The MIT Press, 2010."},{"key":"1629_CR44","doi-asserted-by":"publisher","first-page":"247","DOI":"10.1017\/CBO9780511816802.016","volume-title":"The Cambridge Handbook of Metaphor and Thought","author":"N Yu","year":"2008","unstructured":"N. Yu. Metaphor from body and culture. The Cambridge Handbook of Metaphor and Thought, R. W. Jr. Gibbs, Ed., Cambridge, UK: Cambridge University Press, pp. 247\u2013261, 2008. DOI: https:\/\/doi.org\/10.1017\/CBO9780511816802.016."},{"key":"1629_CR45","volume-title":"International glossary of gestalt psychotherapy","author":"G Serge","year":"1995","unstructured":"G. Serge. International glossary of gestalt psychotherapy. FORGE, 1995, [Online], Available, https:\/\/www.behavenet.com\/ego-gestalt-glossary."},{"key":"1629_CR46","volume-title":"Body Language for Dummies","author":"E Kuhnke","year":"2012","unstructured":"E. Kuhnke. Body Language for Dummies, Hoboken, USA: Wiley, 2012."},{"issue":"1","key":"1629_CR47","doi-asserted-by":"publisher","first-page":"205","DOI":"10.1196\/annals.1280.010","volume":"1000","author":"P Ekman","year":"2003","unstructured":"P. Ekman. Darwin, deception, and facial expression. Annals of the New York Academy of Sciences, vol. 1000, no. 1, pp. 205\u2013221, 2003. DOI: https:\/\/doi.org\/10.1196\/annals.1280.010.","journal-title":"Annals of the New York Academy of Sciences"},{"issue":"3","key":"1629_CR48","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1037\/rev0000059","volume":"124","author":"S Kita","year":"2017","unstructured":"S. Kita, M. W. Alibali, M. Chu. How do gestures influence thinking and speaking? The gesture-for-conceptualization hypothesis. Psychological Review, vol. 124, no. 3, pp. 245\u2013266, 2017. DOI: https:\/\/doi.org\/10.1037\/rev0000059.","journal-title":"Psychological Review"},{"issue":"6","key":"1629_CR49","doi-asserted-by":"publisher","first-page":"1346","DOI":"10.1007\/s11263-023-01761-6","volume":"131","author":"H Chen","year":"2023","unstructured":"H. Chen, H. Shi, X. Liu, X. Li, G. Zhao. SMG: A micro-gesture dataset towards spontaneous body gestures for emotional stress state analysis. International Journal of Computer Vision, vol. 131, no. 6, pp. 1346\u20131366, 2023. DOI: https:\/\/doi.org\/10.1007\/s11263-023-01761-6.","journal-title":"International Journal of Computer Vision"},{"key":"1629_CR50","doi-asserted-by":"publisher","first-page":"5707","DOI":"10.1145\/3746027.3755411","volume-title":"Proceedings of the 33rd ACM International Conference on Multimedia","author":"D Li","year":"2025","unstructured":"D. Li, B. Xing, X. Liu, B. Xia, B. Wen, H. K\u00e4lvi\u00e4inen. DEEMO: De-identity multimodal emotion recognition and reasoning. In Proceedings of the 33rd ACM International Conference on Multimedia, Dublin, Ireland, pp. 5707\u20135716, 2025. DOI: https:\/\/doi.org\/10.1145\/3746027.3755411."},{"key":"1629_CR51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2008.4563173","volume-title":"Proceedings of IEEE Computer Society Conference on Computer Vision and Pattern Recognition Workshops","author":"D Glowinski","year":"2008","unstructured":"D. Glowinski, A. Camurri, G. Volpe, N. Dael, K. Scherer. Technique for automatic emotion recognition by body gesture analysis. In Proceedings of IEEE Computer Society Conference on Computer Vision and Pattern Recognition Workshops, Anchorage, USA, 2008. DOI: https:\/\/doi.org\/10.1109\/CVPRW.2008.4563173."},{"key":"1629_CR52","doi-asserted-by":"publisher","first-page":"488","DOI":"10.1007\/978-3-540-74889-2_43","volume-title":"Proceedings of the 2nd International Conference on Affective Computing and Intelligent Interaction","author":"E Douglas-Cowie","year":"2007","unstructured":"E. Douglas-Cowie, R. Cowie, I. Sneddon, C. Cox, O. Lowry, M. McRorie, J. C. Martin, L. Devillers, S. Abrilian, A. Batliner, N. Amir, K. Karpouzis. The HUMAINE database: Addressing the collection and annotation of naturalistic and induced emotional data. In Proceedings of the 2nd International Conference on Affective Computing and Intelligent Interaction, Lisbon, Portugal, pp. 488\u2013500, 2007. DOI: https:\/\/doi.org\/10.1007\/978-3-540-74889-2_43."},{"key":"1629_CR53","volume-title":"EALD-MLLM: Emotion analysis in long-sequential and de-identity videos with multi-modal large language model","author":"D Li","year":"2025","unstructured":"D. Li, X. Liu, B. Xing, B. Xia, Y. Zong, B. Wen, H. K\u00e4lvi\u00e4inen. EALD-MLLM: Emotion analysis in long-sequential and de-identity videos with multi-modal large language model, [Online], Available: https:\/\/arxiv.org\/abs\/2405.00574, 2025."},{"key":"1629_CR54","series-title":"Technical Report IDIAP\u2013RR","volume-title":"On the Use of Speech and Face Information for Identity Verification","author":"C Sanderson","year":"2004","unstructured":"C. Sanderson, K. K. Paliwal. On the Use of Speech and Face Information for Identity Verification, Technical Report IDIAP\u2013RR 04-10, Martigny, Switzerland, 2004."},{"key":"1629_CR55","volume-title":"GPT-4o system card","author":"OpenAI","year":"2024","unstructured":"OpenAI. GPT-4o system card, [Online], Available: https:\/\/arxiv.org\/abs\/2410.21276, 2024."},{"key":"1629_CR56","volume-title":"Qwen2.5-VL technical report","author":"Qwen Team, Alibaba Group","year":"2025","unstructured":"Qwen Team, Alibaba Group. Qwen2.5-VL technical report, [Online], Available: https:\/\/arxiv.org\/abs\/2502.13923, 2025."},{"key":"1629_CR57","doi-asserted-by":"publisher","first-page":"203","DOI":"10.1007\/978-3-319-46478-7_13","volume-title":"Proceedings of the 14th European Conference on Computer Vision","author":"Y Li","year":"2016","unstructured":"Y. Li, C. Lan, J. Xing, W. Zeng, C. Yuan, J. Liu. Online human action detection using joint classification-regression recurrent neural networks. In Proceedings of the 14th European Conference on Computer Vision, Amsterdam, The Netherlands, pp. 203\u2013220, 2016. DOI: https:\/\/doi.org\/10.1007\/978-3-319-46478-7_13."},{"key":"1629_CR58","doi-asserted-by":"publisher","first-page":"1302","DOI":"10.1109\/CVPR.2017.143","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"Z Cao","year":"2017","unstructured":"Z. Cao, T. Simon, S. E. Wei, Y. Sheikh. Realtime multi-person 2D pose estimation using part affinity fields. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Honolulu, USA, pp. 1302\u20131310, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.143."},{"key":"1629_CR59","doi-asserted-by":"publisher","first-page":"1297","DOI":"10.1109\/CVPR.2011.5995316","volume-title":"Proceedings of CVPR","author":"J Shotton","year":"2011","unstructured":"J. Shotton, A. Fitzgibbon, M. Cook, T. Sharp, M. Finocchio, R. Moore, A. Kipman, A. Blake. Real-time human pose recognition in parts from single depth images. In Proceedings of CVPR, Colorado Springs, USA, pp. 1297\u20131304, 2011. DOI: https:\/\/doi.org\/10.1109\/CVPR.2011.5995316."},{"key":"1629_CR60","doi-asserted-by":"publisher","first-page":"7444","DOI":"10.1609\/aaai.v32i1.12328","volume-title":"Proceedings of the 32nd AAAI Conference on Artificial Intelligence","author":"S Yan","year":"2018","unstructured":"S. Yan, Y. Xiong, D. Lin. Spatial temporal graph convolutional networks for skeleton-based action recognition. In Proceedings of the 32nd AAAI Conference on Artificial Intelligence, New Orleans, USA, pp. 7444\u20137452, 2018. DOI: https:\/\/doi.org\/10.1609\/aaai.v32i1.12328."},{"issue":"11","key":"1629_CR61","doi-asserted-by":"publisher","first-page":"2740","DOI":"10.1109\/TPAMI.2018.2868668","volume":"41","author":"L Wang","year":"2019","unstructured":"L. Wang, Y. Xiong, Z. Wang, Y. Qiao, D. Lin, X. Tang, L. Van Gool. Temporal segment networks for action recognition in videos. IEEE Transactions on Pattern Analysis and Machine Intelligence, vol. 41, no. 11, pp. 2740\u20132755, 2019. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2018.2868668.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1629_CR62","volume-title":"Proceedings of British Machine Vision Conference","author":"H Shi","year":"2018","unstructured":"H. Shi, X. Liu, X. Hong, G. Zhao. Bidirectional long short-term memory variational autoencoder. In Proceedings of British Machine Vision Conference, Newcastle, UK, Article number 165, 2018."},{"key":"1629_CR63","doi-asserted-by":"publisher","first-page":"831","DOI":"10.1007\/978-3-030-01246-5_49","volume-title":"Proceedings of the 15th European Conference on Computer Vision","author":"B Zhou","year":"2018","unstructured":"B. Zhou, A. Andonian, A. Oliva, A. Torralba. Temporal relational reasoning in videos. In Proceedings of the 15th European Conference on Computer Vision, Munich, Germany, pp. 831\u2013846, 2018. DOI: https:\/\/doi.org\/10.1007\/978-3-030-01246-5_49."},{"key":"1629_CR64","doi-asserted-by":"publisher","first-page":"12018","DOI":"10.1109\/CVPR.2019.01230","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"L Shi","year":"2019","unstructured":"L. Shi, Y. Zhang, J. Cheng, H. Lu. Two-stream adaptive graph convolutional networks for skeleton-based action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, USA, pp. 12018\u201312027, 2019. DOI: https:\/\/doi.org\/10.1109\/CVPR.2019.01230."},{"key":"1629_CR65","doi-asserted-by":"publisher","first-page":"7082","DOI":"10.1109\/ICCV.2019.00718","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"J Lin","year":"2019","unstructured":"J. Lin, C. Gan, S. Han. TSM: Temporal shift module for efficient video understanding. In Proceedings of IEEE\/CVF International Conference on Computer Vision, Seoul, Republic of Korea, pp. 7082\u20137092, 2019. DOI: https:\/\/doi.org\/10.1109\/ICCV.2019.00718."},{"key":"1629_CR66","doi-asserted-by":"publisher","first-page":"9628","DOI":"10.1109\/CVPR42600.2020.00965","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K Su","year":"2020","unstructured":"K. Su, X. Liu, E. Shlizerman. PREDICT & Cluster: Unsupervised skeleton based action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, USA, pp. 9628\u20139637, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00965."},{"key":"1629_CR67","doi-asserted-by":"publisher","first-page":"2669","DOI":"10.1609\/aaai.v34i03.5652","volume-title":"Proceedings of the 34th AAAI Conference on Artificial Intelligence","author":"W Peng","year":"2020","unstructured":"W. Peng, X. Hong, H. Chen, G. Zhao. Learning graph convolutional network for skeleton-based human action recognition by neural searching. In Proceedings of the 34th AAAI Conference on Artificial Intelligence, New York, USA, pp. 2669\u20132676, 2020. DOI: https:\/\/doi.org\/10.1609\/aaai.v34i03.5652."},{"key":"1629_CR68","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1016\/j.ins.2021.04.023","volume":"569","author":"H Rao","year":"2021","unstructured":"H. Rao, S. Xu, X. Hu, J. Cheng, B. Hu. Augmented skeleton based contrastive action learning with momentum LSTM for unsupervised action recognition. Information Sciences, vol. 569, pp. 90\u2013109, 2021. DOI: https:\/\/doi.org\/10.1016\/j.ins.2021.04.023.","journal-title":"Information Sciences"},{"key":"1629_CR69","doi-asserted-by":"publisher","DOI":"10.1109\/VCIP56404.2022.10008837","volume-title":"Proceedings of IEEE International Conference on Visual Communications and Image Processing","author":"R Gao","year":"2022","unstructured":"R. Gao, X. Liu, J. Yang, H. Yue. CdCLR: Clip-driven contrastive learning for skeleton-based action recognition. In Proceedings of IEEE International Conference on Visual Communications and Image Processing, Suzhou, China, 2022. DOI: https:\/\/doi.org\/10.1109\/VCIP56404.2022.10008837."},{"key":"1629_CR70","doi-asserted-by":"publisher","first-page":"2686","DOI":"10.1109\/ICPR56361.2022.9956565","volume-title":"Proceedings of the 26th International Conference on Pattern Recognition","author":"A Shah","year":"2022","unstructured":"A. Shah, H. Chen, H. Shi, G. Zhao. Efficient densegraph convolutional network with inductive prior augmentations for unsupervised micro-gesture recognition. In Proceedings of the 26th International Conference on Pattern Recognition, Montreal, Canada, pp. 2686\u20132692, 2022. DOI: https:\/\/doi.org\/10.1109\/ICPR56361.2022.9956565."},{"key":"1629_CR71","volume-title":"Joint skeletal and semantic embedding loss for micro-gesture classification","author":"K Li","year":"2023","unstructured":"K. Li, D. Guo, G. Chen, X. Peng, M. Wang. Joint skeletal and semantic embedding loss for micro-gesture classification, [Online], Available: https:\/\/arxiv.org\/abs\/2307.10624, 2023."},{"key":"1629_CR72","volume-title":"Proceedings of the 32nd International Joint Conference on Artificial Intelligence","author":"H Huang","year":"2023","unstructured":"H. Huang, X. Guo, W. Peng, Z. Xia. Micro-gesture classification based on ensemble hypergraph-convolution transformer. In Proceedings of the 32nd International Joint Conference on Artificial Intelligence, Macau, China, 2023."},{"key":"1629_CR73","volume-title":"Proceedings of the 33rd International Joint Conference on Artificial Intelligence","author":"H Huang","year":"2024","unstructured":"H. Huang, Y. Wang, K. Linghu, Z. Xia. Multi-modal micro-gesture classification via multi-scale heterogeneous ensemble network. In Proceedings of the 33rd International Joint Conference on Artificial Intelligence, Jeju, Republic of Korea, 2024."},{"key":"1629_CR74","doi-asserted-by":"publisher","first-page":"1309","DOI":"10.1109\/LSP.2024.3396656","volume":"31","author":"D Li","year":"2024","unstructured":"D. Li, B. Xing, X. Liu. Enhancing micro gesture recognition for emotion understanding via context-aware visual-text contrastive learning. IEEE Signal Processing Letters, vol. 31, pp. 1309\u20131313, 2024. DOI: https:\/\/doi.org\/10.1109\/LSP.2024.3396656.","journal-title":"IEEE Signal Processing Letters"},{"key":"1629_CR75","volume-title":"Prototype learning for micro-gesture classification","author":"G Chen","year":"2024","unstructured":"G. Chen, F. Wang, K. Li, Z. Wu, H. Fan, Y. Yang, M. Wang, D. Guo. Prototype learning for micro-gesture classification, [Online], Available: https:\/\/arxiv.org\/abs\/2408.03097, 2024."},{"key":"1629_CR76","volume-title":"CLIP-MG: Guiding semantic attention with skeletal pose features and RGB data for micro-gesture recognition on the iMiGUE dataset","author":"S Patapati","year":"2025","unstructured":"S. Patapati, T. Srinivasan, A. Adiraju. CLIP-MG: Guiding semantic attention with skeletal pose features and RGB data for micro-gesture recognition on the iMiGUE dataset, [Online], Available: https:\/\/arxiv.org\/abs\/2506.16385, 2025."},{"key":"1629_CR77","volume-title":"Towards fine-grained emotion understanding via skeleton-based micro-gesture recognition","author":"H Xu","year":"2025","unstructured":"H. Xu, L. Cheng, Y. Wang, S. Tang, Z. Zhong. Towards fine-grained emotion understanding via skeleton-based micro-gesture recognition, [Online], Available: https:\/\/arxiv.org\/abs\/2506.12848, 2025."},{"key":"1629_CR78","doi-asserted-by":"publisher","first-page":"4489","DOI":"10.1109\/ICCV.2015.510","volume-title":"Proceedings of IEEE International Conference on Computer Vision","author":"D Tran","year":"2015","unstructured":"D. Tran, L. Bourdev, R. Fergus, L. Torresani, M. Paluri. Learning spatiotemporal features with 3D convolutional networks. In Proceedings of IEEE International Conference on Computer Vision, Santiago, Chile, pp. 4489\u20134497, 2015. DOI: https:\/\/doi.org\/10.1109\/ICCV.2015.510."},{"key":"1629_CR79","volume-title":"The kinetics human action video dataset","author":"W Kay","year":"2017","unstructured":"W. Kay, J. Carreira, K. Simonyan, B. Zhang, C. Hillier, S. Vijayanarasimhan, F. Viola, T. Green, T. Back, P. Natsev, M. Suleyman, A. Zisserman. The kinetics human action video dataset, [Online], Available: https:\/\/arxiv.org\/abs\/1705.06950, 2017."},{"key":"1629_CR80","doi-asserted-by":"publisher","first-page":"6546","DOI":"10.1109\/CVPR.2018.00685","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K Hara","year":"2018","unstructured":"K. Hara, H. Kataoka, Y. Satoh. Can spatiotemporal 3D CNNs retrace the history of 2D CNNs and ImageNet? In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Salt Lake City, USA, pp. 6546\u20136555, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00685."},{"key":"1629_CR81","doi-asserted-by":"publisher","first-page":"4724","DOI":"10.1109\/CVPR.2017.502","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"J Carreira","year":"2017","unstructured":"J. Carreira, A. Zisserman. Quo Vadis, action recognition? A new model and the kinetics dataset. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Honolulu, USA, pp. 4724\u20134733, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.502."},{"issue":"1","key":"1629_CR82","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2013","unstructured":"S. Ji, W. Xu, M. Yang, K. Yu. 3D convolutional neural networks for human action recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, vol. 35, no. 1, pp. 221\u2013231, 2013. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2012.59.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1629_CR83","doi-asserted-by":"publisher","first-page":"5626","DOI":"10.1109\/TIP.2021.3087348","volume":"30","author":"Z Yu","year":"2021","unstructured":"Z. Yu, B. Zhou, J. Wan, P. Wang, H. Chen, X. Liu, S. Li, G. Zhao. Searching multi-rate and multi-modal temporal enhanced networks for gesture recognition. IEEE Transactions on Image Processing, vol. 30, pp. 5626\u20135640, 2021. DOI: https:\/\/doi.org\/10.1109\/TIP.2021.3087348.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1629_CR84","doi-asserted-by":"publisher","first-page":"568","DOI":"10.5555\/2968826.2968890","volume-title":"Proceedings of the 28th International Conference on Neural Information Processing Systems","author":"K Simonyan","year":"2014","unstructured":"K. Simonyan, A. Zisserman. Two-stream convolutional networks for action recognition in videos. In Proceedings of the 28th International Conference on Neural Information Processing Systems, Montreal, Canada, pp. 568\u2013576, 2014. DOI: https:\/\/doi.org\/10.5555\/2968826.2968890."},{"key":"1629_CR85","doi-asserted-by":"publisher","first-page":"6450","DOI":"10.1109\/CVPR.2018.00675","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"D Tran","year":"2018","unstructured":"D. Tran, H. Wang, L. Torresani, J. Ray, Y. LeCun, M. Paluri. A closer look at spatiotemporal convolutions for action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Salt Lake City, USA, pp. 6450\u20136459, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00675."},{"key":"1629_CR86","doi-asserted-by":"publisher","first-page":"6201","DOI":"10.1109\/ICCV.2019.00630","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"C Feichtenhofer","year":"2019","unstructured":"C. Feichtenhofer, H. Fan, J. Malik, K. He. SlowFast networks for video recognition. In Proceedings of IEEE\/CVF International Conference on Computer Vision, Seoul, Republic of Korea, pp. 6201\u20136210, 2019. DOI: https:\/\/doi.org\/10.1109\/ICCV.2019.00630."},{"issue":"3","key":"1629_CR87","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"O. Russakovsky, J. Deng, H. Su, J. Krause, S. Satheesh, S. Ma, Z. Huang, A. Karpathy, A. Khosla, M. Bernstein, A. C. Berg, Li F. F. ImageNet large scale visual recognition challenge. International Journal of Computer Vision, vol. 115, no. 3, pp. 211\u2013252, 2015. DOI: https:\/\/doi.org\/10.1007\/s11263-015-0816-y.","journal-title":"International Journal of Computer Vision"},{"issue":"2","key":"1629_CR88","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1109\/MMUL.2012.24","volume":"19","author":"Z Zhang","year":"2012","unstructured":"Z. Zhang. Microsoft Kinect sensor and its effect. IEEE Multimedia, vol. 19, no. 2, pp. 4\u201310, 2012. DOI: https:\/\/doi.org\/10.1109\/MMUL.2012.24.","journal-title":"IEEE Multimedia"},{"key":"1629_CR89","doi-asserted-by":"publisher","first-page":"1227","DOI":"10.1109\/CVPR.2019.00132","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"C Si","year":"2019","unstructured":"C. Si, W. Chen, W. Wang, L. Wang, T. Tan. An attention enhanced graph convolutional LSTM network for skeleton-based action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, USA, pp. 1227\u20131236, 2019. DOI: https:\/\/doi.org\/10.1109\/CVPR.2019.00132."},{"issue":"9","key":"1629_CR90","doi-asserted-by":"publisher","first-page":"2330","DOI":"10.1109\/TMM.2018.2802648","volume":"20","author":"S Zhang","year":"2018","unstructured":"S. Zhang, Y. Yang, J. Xiao, X. Liu, Y. Yang, D. Xie, Y. Zhuang. Fusing geometric features for skeleton-based action recognition using multilayer LSTM networks. IEEE Transactions on Multimedia, vol. 20, no. 9, pp. 2330\u20132343, 2018. DOI: https:\/\/doi.org\/10.1109\/TMM.2018.2802648.","journal-title":"IEEE Transactions on Multimedia"},{"issue":"4","key":"1629_CR91","doi-asserted-by":"publisher","first-page":"1586","DOI":"10.1109\/TIP.2017.2785279","volume":"27","author":"J Liu","year":"2018","unstructured":"J. Liu, G. Wang, L. Y. Duan, K. Abdiyeva, A. C. Kot. Skeleton-based human action recognition with global context-aware attention LSTM networks. IEEE Transactions on Image Processing, vol. 27, no. 4, pp. 1586\u20131599, 2018. DOI: https:\/\/doi.org\/10.1109\/TIP.2017.2785279.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1629_CR92","doi-asserted-by":"publisher","first-page":"148","DOI":"10.1109\/WACV.2017.24","volume-title":"Proceedings of IEEE Winter Conference on Applications of Computer Vision","author":"S Zhang","year":"2017","unstructured":"S. Zhang, X. Liu, J. Xiao. On geometric features for skeleton-based action recognition using multilayer LSTM networks. In Proceedings of IEEE Winter Conference on Applications of Computer Vision, Santa Rosa, USA, pp. 148\u2013157, 2017. DOI: https:\/\/doi.org\/10.1109\/WACV.2017.24."},{"key":"1629_CR93","doi-asserted-by":"publisher","first-page":"4583","DOI":"10.1109\/TIP.2020.2974061","volume":"29","author":"X Liu","year":"2020","unstructured":"X. Liu, H. Shi, X. Hong, H. Chen, D. Tao, G. Zhao. 3D skeletal gesture recognition via hidden states exploration. IEEE Transactions on Image Processing, vol. 29, pp. 4583\u20134597, 2020. DOI: https:\/\/doi.org\/10.1109\/TIP.2020.2974061.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1629_CR94","doi-asserted-by":"publisher","first-page":"180","DOI":"10.1109\/CVPR42600.2020.00026","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"K Cheng","year":"2020","unstructured":"K. Cheng, Y. Zhang, X. He, W. Chen, J. Cheng, H. Lu. Skeleton-based action recognition with shift graph convolutional network. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, USA, pp. 180\u2013189, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00026."},{"key":"1629_CR95","doi-asserted-by":"publisher","first-page":"2272","DOI":"10.1109\/ICCV.2019.00236","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"Y Cai","year":"2019","unstructured":"Y. Cai, L. Ge, J. Liu, J. Cai, T. J. Cham, J. Yuan, N. M. Thalmann. Exploiting spatial-temporal relationships for 3D pose estimation via graph convolutional networks. In Proceedings of IEEE\/CVF International Conference on Computer Vision, Seoul, Republic of Korea, pp. 2272\u20132281, 2019. DOI: https:\/\/doi.org\/10.1109\/ICCV.2019.00236."},{"issue":"4","key":"1629_CR96","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1109\/MIS.2022.3147585","volume":"37","author":"H Shi","year":"2022","unstructured":"H. Shi, W. Peng, H. Chen, X. Liu, G. Zhao. Multiscale 3D-shift graph convolution network for emotion recognition from human actions. IEEE Intelligent Systems, vol. 37, no. 4, pp. 103\u2013110, 2022. DOI: https:\/\/doi.org\/10.1109\/MIS.2022.3147585.","journal-title":"IEEE Intelligent Systems"},{"key":"1629_CR97","doi-asserted-by":"publisher","first-page":"140","DOI":"10.1109\/CVPR42600.2020.00022","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Liu","year":"2020","unstructured":"Z. Liu, H. Zhang, Z. Chen, Z. Wang, W. Ouyang. Disentangling and unifying graph convolutions for skeleton-based action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, USA, pp. 140\u2013149, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00022."},{"key":"1629_CR98","doi-asserted-by":"publisher","first-page":"244","DOI":"10.1109\/LSP.2021.3049691","volume":"28","author":"W Peng","year":"2021","unstructured":"W. Peng, J. Shi, G. Zhao. Spatial temporal graph deconvolutional network for skeleton-based human action recognition. IEEE Signal Processing Letters, vol. 28, pp. 244\u2013248, 2021. DOI: https:\/\/doi.org\/10.1109\/LSP.2021.3049691.","journal-title":"IEEE Signal Processing Letters"},{"key":"1629_CR99","doi-asserted-by":"publisher","first-page":"761","DOI":"10.1007\/978-3-030-58565-5_45","volume-title":"Proceedings of the 16th European Conference on Computer Vision","author":"M Korban","year":"2020","unstructured":"M. Korban, X. Li. DDGCN: A dynamic directed graph convolutional network for action recognition. In Proceedings of the 16th European Conference on Computer Vision, Glasgow, UK, pp. 761\u2013776, 2020. DOI: https:\/\/doi.org\/10.1007\/978-3-030-58565-5_45."},{"key":"1629_CR100","doi-asserted-by":"publisher","first-page":"13339","DOI":"10.1109\/ICCV48922.2021.01311","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"Y Chen","year":"2021","unstructured":"Y. Chen, Z. Zhang, C. Yuan, B. Li, Y. Deng, W. Hu. Channel-wise topology refinement graph convolution for skeleton-based action recognition. In Proceedings of IEEE\/CVF International Conference on Computer Vision, Montreal, Canada, pp. 13339\u201313348, 2021. DOI: https:\/\/doi.org\/10.1109\/ICCV48922.2021.01311."},{"key":"1629_CR101","doi-asserted-by":"publisher","first-page":"2049","DOI":"10.1109\/CVPR52733.2024.00200","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Zhou","year":"2024","unstructured":"Y. Zhou, X. Yan, Z. Q. Cheng, Y. Yan, Q. Dai, X. S. Hua. BlockGCN: Redefine topology awareness for skeleton-based action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, USA, pp. 2049\u20132058, 2024. DOI: https:\/\/doi.org\/10.1109\/CVPR52733.2024.00200."},{"key":"1629_CR102","doi-asserted-by":"publisher","first-page":"8527","DOI":"10.1109\/TMM.2023.3318325","volume":"25","author":"H Tian","year":"2023","unstructured":"H. Tian, X. Ma, X. Li, Y. Li. Skeleton-based action recognition with select-assemble-normalize graph convolutional networks. IEEE Transactions on Multimedia, vol. 25, pp. 8527\u20138538, 2023. DOI: https:\/\/doi.org\/10.1109\/TMM.2023.3318325.","journal-title":"IEEE Transactions on Multimedia"},{"key":"1629_CR103","doi-asserted-by":"publisher","first-page":"10410","DOI":"10.1109\/ICCV51070.2023.00958","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"J Lee","year":"2023","unstructured":"J. Lee, M. Lee, D. Lee, S. Lee. Hierarchically decomposed graph convolutional networks for skeleton-based action recognition. In Proceedings of IEEE\/CVF International Conference on Computer Vision, Paris, France, pp. 10410\u201310419, 2023. DOI: https:\/\/doi.org\/10.1109\/ICCV51070.2023.00958."},{"key":"1629_CR104","doi-asserted-by":"publisher","first-page":"536","DOI":"10.1007\/978-3-030-58586-0_32","volume-title":"Proceedings of the 16th European Conference on Computer Vision","author":"K Cheng","year":"2020","unstructured":"K. Cheng, Y. Zhang, C. Cao, L. Shi, J. Cheng, H. Lu. Decoupling GCN with DropGraph module for skeleton-based action recognition. In Proceedings of the 16th European Conference on Computer Vision, Glasgow, UK, pp. 536\u2013553, 2020. DOI: https:\/\/doi.org\/10.1007\/978-3-030-58586-0_32."},{"key":"1629_CR105","volume-title":"Skeleton-based action recognition via temporal-channel aggregation","author":"S Wang","year":"2022","unstructured":"S. Wang, Y. Zhang, M. Zhao, H. Qi, K. Wang, F. Wei, Y. Jiang. Skeleton-based action recognition via temporal-channel aggregation, [Online], Available: https:\/\/arxiv.org\/abs\/2205.15936, 2022."},{"key":"1629_CR106","doi-asserted-by":"publisher","first-page":"441","DOI":"10.1109\/LSP.2024.3356808","volume":"31","author":"Y Zhang","year":"2024","unstructured":"Y. Zhang, Z. Sun, M. Dai, J. Feng, K. Jia. Cross-scale spatiotemporal refinement learning for skeleton-based action recognition. IEEE Signal Processing Letters, vol. 31, pp. 441\u2013445, 2024. DOI: https:\/\/doi.org\/10.1109\/LSP.2024.3356808.","journal-title":"IEEE Signal Processing Letters"},{"key":"1629_CR107","doi-asserted-by":"publisher","first-page":"2959","DOI":"10.1109\/CVPR52688.2022.00298","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"H Duan","year":"2022","unstructured":"H. Duan, Y. Zhao, K. Chen, D. Lin, B. Dai. Revisiting skeleton-based action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, USA, pp. 2959\u20132968, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.00298."},{"key":"1629_CR108","doi-asserted-by":"publisher","first-page":"10608","DOI":"10.1109\/CVPR52729.2023.01022","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"H Zhou","year":"2023","unstructured":"H. Zhou, Q. Liu, Y. Wang. Learning discriminative representations for skeleton based action recognition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Vancouver, Canada, pp. 10608\u201310617, 2023. DOI: https:\/\/doi.org\/10.1109\/CVPR52729.2023.01022."},{"key":"1629_CR109","doi-asserted-by":"publisher","first-page":"5971","DOI":"10.18653\/v1\/2024.emnlp-main.342","volume-title":"Proceedings of Conference on Empirical Methods in Natural Language Processing","author":"B Lin","year":"2024","unstructured":"B. Lin, Y. Ye, B. Zhu, J. Cui, M. Ning, P. Jin, L. Yuan. Video-LLaVA: Learning united visual representation by alignment before projection. In Proceedings of Conference on Empirical Methods in Natural Language Processing, Miami, USA, pp. 5971\u20135984, 2024. DOI: https:\/\/doi.org\/10.18653\/v1\/2024.emnlp-main.342."},{"key":"1629_CR110","doi-asserted-by":"publisher","DOI":"10.5555\/3666122.3667638","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"H Liu","year":"2023","unstructured":"H. Liu, C. Li, Q. Wu, Y. J. Lee. Visual instruction tuning. In Proceedings of the 37th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 1516, 2023. DOI: https:\/\/doi.org\/10.5555\/3666122.3667638."},{"key":"1629_CR111","doi-asserted-by":"publisher","first-page":"1447","DOI":"10.1609\/aaai.v39i2.32135","volume-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence","author":"H Lu","year":"2025","unstructured":"H. Lu, J. Chen, F. Liang, M. Tan, R. Zeng, X. Hu. Understanding emotional body expressions via large language models. In Proceedings of the 39th AAAI Conference on Artificial Intelligence, Philadelphia, USA, pp. 1447\u20131455, 2025. DOI: https:\/\/doi.org\/10.1609\/aaai.v39i2.32135."},{"key":"1629_CR112","doi-asserted-by":"publisher","first-page":"6931","DOI":"10.1609\/aaai.v39i7.32744","volume-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence","author":"A Sinha","year":"2025","unstructured":"A. Sinha, D. Reilly, F. Bremond, P. Wang, S. Das. SKI models: Skeleton induced vision-language embeddings for understanding activities of daily living. In Proceedings of the 39th AAAI Conference on Artificial Intelligence, Philadelphia, USA, pp. 6931\u20136939, 2025. DOI: https:\/\/doi.org\/10.1609\/aaai.v39i7.32744."},{"issue":"10","key":"1629_CR113","doi-asserted-by":"publisher","first-page":"2222","DOI":"10.1109\/TNNLS.2016.2582924","volume":"28","author":"K Greff","year":"2017","unstructured":"K. Greff, R. K. Srivastava, J. Koutn\u00edk, B. R. Steunebrink, J. Schmidhuber. LSTM: A search space odyssey. IEEE Transactions on Neural Networks and Learning Systems, vol. 28, no. 10, pp. 2222\u20132232, 2017. DOI: https:\/\/doi.org\/10.1109\/TNNLS.2016.2582924.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"1629_CR114","volume-title":"Hypergraph transformer for skeleton-based action recognition","author":"Y Zhou","year":"2022","unstructured":"Y. Zhou, Z. Q. Cheng, C. Li, Y. Fang, Y. Geng, X. Xie, M. Keuper. Hypergraph transformer for skeleton-based action recognition, [Online], Available: https:\/\/arxiv.org\/abs\/2211.09590, 2022."},{"key":"1629_CR115","volume-title":"MSF-mamba: Motion-aware state fusion mamba for efficient micro-gesture recognition","author":"D Li","year":"2025","unstructured":"D. Li, J. Shao, B. Xing, R. Gao, B. Wen, H. K\u00e4lvi\u00e4inen, X. Liu. MSF-mamba: Motion-aware state fusion mamba for efficient micro-gesture recognition, [Online], Available: https:\/\/arxiv.org\/abs\/2510.10478, 2025."},{"key":"1629_CR116","volume-title":"MM-gesture: Towards precise micro-gesture recognition through multimodal fusion","author":"J Gu","year":"2025","unstructured":"J. Gu, F. Wang, K. Li, Y. Wei, Z. Wu, D. Guo. MM-gesture: Towards precise micro-gesture recognition through multimodal fusion, [Online], Available: https:\/\/arxiv.org\/abs\/2507.08344, 2025."},{"key":"1629_CR117","doi-asserted-by":"publisher","unstructured":"S. Bai, F. Zhang, P. H. S. Torr. Hypergraph convolution and hypergraph attention. Pattern Recognition, vol. 110, Article number 107637, 2021. DOI: https:\/\/doi.org\/10.1016\/j.patcog.2020.107637.","DOI":"10.1016\/j.patcog.2020.107637"},{"key":"1629_CR118","doi-asserted-by":"publisher","DOI":"10.1109\/IJCB62174.2024.10744468","volume-title":"Proceedings of IEEE International Joint Conference on Biometrics","author":"C Liu","year":"2024","unstructured":"C. Liu, X. Liu, Z. Yu, Y. Hou, H. Yue, J. Yang. Adversarial robustness in RGB-skeleton action recognition: Leveraging attention modality reweighter. In Proceedings of IEEE International Joint Conference on Biometrics, Buffalo, USA, 2024. DOI: https:\/\/doi.org\/10.1109\/IJCB62174.2024.10744468."},{"key":"1629_CR119","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"A. Radford, J. W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, G. Krueger, I. Sutskever. Learning transferable visual models from natural language supervision. In Proceedings of the 38th International Conference on Machine Learning, pp. 8748\u20138763, 2021."},{"key":"1629_CR120","volume-title":"DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter","author":"V Sanh","year":"2020","unstructured":"V. Sanh, L. Debut, J. Chaumond, T. Wolf. DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter, [Online], Available: https:\/\/arxiv.org\/abs\/1910.01108, 2020."},{"key":"1629_CR121","doi-asserted-by":"publisher","first-page":"3192","DOI":"10.1109\/CVPR52688.2022.00320","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Liu","year":"2022","unstructured":"Z. Liu, J. Ning, Y. Cao, Y. Wei, Z. Zhang, S. Lin, H. Hu. Video swin transformer. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, USA, pp. 3192\u20133201, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.00320."}],"container-title":["Machine Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-025-1629-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11633-025-1629-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-025-1629-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T12:05:19Z","timestamp":1775649919000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11633-025-1629-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":121,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["1629"],"URL":"https:\/\/doi.org\/10.1007\/s11633-025-1629-x","relation":{},"ISSN":["2731-538X","2731-5398"],"issn-type":[{"value":"2731-538X","type":"print"},{"value":"2731-5398","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]},"assertion":[{"value":"1 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declared that they have no conflicts of interest to this work.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations of conflict of interest"}}]}}