{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:55:23Z","timestamp":1784300123107,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","funder":[{"name":"Anhui Provincial Key Research and Development Project","award":["202304a05020068"],"award-info":[{"award-number":["202304a05020068"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472381, U24A20331"],"award-info":[{"award-number":["62472381, U24A20331"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Postdoctoral Fellowship Program of CPSF","award":["GZC20251086"],"award-info":[{"award-number":["GZC20251086"]}]},{"name":"Fundamental Research Funds for the Central Universities of China","award":["PA2025IISL0109"],"award-info":[{"award-number":["PA2025IISL0109"]}]},{"name":"the Earth System Big Data Platform of the School of Earth Sciences, Zhejiang University"},{"name":"HPC Platform of Hefei University of Technology"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754722","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:55Z","timestamp":1761377215000},"page":"5461-5470","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Motion Matters: Motion-guided Modulation Network for Skeleton-based Micro-Action Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-0141-4807","authenticated-orcid":false,"given":"Jihao","family":"Gu","sequence":"first","affiliation":[{"name":"University College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5083-2145","authenticated-orcid":false,"given":"Kun","family":"Li","sequence":"additional","affiliation":[{"name":"ReLER, CCAI, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1142-6434","authenticated-orcid":false,"given":"Fei","family":"Wang","sequence":"additional","affiliation":[{"name":"Hefei University of Technology, Hefei, China and Institute of Artificial Intelligence, Hefei Comprehensive National Science Center, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8818-6740","authenticated-orcid":false,"given":"Yanyan","family":"Wei","sequence":"additional","affiliation":[{"name":"Hefei University of Technology, Hefei, China and Intelligent Interconnected Systems Laboratory of Anhui Province, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6597-8048","authenticated-orcid":false,"given":"Zhiliang","family":"Wu","sequence":"additional","affiliation":[{"name":"ReLER, CCAI, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9572-2345","authenticated-orcid":false,"given":"Hehe","family":"Fan","sequence":"additional","affiliation":[{"name":"ReLER, CCAI, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3094-7735","authenticated-orcid":false,"given":"Meng","family":"Wang","sequence":"additional","affiliation":[{"name":"Hefei University of Technology, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"1225","volume-title":"Science","volume":"338","author":"Aviezer Hillel","year":"2012","unstructured":"Hillel Aviezer, Yaacov Trope, and Alexander Todorov. 2012. Body cues, not facial expressions, discriminate between intense positive and negative emotions. Science, Vol. 338, 6111 (2012), 1225-1229."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548363"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of International Conference on Machine Learning. 813-824","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is space-time attention all you need for video understanding?. In Proceedings of International Conference on Machine Learning. 813-824."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_5_1","volume-title":"Prototype learning for micro-gesture classification. arXiv preprint arXiv:2408.03097","author":"Chen Guoliang","year":"2024","unstructured":"Guoliang Chen, Fei Wang, Kun Li, Zhiliang Wu, Hehe Fan, Yi Yang, Meng Wang, and Dan Guo. 2024b. Prototype learning for micro-gesture classification. arXiv preprint arXiv:2408.03097 (2024)."},{"key":"e_1_3_2_1_6_1","unstructured":"Haoyu Chen Bj\u00f6rn W Schuller Ehsan Adeli and Guoying Zhao. 2024a. The 2nd Challenge on Micro-gesture Analysis for Hidden Emotion Understanding (MiGA) 2024: Dataset and Results. In MiGA 2024: Proceedings of IJCAI 2024 Workshop&Challenge on Micro-gesture Analysis for Hidden Emotion Understanding (MiGA 2024) co-located with 33rd International Joint Conference on Artificial Intelligence (IJCAI 2024). RWTH Aachen."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01761-6"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01311"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i2.16197"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58586-0_32"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00026"},{"key":"e_1_3_2_1_12_1","volume-title":"European Conference on Computer Vision. 401-420","author":"Do Jeonghyeok","year":"2024","unstructured":"Jeonghyeok Do and Munchurl Kim. 2024. Skateformer: skeletal-temporal transformer for human action recognition. In European Conference on Computer Vision. 401-420."},{"key":"e_1_3_2_1_13_1","volume-title":"Dg-stgcn: Dynamic spatial-temporal modeling for skeleton-based action recognition. arXiv preprint arXiv:2210.05895","author":"Duan Haodong","year":"2022","unstructured":"Haodong Duan, Jiaqi Wang, Kai Chen, and Dahua Lin. 2022a. Dg-stgcn: Dynamic spatial-temporal modeling for skeleton-based action recognition. arXiv preprint arXiv:2210.05895 (2022)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548546"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00298"},{"key":"e_1_3_2_1_16_1","volume-title":"AlphaPose: Whole-Body Regional Multi-Person Pose Estimation and Tracking in Real-Time","author":"Fang Hao-Shu","year":"2022","unstructured":"Hao-Shu Fang, Jiefeng Li, Hongyang Tang, Chao Xu, Haoyi Zhu, Yuliang Xiu, Yong-Lu Li, and Cewu Lu. 2022. AlphaPose: Whole-Body Regional Multi-Person Pose Estimation and Tracking in Real-Time. IEEE Transactions on Pattern Analysis and Machine Intelligence (2022)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_1_18_1","volume-title":"MM-Gesture: Towards Precise Micro-Gesture Recognition through Multimodal Fusion. arXiv preprint arXiv:2507.08344","author":"Gu Jihao","year":"2025","unstructured":"Jihao Gu, Fei Wang, Kun Li, Zhiliang Wu, and Dan Guo. 2025. MM-Gesture: Towards Precise Micro-Gesture Recognition through Multimodal Fusion. arXiv preprint arXiv:2507.08344 (2025)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3358415"},{"key":"e_1_3_2_1_20_1","volume-title":"MAC 2024: Micro-Action Analysis Grand Challenge. In Proceedings of the 32nd ACM International Conference on Multimedia. 11304-11305","author":"Guo Dan","year":"2024","unstructured":"Dan Guo, Xiaobai Li, Kun Li, Haoyu Chen, Jingjing Hu, Guoying Zhao, Yi Yang, and Meng Wang. 2024b. MAC 2024: Micro-Action Analysis Grand Challenge. In Proceedings of the 32nd ACM International Conference on Multimedia. 11304-11305."},{"key":"e_1_3_2_1_21_1","volume-title":"Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415","author":"Hendrycks Dan","year":"2016","unstructured":"Dan Hendrycks and Kevin Gimpel. 2016. Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)."},{"key":"e_1_3_2_1_22_1","volume-title":"Ddgcn: A dynamic directed graph convolutional network for action recognition. In Computer Vision-ECCV 2020: 16th European Conference","author":"Korban Matthew","year":"2020","unstructured":"Matthew Korban and Xin Li. 2020. Ddgcn: A dynamic directed graph convolutional network for action recognition. In Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23-28, 2020, Proceedings, Part XX 16. Springer, 761-776."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00998"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00958"},{"key":"e_1_3_2_1_26_1","volume-title":"Eald-mllm: Emotion analysis in long-sequential and de-identity videos with multi-modal large language model. arXiv preprint arXiv:2405.00574","author":"Li Deng","year":"2024","unstructured":"Deng Li, Xin Liu, Bohao Xing, Baiqiang Xia, Yuan Zong, Bihan Wen, and Heikki K\u00e4lvi\u00e4inen. 2024c. Eald-mllm: Emotion analysis in long-sequential and de-identity videos with multi-modal large language model. arXiv preprint arXiv:2405.00574 (2024)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i5.32509"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612856"},{"key":"e_1_3_2_1_29_1","volume-title":"Joint skeletal and semantic embedding loss for micro-gesture classification. arXiv preprint arXiv:2307.10624","author":"Li Kun","year":"2023","unstructured":"Kun Li, Dan Guo, Guoliang Chen, Xinge Peng, and Meng Wang. 2023c. Joint skeletal and semantic embedding loss for micro-gesture classification. arXiv preprint arXiv:2307.10624 (2023)."},{"key":"e_1_3_2_1_30_1","volume-title":"Mmad: Multi-label micro-action detection in videos. arXiv preprint arXiv:2407.05311","author":"Li Kun","year":"2024","unstructured":"Kun Li, Dan Guo, Pengyu Liu, Guoliang Chen, and Meng Wang. 2024a. Mmad: Multi-label micro-action detection in videos. arXiv preprint arXiv:2407.05311 (2024)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-022-3783-3"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2025.3535385"},{"key":"e_1_3_2_1_33_1","volume-title":"UniFormer: Unified Transformer for Efficient Spatial-Temporal Representation Learning. In International Conference on Learning Representations.","author":"Li Kunchang","year":"2022","unstructured":"Kunchang Li, Yali Wang, Gao Peng, Guanglu Song, Yu Liu, Hongsheng Li, and Yu Qiao. 2022. UniFormer: Unified Transformer for Efficient Spatial-Temporal Representation Learning. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00371"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688975"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612496"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00572"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00718"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46487-9_50"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2017.2785279"},{"key":"e_1_3_2_1_41_1","volume-title":"Online Micro-gesture Recognition Using Data Augmentation and Spatial-Temporal Attention. arXiv preprint arXiv:2507.09512","author":"Liu Pengyu","year":"2025","unstructured":"Pengyu Liu, Kun Li, Fei Wang, Yanyan Wei, Junhui She, and Dan Guo. 2025. Online Micro-gesture Recognition Using Data Augmentation and Spatial-Temporal Attention. arXiv preprint arXiv:2507.09512 (2025)."},{"key":"e_1_3_2_1_42_1","volume-title":"Micro-gesture Online Recognition using Learnable Query Points. arXiv preprint arXiv:2407.04490","author":"Liu Pengyu","year":"2024","unstructured":"Pengyu Liu, Fei Wang, Kun Li, Guoliang Chen, Yanyan Wei, Shengeng Tang, Zhiliang Wu, and Dan Guo. 2024. Micro-gesture Online Recognition using Learnable Query Points. arXiv preprint arXiv:2407.04490 (2024)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01049"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-60639-8_40"},{"key":"e_1_3_2_1_46_1","volume-title":"Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3050642"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.10.084"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6872"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01230"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3028207"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00132"},{"key":"e_1_3_2_1_53_1","volume-title":"UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402","author":"Soomro K","year":"2012","unstructured":"K Soomro. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_55_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in Neural Information Processing Systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i6.28342"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01796"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701716.3717518"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00544"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2868668"},{"key":"e_1_3_2_1_62_1","volume-title":"Taylor videos for action recognition. arXiv preprint arXiv:2402.03019","author":"Wang Lei","year":"2024","unstructured":"Lei Wang, Xiuyuan Yuan, Tom Gedeon, and Liang Zheng. 2024c. Taylor videos for action recognition. arXiv preprint arXiv:2402.03019 (2024)."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01021"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_1_65_1","volume-title":"Action recognition with multi-stream motion modeling and mutual information maximization. arXiv preprint arXiv:2306.07576","author":"Yang Yuheng","year":"2023","unstructured":"Yuheng Yang, Haipeng Chen, Zhenguang Liu, Yingda Lyu, Beibei Zhang, Shuang Wu, Zhibo Wang, and Kui Ren. 2023. Action recognition with multi-stream motion modeling and mutual information maximization. arXiv preprint arXiv:2306.07576 (2023)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.233"},{"key":"e_1_3_2_1_67_1","volume-title":"Temporal-Frequency State Space Duality: An Efficient Paradigm for Speech Emotion Recognition. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing. 1-5.","author":"Zhao Jiaqi","year":"2025","unstructured":"Jiaqi Zhao, Fei Wang, Kun Li, Yanyan Wei, Shengeng Tang, Shu Zhao, and Xiao Sun. 2025. Temporal-Frequency State Space Duality: An Efficient Paradigm for Speech Emotion Recognition. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing. 1-5."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01022"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10451"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754722","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:06:46Z","timestamp":1765343206000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754722"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":69,"alternative-id":["10.1145\/3746027.3754722","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754722","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}