{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T00:39:35Z","timestamp":1783816775089,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3762095","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:54:15Z","timestamp":1761375255000},"page":"14222-14228","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Progressive Large-Scale Modeling via Temporal-Spatial Focus Connector for Micro-Action Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5121-1682","authenticated-orcid":false,"given":"Qiankun","family":"Li","sequence":"first","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1036-6762","authenticated-orcid":false,"given":"Qiupu","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Henan University, Zhengzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4300-3017","authenticated-orcid":false,"given":"Huabao","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5059-4423","authenticated-orcid":false,"given":"Feng","family":"He","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2281-7535","authenticated-orcid":false,"given":"Depeng","family":"Li","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4587-3588","authenticated-orcid":false,"given":"Zhigang","family":"Zeng","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688977"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the IEEE International Conference on Computer Vision (ICCV).","author":"Goyal Raghav","year":"2017","unstructured":"Raghav Goyal, Samira Ebrahimi Kahou, Vincent Michalski, Joanna Materzynska, Susanne Westphal, Heuna Kim, Valentin Haenel, Felix Fruend, Peter Yianilos, Moritz Mueller-Freitag, et al., 2017. The Something-Something Dataset for Learning and Evaluating Visual Common Sense. In Proceedings of the IEEE International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00633"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754722"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053928"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3358415"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_1_10_1","unstructured":"Will Kay Joao Carreira Andrew Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Pavel Natsev et al. 2017. The Kinetics Human Action Video Dataset. In arXiv preprint arXiv:1705.06950."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i5.32509"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612856"},{"key":"e_1_3_2_1_13_1","volume-title":"Joint skeletal and semantic embedding loss for micro-gesture classification. arXiv preprint arXiv:2307.10624","author":"Li Kun","year":"2023","unstructured":"Kun Li, Dan Guo, Guoliang Chen, Xinge Peng, and Meng Wang. 2023b. Joint skeletal and semantic embedding loss for micro-gesture classification. arXiv preprint arXiv:2307.10624 (2023)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2025.3535385"},{"key":"e_1_3_2_1_15_1","volume-title":"Uniformerv2: Spatiotemporal learning by arming image vits with video uniformer. arXiv preprint arXiv:2211.09552","author":"Li Kunchang","year":"2022","unstructured":"Kunchang Li, Yali Wang, Yinan He, Yizhuo Li, Yi Wang, Limin Wang, and Yu Qiao. 2022. Uniformerv2: Spatiotemporal learning by arming image vits with video uniformer. arXiv preprint arXiv:2211.09552 (2022)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01826"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688975"},{"key":"e_1_3_2_1_18_1","volume-title":"Video is graph: Structured graph module for video action recognition. arXiv preprint arXiv:2110.05904","author":"Li Rongchang","year":"2021","unstructured":"Rongchang Li, Xiao-Jun Wu, and Tianyang Xu. 2021. Video is graph: Structured graph module for video action recognition. arXiv preprint arXiv:2110.05904 (2021)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00718"},{"key":"e_1_3_2_1_20_1","volume-title":"Fineaction: A fine-grained video dataset for temporal action localization","author":"Liu Yi","year":"2022","unstructured":"Yi Liu, Limin Wang, Yali Wang, Xiao Ma, and Yu Qiao. 2022b. Fineaction: A fine-grained video dataset for temporal action localization. IEEE transactions on image processing, Vol. 31 (2022), 6937-6950."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_2_1_23_1","volume-title":"David Anastasiu, Shuo Wang, and Anuj Sharma.","author":"Rahman Mohammed Shaiqur","year":"2022","unstructured":"Mohammed Shaiqur Rahman, Jiyang Wang, Senem Velipasalar Gursoy, David Anastasiu, Shuo Wang, and Anuj Sharma. 2022. Synthetic Distracted Driving (SynDD2) dataset for analyzing distracted behaviors and various gaze zones of a driver. arXiv e-prints (2022), arXiv-2204."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00269"},{"key":"e_1_3_2_1_25_1","volume-title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems","author":"Tong Zhan","year":"2022","unstructured":"Zhan Tong, Yibing Song, Jue Wang, and Limin Wang. 2022. Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems, Vol. 35 (2022), 10078-10093."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688976"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01398"},{"key":"e_1_3_2_1_29_1","volume-title":"So Kweon, and Saining Xie.","author":"Woo Sanghyun","year":"2023","unstructured":"Sanghyun Woo, Shoubhik Debnath, Ronghang Hu, Xinlei Chen, Zhuang Liu, In So Kweon, and Saining Xie. 2023. Convnext v2: Co-designing and scaling convnets with masked autoencoders. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 16133-16142."},{"key":"e_1_3_2_1_30_1","volume-title":"DualActNet: Exploiting SlowFast Architecture for Micro-action Recognition. In Chinese Conference on Biometric Recognition. Springer, 59-68","author":"Yu Churan","year":"2024","unstructured":"Churan Yu, Yiwei Ru, Zhenbo Xu, Huijia Wu, Hujiang Yang, and Zhaofeng He. 2024b. DualActNet: Exploiting SlowFast Architecture for Micro-action Recognition. In Chinese Conference on Biometric Recognition. Springer, 59-68."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688974"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00542"},{"key":"e_1_3_2_1_33_1","volume-title":"Towards Micro-Action Recognition with Limited Annotations: An Asynchronous Pseudo Labeling and Training Approach. arXiv preprint arXiv:2504.07785","author":"Zhang Yan","year":"2025","unstructured":"Yan Zhang, Lechao Cheng, Yaxiong Wang, Zhun Zhong, and Meng Wang. 2025. Towards Micro-Action Recognition with Limited Annotations: An Asynchronous Pseudo Labeling and Training Approach. arXiv preprint arXiv:2504.07785 (2025)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3762095","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:03:54Z","timestamp":1765339434000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3762095"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":33,"alternative-id":["10.1145\/3746027.3762095","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3762095","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}