{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T15:45:17Z","timestamp":1781797517999,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3688975","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"11313-11319","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Advancing Micro-Action Recognition with Multi-Auxiliary Heads and Hybrid Loss Optimization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5121-1682","authenticated-orcid":false,"given":"Qiankun","family":"Li","sequence":"first","affiliation":[{"name":"HFIPS, Chinese Academy of Sciences &amp; University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5502-2535","authenticated-orcid":false,"given":"Xiaolong","family":"Huang","sequence":"additional","affiliation":[{"name":"Mila - Quebec AI Institute &amp; Concordia University, Canada, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4300-3017","authenticated-orcid":false,"given":"Huabao","family":"Chen","sequence":"additional","affiliation":[{"name":"HFIPS, Chinese Academy of Sciences &amp; University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5059-4423","authenticated-orcid":false,"given":"Feng","family":"He","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1036-6762","authenticated-orcid":false,"given":"Qiupu","family":"Chen","sequence":"additional","affiliation":[{"name":"HFIPS, Chinese Academy of Sciences &amp; University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1859-900X","authenticated-orcid":false,"given":"Zengfu","family":"Wang","sequence":"additional","affiliation":[{"name":"HFIPS, Chinese Academy of Sciences &amp; University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"2","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is space-time attention all you need for video understanding?. In ICML, Vol. 2. 4.","journal-title":"ICML"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"e_1_3_2_1_4_1","volume-title":"From static to dynamic: Adapting landmark-aware image models for facial expression recognition in videos. arXiv preprint arXiv:2312.05447","author":"Chen Yin","year":"2023","unstructured":"Yin Chen, Jia Li, Shiguang Shan, Meng Wang, and Richang Hong. 2023. From static to dynamic: Adapting landmark-aware image models for facial expression recognition in videos. arXiv preprint arXiv:2312.05447 (2023)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45103-X_50"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_1_7_1","first-page":"35946","article-title":"Masked autoencoders as spatiotemporal learners","volume":"35","author":"Feichtenhofer Christoph","year":"2022","unstructured":"Christoph Feichtenhofer, Yanghao Li, Kaiming He, et al. 2022. Masked autoencoders as spatiotemporal learners. Advances in Neural Information Processing Systems, Vol. 35 (2022), 35946--35958.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01003"},{"key":"e_1_3_2_1_9_1","volume-title":"Theory, Applications and Future Trends. arXiv preprint arXiv:2301.05712","author":"Gui Jie","year":"2023","unstructured":"Jie Gui, Tuo Chen, Qiong Cao, Zhenan Sun, Hao Luo, and Dacheng Tao. 2023. A Survey of Self-Supervised Learning from Multiple Perspectives: Algorithms, Theory, Applications and Future Trends. arXiv preprint arXiv:2301.05712 (2023)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3358415"},{"key":"e_1_3_2_1_11_1","volume-title":"MAC 2024: Micro-Action Analysis Grand Challenge. In Proceedings of the 32nd ACM International Conference on Multimedia.","author":"Guo Dan","year":"2024","unstructured":"Dan Guo, Xiaobai Li, Kun Li, Haoyu Chen, Jingjing Hu, Guoying Zhao, Yi Yang, and Meng Wang. 2024. MAC 2024: Micro-Action Analysis Grand Challenge. In Proceedings of the 32nd ACM International Conference on Multimedia."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_13_1","volume-title":"One Step Review. In Proceedings of the AAAI Conference on Artificial Intelligence","volume":"38","author":"Huang Xiaolong","year":"2024","unstructured":"Xiaolong Huang, Qiankun Li, Xueran Li, and Xuesong Gao. 2024. One Step Learning, One Step Review. In Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 38. 12644--12652."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Yuqi Huo Mingyu Ding Haoyu Lu Zhiwu Lu Tao Xiang Ji-Rong Wen Ziyuan Huang Jianwen Jiang Shiwei Zhang Mingqian Tang et al. 2021. Self-supervised video representation learning with constrained spatiotemporal jigsaw. (2021).","DOI":"10.24963\/ijcai.2021\/104"},{"key":"e_1_3_2_1_15_1","volume-title":"3D convolutional neural networks for human action recognition","author":"Ji Shuiwang","year":"2012","unstructured":"Shuiwang Ji, Wei Xu, Ming Yang, and Kai Yu. 2012. 3D convolutional neural networks for human action recognition. IEEE transactions on pattern analysis and machine intelligence, Vol. 35, 1 (2012), 221--231."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00559"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"e_1_3_2_1_18_1","volume-title":"Emotion separation and recognition from a facial expression by generating the poker face with vision transformers. arXiv preprint arXiv:2207.11081","author":"Li Jia","year":"2022","unstructured":"Jia Li, Jiantao Nie, Dan Guo, Richang Hong, and Meng Wang. 2022. Emotion separation and recognition from a facial expression by generating the poker face with vision transformers. arXiv preprint arXiv:2207.11081 (2022)."},{"key":"e_1_3_2_1_19_1","volume-title":"MMAD: Multi-label Micro-Action Detection in Videos. arXiv preprint arXiv:2407.05311","author":"Li Kun","year":"2024","unstructured":"Kun Li, Dan Guo, Pengyu Liu, Guoliang Chen, and Meng Wang. 2024. MMAD: Multi-label Micro-Action Detection in Videos. arXiv preprint arXiv:2407.05311 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01826"},{"key":"e_1_3_2_1_21_1","volume-title":"Embracing large natural data: Enhancing medical image analysis via cross-domain fine-tuning","author":"Li Qiankun","year":"2023","unstructured":"Qiankun Li, Xiaolong Huang, Bo Fang, Huabao Chen, Siyuan Ding, and Xu Liu. 2023. Embracing large natural data: Enhancing medical image analysis via cross-domain fine-tuning. IEEE Journal of Biomedical and Health Informatics (2023)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3606042.3616458"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612496"},{"key":"e_1_3_2_1_24_1","volume-title":"Fuzzy-ViT: A Deep Neuro-Fuzzy System for Cross-Domain Transfer Learning from Large-scale General Data to Medical Image","author":"Li Qiankun","year":"2024","unstructured":"Qiankun Li, Yimou Wang, Yani Zhang, Zhaoyu Zuo, Junxin Chen, and Wei Wang. 2024. Fuzzy-ViT: A Deep Neuro-Fuzzy System for Cross-Domain Transfer Learning from Large-scale General Data to Medical Image. IEEE Transactions on Fuzzy Systems (2024)."},{"key":"e_1_3_2_1_25_1","volume-title":"2023 e. PGA-Net: Polynomial Global Attention Network With Mean Curvature Loss for Lane Detection","author":"Li Qiankun","year":"2023","unstructured":"Qiankun Li, Xianwang Yu, Junxin Chen, Ben-Guo He, Wei Wang, Danda B Rawat, and Zhihan Lyu. 2023 e. PGA-Net: Polynomial Global Attention Network With Mean Curvature Loss for Lane Detection. IEEE Transactions on Intelligent Transportation Systems (2023)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01049"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548238"},{"key":"e_1_3_2_1_29_1","volume-title":"Dorota Kami'nska, Tomasz Sapi'nski, Sergio Escalera, and Gholamreza Anbarjafari.","author":"Noroozi Fatemeh","year":"2018","unstructured":"Fatemeh Noroozi, Ciprian Adrian Corneanu, Dorota Kami'nska, Tomasz Sapi'nski, Sergio Escalera, and Gholamreza Anbarjafari. 2018. Survey on emotional body gesture recognition. IEEE transactions on affective computing, Vol. 12, 2 (2018), 505--523."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.590"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3551791"},{"key":"e_1_3_2_1_32_1","volume-title":"Two-stream convolutional networks for action recognition in videos. Advances in neural information processing systems","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-stream convolutional networks for action recognition in videos. Advances in neural information processing systems, Vol. 27 (2014)."},{"key":"e_1_3_2_1_33_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Roshan Zamir, and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_1_34_1","volume-title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems","author":"Tong Zhan","year":"2022","unstructured":"Zhan Tong, Yibing Song, Jue Wang, and Limin Wang. 2022. Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems, Vol. 35 (2022), 10078--10093."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Du Tran Heng Wang Lorenzo Torresani Jamie Ray Yann LeCun and Manohar Paluri. 2018. A closer look at spatiotemporal convolutions for action recognition. In CVPR. 6450--6459.","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings, Part XVII 16","author":"Wang Jiangliu","year":"2020","unstructured":"Jiangliu Wang, Jianbo Jiao, and Yun-Hui Liu. 2020. Self-supervised video representation learning by pace prediction. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XVII 16. Springer, 504--521."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01398"},{"key":"e_1_3_2_1_39_1","volume-title":"Temporal segment networks for action recognition in videos","author":"Wang Limin","year":"2018","unstructured":"Limin Wang, Yuanjun Xiong, Zhe Wang, Yu Qiao, Dahua Lin, Xiaoou Tang, and Luc Van Gool. 2018. Temporal segment networks for action recognition in videos. IEEE transactions on pattern analysis and machine intelligence, Vol. 41, 11 (2018), 2740--2755."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01432"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Yi Wang Kunchang Li Xinhao Li Jiashuo Yu Yinan He Guo Chen Baoqi Pei Rongkun Zheng Jilan Xu Zun Wang et al. 2024. Internvideo2: Scaling video foundation models for multimodal video understanding. arXiv preprint arXiv:2403.15377 (2024).","DOI":"10.1007\/978-3-031-73013-9_23"},{"key":"e_1_3_2_1_42_1","volume-title":"Internvideo: General video foundation models via generative and discriminative learning. arXiv preprint arXiv:2212.03191","author":"Wang Yi","year":"2022","unstructured":"Yi Wang, Kunchang Li, Yizhuo Li, Yinan He, Bingkun Huang, Zhiyu Zhao, Hongjie Zhang, Jilan Xu, Yi Liu, Zun Wang, et al. 2022. Internvideo: General video foundation models via generative and discriminative learning. arXiv preprint arXiv:2212.03191 (2022)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01267-0_19"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSS.2022.3223251"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108282"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3556644"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3688975","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3688975","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:28Z","timestamp":1750295848000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3688975"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":46,"alternative-id":["10.1145\/3664647.3688975","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3688975","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}