{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T15:54:17Z","timestamp":1784994857640,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":74,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China (No. 62203476), Natural Science Foundation of Shenzhen (No. JCYJ20230807120801002)","award":["2"],"award-info":[{"award-number":["2"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681015","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"4909-4918","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":26,"title":["Multi-Modality Co-Learning for Efficient Skeleton-based Action Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0994-9146","authenticated-orcid":false,"given":"Jinfu","family":"Liu","sequence":"first","affiliation":[{"name":"State Key Laboratory of General Artificial Intelligence, Peking University, Shenzhen Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3957-7061","authenticated-orcid":false,"given":"Chen","family":"Chen","sequence":"additional","affiliation":[{"name":"Center for Research in Computer Vision, University of Central Florida, Orlando, FL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6332-8316","authenticated-orcid":false,"given":"Mengyuan","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of General Artificial Intelligence, Peking University, Shenzhen Graduate School, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00333"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2856094"},{"key":"e_1_3_2_2_3_1","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence (AAAI). AAAI, Canada, 3199--3207","author":"Bruce XB","year":"2021","unstructured":"XB Bruce, Yan Liu, and Keith CC Chan. 2021. Multimodal fusion via teacher-student network for indoor action recognition. In Proceedings of the AAAI Conference on Artificial Intelligence (AAAI). AAAI, Canada, 3199--3207."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2015.7350781"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475574"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01311"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i2.16197"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00026"},{"key":"e_1_3_2_2_9_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE","author":"Ha Myoung Hoon","year":"2022","unstructured":"Hyung-gun Chi, Myoung Hoon Ha, Seunggeun Chi, Sang Wan Lee, Qixing Huang, and Karthik Ramani. 2022. InfoGCN: Representation Learning for Human Skeleton-Based Action Recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, New Orleans, USA, 20186--20196."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58545-7_5"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610217"},{"key":"e_1_3_2_2_12_1","volume-title":"Proceedings of the CAAI International Conference on Artificial Intelligence (CICAI). CAAI","author":"Ding Runwei","year":"2023","unstructured":"Runwei Ding, Yuhang Wen, Jinfu Liu, Nan Dai, Fanyang Meng, and Mengyuan Liu. 2023. Integrating Human Parsing and Pose Network for Human Action Recognition. In Proceedings of the CAAI International Conference on Artificial Intelligence (CICAI). CAAI, Fuzhou, China, 182--194."},{"key":"e_1_3_2_2_13_1","unstructured":"Haodong Duan Jiaqi Wang Kai Chen and Dahua Lin. 2022. DG-STGCN: Dynamic Spatial-Temporal Modeling for Skeleton-based Action Recognition."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00298"},{"key":"e_1_3_2_2_15_1","unstructured":"Peng Gao Jiaming Han Renrui Zhang Ziyi Lin Shijie Geng Aojun Zhou Wei Zhang Pan Lu Conghui He Xiangyu Yue Hongsheng Li and Yu Qiao. 2023. LLaMA-Adapter V2: Parameter-Efficient Visual Instruction Model."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01436"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548210"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2640292"},{"key":"e_1_3_2_2_20_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR). OpenReview.net, Kigali, Rwanda.","author":"Huang Xiaohu","year":"2023","unstructured":"Xiaohu Huang, Hao Zhou, Bin Feng, Xinggang Wang, Wenyu Liu, Jian Wang, Haocheng Feng, Junyu Han, Errui Ding, and Jingdong Wang. 2023. Graph contrastive learning for skeleton-based action recognition. In Proceedings of the International Conference on Learning Representations (ICLR). OpenReview.net, Kigali, Rwanda."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01330"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.486"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2812099"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00942"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00958"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3061115"},{"key":"e_1_3_2_2_27_1","unstructured":"Chao Li Qiaoyong Zhong Di Xie and Shiliang Pu. 2018. Co-occurrence Feature Learning from Skeleton Data for Action Recognition and Detection with Hierarchical Aggregation."},{"key":"e_1_3_2_2_28_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). ACM","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In Proceedings of the International Conference on Machine Learning (ICML). ACM, Hawaii, USA, 19730--19742."},{"key":"e_1_3_2_2_29_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). ACM","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Proceedings of the International Conference on Machine Learning (ICML). ACM, Seoul, Korea, 12888--12900."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00572"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW63481.2024.10645397"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW63481.2024.10645440"},{"key":"e_1_3_2_2_33_1","unstructured":"Dongjingdin Liu Pengpeng Chen Miao Yao Yijing Lu Zijie Cai and Yuxin Tian. 2023. TSGCNeXt: Dynamic-Static Multi-Graph Convolution for Efficient Skeleton-Based Action Recognition with Long-term Learning Potential."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02022"},{"key":"e_1_3_2_2_35_1","unstructured":"Hong Liu Juanhui Tu and Mengyuan Liu. 2017. Two-Stream 3D Convolutional Neural Network for Skeleton-Based Action Recognition."},{"key":"e_1_3_2_2_36_1","volume-title":"Explore Human Parsing Modality for Action Recognition. CAAI Transactions on Intelligence Technology","author":"Liu Jinfu","year":"2024","unstructured":"Jinfu Liu, Runwei Ding, Yuhang Wen, Nan Dai, Fanyang Meng, Shen Zhao, and Mengyuan Liu. 2024. Explore Human Parsing Modality for Action Recognition. CAAI Transactions on Intelligence Technology (2024)."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2916873"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2017.2785279"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3271811"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW63481.2024.10645450"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.02.030"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25258"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-60639-8_40"},{"key":"e_1_3_2_2_44_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV). Springer, Tel-Aviv, Israel, 211--227","author":"Neel Trivedi","year":"2022","unstructured":"Trivedi Neel and Sarvadevabhatla Ravi Kiran. 2022. PSUMNet: Unified Modality Part Streams Are All You Need for Efficient Pose-Based Action Recognition. In Proceedings of the European Conference on Computer Vision (ECCV). Springer, Tel-Aviv, Israel, 211--227."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3070127"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00713"},{"key":"e_1_3_2_2_47_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). ACM, Online, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In Proceedings of the International Conference on Machine Learning (ICML). ACM, Online, 8748--8763."},{"key":"e_1_3_2_2_48_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE","author":"Shahroudy Amir","year":"2016","unstructured":"Amir Shahroudy, Jun Liu, Tian-Tsong Ng, and Gang Wang. 2016. NTU RGBD: A Large Scale Dataset for 3D Human Activity Analysis. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, Las Vegas, USA, 1010--1019."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2818328"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_2_51_1","unstructured":"Ultralytics. 2022. ultralytics\/yolov5: v7.0 - YOLOv5 SOTA Realtime Instance Segmentation."},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.387"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.339"},{"key":"e_1_3_2_2_54_1","unstructured":"Mengmeng Wang Jiazheng Xing and Yong Liu. 2021. ActionCLIP: A New Paradigm for Video Action Recognition."},{"key":"e_1_3_2_2_55_1","unstructured":"Shengqin Wang Yongji Zhang Fenglin Wei Kai Wang Minghao Zhao and Yu Jiang. 2022. Skeleton-based Action Recognition via Temporal-Channel Aggregation."},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547811"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01021"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3334954"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10342472"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3077512"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3086758"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00943"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611900"},{"key":"e_1_3_2_2_64_1","unstructured":"Haojun Xu Yan Gao Zheng Hui Jie Li and Xinbo Gao. 2023. Language Knowledge-Assisted Representation Learning for Skeleton-Based Action Recognition."},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413941"},{"key":"e_1_3_2_2_67_1","first-page":"3522","article-title":"MMNet: A Model-Based Multimodal Network for Human Action Recognition in RGB-D Videos","volume":"45","author":"Yu Bruce X.B.","year":"2023","unstructured":"Bruce X.B. Yu, Yan Liu, Xiang Zhang, Sheng-hua Zhong, and Keith C.C. Chan. 2023. MMNet: A Model-Based Multimodal Network for Human Action Recognition in RGB-D Videos. IEEE Transactions on Pattern Analysis and Machine Intelligence, Vol. 45, 3 (2023), 3522--3538.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01240-3_9"},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01460"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01022"},{"key":"e_1_3_2_2_71_1","unstructured":"Yuxuan Zhou Zhi-Qi Cheng Chao Li Yifeng Geng Xuansong Xie and Margret Keuper. 2022. Hypergraph Transformer for Skeleton-based Action Recognition."},{"key":"e_1_3_2_2_72_1","unstructured":"Deyao Zhu Jun Chen Xiaoqian Shen Xiang Li and Mohamed Elhoseiny. 2023. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models."},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00249"},{"key":"e_1_3_2_2_74_1","unstructured":"Yanqiao Zhu Yichen Xu Feng Yu Qiang Liu Shu Wu and Liang Wang. 2020. Deep graph contrastive representation learning."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681015","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681015","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:36Z","timestamp":1750295856000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681015"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":74,"alternative-id":["10.1145\/3664647.3681015","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681015","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}