{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T10:00:58Z","timestamp":1766138458347,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Natural Science Foundation of Guangdong Province of China","award":["2021A1515011905"],"award-info":[{"award-number":["2021A1515011905"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62101543"],"award-info":[{"award-number":["62101543"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612560","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:26:54Z","timestamp":1698391614000},"page":"3657-3666","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Skeletal Spatial-Temporal Semantics Guided Homogeneous-Heterogeneous Multimodal Network for Action Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-6698-0652","authenticated-orcid":false,"given":"Chenwei","family":"Zhang","sequence":"first","affiliation":[{"name":"Sun Yat-Sen University &amp; Shenzhen MSU-BIT University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8571-118X","authenticated-orcid":false,"given":"Yuxuan","family":"Hu","sequence":"additional","affiliation":[{"name":"Sun Yat-Sen University &amp; Shenzhen MSU-BIT University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7345-5071","authenticated-orcid":false,"given":"Min","family":"Yang","sequence":"additional","affiliation":[{"name":"SIAT, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4592-3875","authenticated-orcid":false,"given":"Chengming","family":"Li","sequence":"additional","affiliation":[{"name":"Shenzhen MSU-BIT University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4952-699X","authenticated-orcid":false,"given":"Xiping","family":"Hu","sequence":"additional","affiliation":[{"name":"Shenzhen MSU-BIT University, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00333"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2017.77"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_4_1","volume-title":"Hinton","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey E. Hinton. 2020. A Simple Framework for Contrastive Learning of Visual Representations. CoRR, Vol. abs\/2002.05709 (2020). [arXiv]2002.05709 https:\/\/arxiv.org\/abs\/2002.05709"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01311"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00026"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01955"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3127885"},{"key":"e_1_3_2_1_9_1","volume-title":"Fran\u00e7ois Br\u00e9 mond, and Monique Thonnat","author":"Das Srijan","year":"2020","unstructured":"Srijan Das, Saurav Sharma, Rui Dai, Fran\u00e7ois Br\u00e9 mond, and Monique Thonnat. 2020. VPN: Learning Video-Pose Embedding for Activities of Daily Living. CoRR, Vol. abs\/2007.03056 (2020). [arXiv]2007.03056 https:\/\/arxiv.org\/abs\/2007.03056"},{"key":"e_1_3_2_1_10_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. CoRR","author":"Dosovitskiy Alexey","year":"1929","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2020. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. CoRR, Vol. abs\/2010.11929 (2020). [arXiv]2010.11929 https:\/\/arxiv.org\/abs\/2010.11929"},{"key":"e_1_3_2_1_11_1","volume-title":"Revisiting Skeleton-based Action Recognition. CoRR","author":"Duan Haodong","year":"2021","unstructured":"Haodong Duan, Yue Zhao, Kai Chen, Dian Shao, Dahua Lin, and Bo Dai. 2021. Revisiting Skeleton-based Action Recognition. CoRR, Vol. abs\/2104.13586 (2021). showeprint[arXiv]2104.13586 https:\/\/arxiv.org\/abs\/2104.13586"},{"key":"e_1_3_2_1_12_1","unstructured":"W. Elmenreich. 2002. An Introduction to Sensor Fusion. (2002)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00369"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3073107"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_1_16_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2015. Deep Residual Learning for Image Recognition. arxiv: 1512.03385 [cs.CV]"},{"key":"e_1_3_2_1_17_1","unstructured":"Xiaohu Huang Hao Zhou Bin Feng Xinggang Wang Wenyu Liu Jian Wang Haocheng Feng Junyu Han Errui Ding and Jingdong Wang. 2023. Graph Contrastive Learning for Skeleton-based Action Recognition. arxiv: 2301.10900 [cs.CV]"},{"key":"e_1_3_2_1_18_1","unstructured":"E. Jang S. Gu and B. Poole. 2016. Categorical Reparameterization with Gumbel-Softmax. arXiv e-prints (2016)."},{"key":"e_1_3_2_1_19_1","unstructured":"Sangwon Kim Dasom Ahn and Byoung Chul Ko. 2023. Cross-Modal Learning with 3D Deformable Attention for Action Recognition. arxiv: 2212.05638 [cs.CV]"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00240"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW.2017.8026282"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00471"},{"key":"e_1_3_2_1_23_1","volume-title":"Actional-Structural Graph Convolutional Networks for Skeleton-based Action Recognition. CoRR","author":"Li Maosen","year":"2019","unstructured":"Maosen Li, Siheng Chen, Xu Chen, Ya Zhang, Yanfeng Wang, and Qi Tian. 2019. Actional-Structural Graph Convolutional Networks for Skeleton-based Action Recognition. CoRR, Vol. abs\/1904.12659 (2019). [arXiv]1904.12659 http:\/\/arxiv.org\/abs\/1904.12659"},{"volume-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","author":"Li S.","key":"e_1_3_2_1_24_1","unstructured":"S. Li, W. Li, C. Cook, C. Zhu, and Y. Gao. 2018. Independently Recurrent Neural Network (IndRNN): Building A Longer and Deeper RNN. In 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"X. Li W. Wang X. Hu and J. Yang. 2020. Selective Kernel Networks. IEEE (2020).","DOI":"10.1109\/CVPR.2019.00060"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2916873"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.02.030"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00022"},{"key":"e_1_3_2_1_29_1","unstructured":"Xuran Pan Tianzhu Ye Dongchen Han Shiji Song and Gao Huang. 2022. Contrastive Language-Image Pre-Training with Knowledge Graphs. arxiv: 2210.08901 [cs.CV]"},{"key":"e_1_3_2_1_30_1","volume-title":"Spatial Temporal Transformer Network for Skeleton-Based Action Recognition. In International Conference on Pattern Recognition.","author":"Plizzari Chiara","year":"2021","unstructured":"Chiara Plizzari, Marco Cannici, and Matteo Matteucci. 2021. Spatial Temporal Transformer Network for Skeleton-Based Action Recognition. In International Conference on Pattern Recognition."},{"key":"e_1_3_2_1_31_1","unstructured":"Y. Rao W. Zhao B. Liu J. Lu and C. J. Hsieh. 2021. DynamicViT: Efficient Vision Transformers with Dynamic Token Sparsification. (2021)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2013.2246148"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/KBEI.2017.8324867"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.115"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/JBHI.2021.3122299"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01230"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"L. Shi Y. Zhang J. Cheng and H. Lu. 2020. Decoupled Spatial-Temporal Attention Network for Skeleton-Based Action Recognition. (2020).","DOI":"10.1007\/978-3-030-69541-5_3"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00132"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jonathon Shlens and Zbigniew Wojna. 2015. Rethinking the Inception Architecture for Computer Vision. arxiv: 1512.00567 [cs.CV]","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_40_1","volume-title":"Le","author":"Tan Mingxing","year":"2020","unstructured":"Mingxing Tan and Quoc V. Le. 2020. EfficientNet: Rethinking Model Scaling for Convolutional Neural Networks. arxiv: 1905.11946 [cs.LG]"},{"key":"e_1_3_2_1_41_1","unstructured":"Y. Tian D. Krishnan and P. Isola. 2019. Contrastive Representation Distillation. (2019)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-25072-9_14"},{"key":"e_1_3_2_1_43_1","first-page":"2579","article-title":"Visualizing Data using t-SNE","volume":"9","author":"van der Maaten Laurens","year":"2008","unstructured":"Laurens van der Maaten and Geoffrey Hinton. 2008. Visualizing Data using t-SNE. Journal of Machine Learning Research, Vol. 9, 86 (2008), 2579--2605. http:\/\/jmlr.org\/papers\/v9\/vandermaaten08a.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_1_45_1","volume-title":"Cross-view Action Modeling, Learning and Recognition. CoRR","author":"Wang Jiang","year":"2014","unstructured":"Jiang Wang, Xiaohan Nie, Yin Xia, Ying Wu, and Song-Chun Zhu. 2014a. Cross-view Action Modeling, Learning and Recognition. CoRR, Vol. abs\/1405.2941 (2014). [arXiv]1405.2941 http:\/\/arxiv.org\/abs\/1405.2941"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.339"},{"key":"e_1_3_2_1_47_1","unstructured":"M. Wu C. Zhuang M. Mosse D. Yamins and N. Goodman. 2020. On Mutual Information in Contrastive Learning for Visual Representations. arXiv (2020)."},{"key":"e_1_3_2_1_48_1","volume-title":"Rethinking Spatiotemporal Feature Learning For Video Understanding. CoRR","author":"Xie Saining","year":"2017","unstructured":"Saining Xie, Chen Sun, Jonathan Huang, Zhuowen Tu, and Kevin Murphy. 2017. Rethinking Spatiotemporal Feature Learning For Video Understanding. CoRR, Vol. abs\/1712.04851 (2017). [arXiv]1712.04851 http:\/\/arxiv.org\/abs\/1712.04851"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Sijie Yan Yuanjun Xiong and Dahua Lin. 2018. Spatial Temporal Graph Convolutional Networks for Skeleton-Based Action Recognition. In Proceedings of the Thirty-Second AAAI Conference on Artificial Intelligence and Thirtieth Innovative Applications of Artificial Intelligence Conference and Eighth AAAI Symposium on Educational Advances in Artificial Intelligence (New Orleans Louisiana USA) (AAAI'18\/IAAI'18\/EAAI'18). AAAI Press Article 912 9 pages.","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413941"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16430"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3177813"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00119"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475473"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612560","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612560","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:58:01Z","timestamp":1755820681000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612560"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":54,"alternative-id":["10.1145\/3581783.3612560","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612560","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}