{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:08:52Z","timestamp":1783436932906,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":76,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,6]],"date-time":"2025-05-06T00:00:00Z","timestamp":1746489600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"U.S. National Science Foundation (NSF)","award":["CNS-2112562"],"award-info":[{"award-number":["CNS-2112562"]}]},{"name":"U.S. National Science Foundation (NSF)","award":["CNS-2107060"],"award-info":[{"award-number":["CNS-2107060"]}]},{"name":"U.S. National Science Foundation (NSF)","award":["CNS-2213688"],"award-info":[{"award-number":["CNS-2213688"]}]},{"name":"U.S. National Science Foundation (NSF)","award":["CNS-2312716"],"award-info":[{"award-number":["CNS-2312716"]}]},{"DOI":"10.13039\/100000190","name":"U.S. Department of Commerce","doi-asserted-by":"publisher","award":["70NANB21H043"],"award-info":[{"award-number":["70NANB21H043"]}],"id":[{"id":"10.13039\/100000190","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,6]]},"DOI":"10.1145\/3715014.3722068","type":"proceedings-article","created":{"date-parts":[[2025,5,4]],"date-time":"2025-05-04T23:39:01Z","timestamp":1746401941000},"page":"240-253","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Babel: A Scalable Pre-trained Model for Multi-Modal Sensing via Expandable Modality Alignment"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6316-3980","authenticated-orcid":false,"given":"Shenghong","family":"Dai","sequence":"first","affiliation":[{"name":"University of Wisconsin-Madison, Madison, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4685-9633","authenticated-orcid":false,"given":"Shiqi","family":"Jiang","sequence":"additional","affiliation":[{"name":"Microsoft Research, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5481-2851","authenticated-orcid":false,"given":"Yifan","family":"Yang","sequence":"additional","affiliation":[{"name":"Microsoft Research, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9107-013X","authenticated-orcid":false,"given":"Ting","family":"Cao","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6047-9709","authenticated-orcid":false,"given":"Mo","family":"Li","sequence":"additional","affiliation":[{"name":"HKUST, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5548-8862","authenticated-orcid":false,"given":"Suman","family":"Banerjee","sequence":"additional","affiliation":[{"name":"University of Wisconsin-Madison, Madison, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8131-7439","authenticated-orcid":false,"given":"Lili","family":"Qiu","sequence":"additional","affiliation":[{"name":"UT Austin, MSR Asia, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,5,6]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Sizhe An Yin Li and Umit Ogras. 2022. mRI: Multi-modal 3D Human Pose Estimation Dataset using mmWave RGB-D and Inertial Sensors. arXiv:2210.08394 [cs.CV] https:\/\/arxiv.org\/abs\/2210.08394"},{"key":"e_1_3_2_1_2_1","first-page":"27414","article-title":"mri: Multi-modal 3d human pose estimation dataset using mmwave, rgb-d, and inertial sensors","volume":"35","author":"An Sizhe","year":"2022","unstructured":"Sizhe An, Yin Li, and Umit Ogras. 2022. mri: Multi-modal 3d human pose estimation dataset using mmwave, rgb-d, and inertial sensors. Advances in Neural Information Processing Systems 35 (2022), 27414--27426.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477030"},{"key":"e_1_3_2_1_4_1","unstructured":"Mohammud J. Bocus Wenda Li Shelly Vishwakarma Roget Kou Chong Tang Karl Woodbridge Ian Craddock Ryan McConville Raul Santos-Rodriguez Kevin Chetty and Robert Piechocki. 2021. OPERAnet: A Multimodal Activity Recognition Dataset Acquired from Radio Frequency and Vision-based Sensors. arXiv:2110.04239 [eess.SP]"},{"key":"e_1_3_2_1_5_1","volume-title":"Bagging predictors. Machine learning 24","author":"Breiman Leo","year":"1996","unstructured":"Leo Breiman. 1996. Bagging predictors. Machine learning 24 (1996), 123--140."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.143"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015432"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Anjun Chen Xiangyu Wang Shaohao Zhu Yanxu Li Jiming Chen and Qi Ye. 2023. mmBody Benchmark: 3D Body Reconstruction Dataset and Analysis for Millimeter Wave Radar. arXiv:2209.05070 [cs.CV]","DOI":"10.1145\/3503161.3548262"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2015.7350781"},{"key":"e_1_3_2_1_10_1","volume-title":"International conference on machine learning. PMLR, 1597--1607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In International conference on machine learning. PMLR, 1597--1607."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.3390\/s19071716"},{"key":"e_1_3_2_1_12_1","first-page":"3","volume-title":"Proc. ACM Interact. Mob. Wearable Ubiquitous Technol. 6","author":"Deldari Shohreh","year":"2022","unstructured":"Shohreh Deldari, Hao Xue, Aaqib Saeed, Daniel V. Smith, and Flora D. Salim. 2022. COCOA: Cross Modality Contrastive Learning for Sensor Data. Proc. ACM Interact. Mob. Wearable Ubiquitous Technol. 6, 3 (2022), 108:1--108:28."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3560905.3568435"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2006.79"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2019.06.070"},{"key":"e_1_3_2_1_16_1","volume-title":"Catastrophic forgetting in connectionist networks. Trends in cognitive sciences 3, 4","author":"French Robert M","year":"1999","unstructured":"Robert M French. 1999. Catastrophic forgetting in connectionist networks. Trends in cognitive sciences 3, 4 (1999), 128--135."},{"key":"e_1_3_2_1_17_1","volume-title":"Armand Joulin, and Ishan Misra.","author":"Girdhar Rohit","year":"2023","unstructured":"Rohit Girdhar, Alaaeldin El-Nouby, Zhuang Liu, Mannat Singh, Kalyan Vasudev Alwala, Armand Joulin, and Ishan Misra. 2023. ImageBind: One Embedding Space To Bind Them All. arXiv:2305.05665 [cs.CV]"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2001.977164"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_1_20_1","unstructured":"Jiaming Han Kaixiong Gong Yiyuan Zhang Jiaqi Wang Kaipeng Zhang Dahua Lin Yu Qiao Peng Gao and Xiangyu Yue. 2023. OneLLM: One Framework to Align All Modalities with Language. arXiv:2312.03700 [cs.CV]"},{"key":"e_1_3_2_1_21_1","volume-title":"Animate Anyone: Consistent and Controllable Image-to-Video Synthesis for Character Animation. arXiv:2311.17117 [cs.CV] https:\/\/arxiv.org\/abs\/2311.17117","author":"Hu Li","year":"2024","unstructured":"Li Hu, Xin Gao, Peng Zhang, Ke Sun, Bang Zhang, and Liefeng Bo. 2024. Animate Anyone: Consistent and Controllable Image-to-Video Synthesis for Character Animation. arXiv:2311.17117 [cs.CV] https:\/\/arxiv.org\/abs\/2311.17117"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3027979"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3517246"},{"key":"e_1_3_2_1_24_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev Mustafa Suleyman and Andrew Zisserman. 2017. The Kinetics Human Action Video Dataset. arXiv:1705.06950 [cs.CV]"},{"key":"e_1_3_2_1_25_1","unstructured":"Gregory Koch Richard Zemel Ruslan Salakhutdinov et al. 2015. Siamese neural networks for one-shot image recognition. In ICML deep learning workshop Vol. 2. Lille 1--30."},{"key":"e_1_3_2_1_26_1","unstructured":"Bo Li Yuanhan Zhang Dong Guo Renrui Zhang Feng Li Hao Zhang Kaichen Zhang Peiyuan Zhang Yanwei Li Ziwei Liu and Chunyuan Li. 2024. LLaVA-OneVision: Easy Visual Task Transfer. arXiv:2408.03326 [cs.CV] https:\/\/arxiv.org\/abs\/2408.03326"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jag.2022.102926"},{"key":"e_1_3_2_1_28_1","volume-title":"Action recognition based on a bag of 3d points. In 2010 IEEE computer society conference on computer vision and pattern recognition-workshops","author":"Li Wanqing","unstructured":"Wanqing Li, Zhengyou Zhang, and Zicheng Liu. 2010. Action recognition based on a bag of 3d points. In 2010 IEEE computer society conference on computer vision and pattern recognition-workshops. IEEE, 9--14."},{"key":"e_1_3_2_1_29_1","volume-title":"Video-llava: Learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122","author":"Lin Bin","year":"2023","unstructured":"Bin Lin, Yang Ye, Bin Zhu, Jiaxi Cui, Munan Ning, Peng Jin, and Li Yuan. 2023. Video-llava: Learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122 (2023)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458864.3466904"},{"key":"e_1_3_2_1_31_1","unstructured":"Haokun Liu Derek Tam Mohammed Muqeeth Jay Mohta Tenghao Huang Mohit Bansal and Colin Raffel. 2022. Few-Shot Parameter-Efficient Fine-Tuning is Better and Cheaper than In-Context Learning. arXiv:2205.05638 [cs.LG]"},{"key":"e_1_3_2_1_32_1","volume-title":"FOCAL: Contrastive learning for multimodal time-series sensing signals in factorized orthogonal latent space. Advances in Neural Information Processing Systems 36","author":"Liu Shengzhong","year":"2024","unstructured":"Shengzhong Liu, Tomoyoshi Kimura, Dongxin Liu, Ruijie Wang, Jinyang Li, Suhas Diggavi, Mani Srivastava, and Tarek Abdelzaher. 2024. FOCAL: Contrastive learning for multimodal time-series sensing signals in factorized orthogonal latent space. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447993.3483244"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00934"},{"key":"e_1_3_2_1_35_1","unstructured":"Ilya Loshchilov and Frank Hutter. 2019. Decoupled Weight Decay Regularization. arXiv:1711.05101 [cs.LG]"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00810"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3560519"},{"key":"e_1_3_2_1_38_1","first-page":"16131","article-title":"DualNet: Continual learning, fast and slow","volume":"34","author":"Pham Quang","year":"2021","unstructured":"Quang Pham, Chenghao Liu, and Steven Hoi. 2021. DualNet: Continual learning, fast and slow. Advances in Neural Information Processing Systems 34 (2021), 16131--16144.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_39_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. arXiv:2103.00020 [cs.CV]"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3161174"},{"key":"e_1_3_2_1_41_1","unstructured":"Aditya Ramesh Prafulla Dhariwal Alex Nichol Casey Chu and Mark Chen. 2022. Hierarchical Text-Conditional Image Generation with CLIP Latents. arXiv:2204.06125 [cs.CV]"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2015.07.085"},{"key":"e_1_3_2_1_43_1","volume-title":"APE: Aligning Pretrained Encoders to Quickly Learn Aligned Multimodal Representations. arXiv:2210.03927 [cs.LG]","author":"Rosenfeld Elan","year":"2022","unstructured":"Elan Rosenfeld, Preetum Nakkiran, Hadi Pouransari, Oncel Tuzel, and Fartash Faghri. 2022. APE: Aligning Pretrained Encoders to Quickly Learn Aligned Multimodal Representations. arXiv:2210.03927 [cs.LG]"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2022.3170733"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Amir Shahroudy Jun Liu Tian-Tsong Ng and Gang Wang. 2016. NTU RGB+D: A Large Scale Dataset for 3D Human Activity Analysis. arXiv:1604.02808 [cs.CV]","DOI":"10.1109\/CVPR.2016.115"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2020.2994292"},{"key":"e_1_3_2_1_47_1","unstructured":"Zineng Tang Ziyi Yang Chenguang Zhu Michael Zeng and Mohit Bansal. 2023. Any-to-Any Generation via Composable Diffusion. arXiv:2305.11846 [cs.CV] https:\/\/arxiv.org\/abs\/2305.11846"},{"key":"e_1_3_2_1_48_1","volume-title":"Contrastive Multiview Coding. CoRR abs\/1906.05849","author":"Tian Yonglong","year":"2019","unstructured":"Yonglong Tian, Dilip Krishnan, and Phillip Isola. 2019. Contrastive Multiview Coding. CoRR abs\/1906.05849 (2019). arXiv:1906.05849 http:\/\/arxiv.org\/abs\/1906.05849"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Du Tran Heng Wang Lorenzo Torresani Jamie Ray Yann LeCun and Manohar Paluri. 2018. A Closer Look at Spatiotemporal Convolutions for Action Recognition. arXiv:1711.11248 [cs.CV]","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_2_1_50_1","volume-title":"Representation Learning with Contrastive Predictive Coding. CoRR abs\/1807.03748","author":"van den Oord A\u00e4ron","year":"2018","unstructured":"A\u00e4ron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation Learning with Contrastive Predictive Coding. CoRR abs\/1807.03748 (2018). arXiv:1807.03748 http:\/\/arxiv.org\/abs\/1807.03748"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643543"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3558277"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Yuxuan Weng Guoquan Wu Tianyue Zheng Yanbing Yang and Jun Luo. 2024. Large Model for Small Data: Foundation Model for Cross-Modal RF Human Activity Recognition. arXiv:2410.19766 [cs.CV] https:\/\/arxiv.org\/abs\/2410.19766","DOI":"10.1145\/3666025.3699349"},{"key":"e_1_3_2_1_54_1","unstructured":"Jay Zhangjie Wu Yixiao Ge Xintao Wang Weixian Lei Yuchao Gu Yufei Shi Wynne Hsu Ying Shan Xiaohu Qie and Mike Zheng Shou. 2023. Tune-A-Video: One-Shot Tuning of Image Diffusion Models for Text-to-Video Generation. arXiv:2212.11565 [cs.CV] https:\/\/arxiv.org\/abs\/2212.11565"},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition. 1912--1920","author":"Wu Zhirong","year":"2015","unstructured":"Zhirong Wu, Shuran Song, Aditya Khosla, Fisher Yu, Linguang Zhang, Xiaoou Tang, and Jianxiong Xiao. 2015. 3d shapenets: A deep representation for volumetric shapes. In Proceedings of the IEEE conference on computer vision and pattern recognition. 1912--1920."},{"key":"e_1_3_2_1_56_1","volume-title":"MuseV: Infinite-length and High Fidelity Virtual Human Video Generation with Visual Conditioned Parallel Denoising. arxiv","author":"Xia Zhiqiang","year":"2024","unstructured":"Zhiqiang Xia, Zhaokang Chen, Bin Wu, Chao Li, Kwok-Wai Hung, Chao Zhan, Yingjie He, and Wenjiang Zhou. 2024. MuseV: Infinite-length and High Fidelity Virtual Human Video Generation with Visual Conditioned Parallel Denoising. arxiv (2024)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485730.3485936"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485730.3485937"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3568113.3568124"},{"key":"e_1_3_2_1_60_1","volume-title":"MESEN: Exploit Multimodal Data to Design Unimodal Human Activity Recognition with Few Labels.","author":"Xu Lilin","year":"2023","unstructured":"Lilin Xu, Chaojie Gu, Rui Tan, Shibo He, and Jiming Chen. 2023. MESEN: Exploit Multimodal Data to Design Unimodal Human Activity Recognition with Few Labels. (2023)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"crossref","unstructured":"Sijie Yan Yuanjun Xiong and Dahua Lin. 2018. Spatial Temporal Graph Convolutional Networks for Skeleton-Based Action Recognition. arXiv:1801.07455 [cs.CV]","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_1_62_1","volume-title":"Sumei Sun, and Lihua Xie.","author":"Yang Jianfei","year":"2023","unstructured":"Jianfei Yang, Xinyan Chen, Dazhuo Wang, Han Zou, Chris Xiaoxuan Lu, Sumei Sun, and Lihua Xie. 2023. SenseFi: A Library and Benchmark on Deep-Learning-Empowered WiFi Human Sensing. arXiv:2207.07859 [cs.LG]"},{"key":"e_1_3_2_1_63_1","volume-title":"Sumei Sun, and Lihua Xie.","author":"Yang Jianfei","year":"2023","unstructured":"Jianfei Yang, Xinyan Chen, Dazhuo Wang, Han Zou, Chris Xiaoxuan Lu, Sumei Sun, and Lihua Xie. 2023. SenseFi: A Library and Benchmark on Deep-Learning-Empowered WiFi Human Sensing. arXiv:2207.07859 [cs.LG] https:\/\/arxiv.org\/abs\/2207.07859"},{"key":"e_1_3_2_1_64_1","volume-title":"Chris Xiaoxuan Lu, and Lihua Xie","author":"Yang Jianfei","year":"2023","unstructured":"Jianfei Yang, He Huang, Yunjiao Zhou, Xinyan Chen, Yuecong Xu, Shenghai Yuan, Han Zou, Chris Xiaoxuan Lu, and Lihua Xie. 2023. MM-Fi: Multi-Modal Non-Intrusive 4D Human Dataset for Versatile Wireless Sensing. arXiv:2305.10345 [eess.SP]"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.2017.1700082"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649361"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"crossref","unstructured":"Xiaohua Zhai Xiao Wang Basil Mustafa Andreas Steiner Daniel Keysers Alexander Kolesnikov and Lucas Beyer. 2022. LiT: Zero-Shot Transfer with Locked-image text Tuning. arXiv:2111.07991 [cs.CV]","DOI":"10.1109\/CVPR52688.2022.01759"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"crossref","unstructured":"Hang Zhang Xin Li and Lidong Bing. 2023. Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding. arXiv:2306.02858 [cs.CL]","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"e_1_3_2_1_69_1","unstructured":"Pan Zhang Xiaoyi Dong Yuhang Zang Yuhang Cao Rui Qian Lin Chen Qipeng Guo Haodong Duan Bin Wang Linke Ouyang et al. 2024. Internlm-xcomposer-2.5: A versatile large vision language model supporting long-contextual input and output. arXiv preprint arXiv:2407.03320 (2024)."},{"key":"e_1_3_2_1_70_1","unstructured":"Yiyuan Zhang Kaixiong Gong Kaipeng Zhang Hongsheng Li Yu Qiao Wanli Ouyang and Xiangyu Yue. 2023. Meta-Transformer: A Unified Framework for Multimodal Learning. arXiv:2307.10802 [cs.CV]"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3105387"},{"key":"e_1_3_2_1_72_1","volume-title":"CoRR abs\/2012.09164","author":"Zhao Hengshuang","year":"2020","unstructured":"Hengshuang Zhao, Li Jiang, Jiaya Jia, Philip H. S. Torr, and Vladlen Koltun. 2020. Point Transformer. CoRR abs\/2012.09164 (2020). arXiv:2012.09164 https:\/\/arxiv.org\/abs\/2012.09164"},{"key":"e_1_3_2_1_73_1","volume-title":"Through-Wall Human Pose Estimation Using Radio Signals. In 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR '18)","author":"Zhao Mingmin","year":"2018","unstructured":"Mingmin Zhao, Tianhong Li, Mohammad Abu Alsheikh, Yonglong Tian, Hang Zhao, Antonio Torralba, and Dina Katabi. 2018. Through-Wall Human Pose Estimation Using Radio Signals. In 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR '18)."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1145\/2639108.2639110"},{"key":"e_1_3_2_1_75_1","volume-title":"TENT: Connect Language Models with IoT Sensors for Zero-Shot Activity Recognition. arXiv:2311.08245 [cs.CV] https:\/\/arxiv.org\/abs\/2311.08245","author":"Zhou Yunjiao","year":"2023","unstructured":"Yunjiao Zhou, Jianfei Yang, Han Zou, and Lihua Xie. 2023. TENT: Connect Language Models with IoT Sensors for Zero-Shot Activity Recognition. arXiv:2311.08245 [cs.CV] https:\/\/arxiv.org\/abs\/2311.08245"},{"key":"e_1_3_2_1_76_1","unstructured":"Bin Zhu Bin Lin Munan Ning Yang Yan Jiaxi Cui HongFa Wang Yatian Pang Wenhao Jiang Junwu Zhang Zongwei Li Wancai Zhang Zhifeng Li Wei Liu and Li Yuan. 2024. LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment. arXiv:2310.01852 [cs.CV] https:\/\/arxiv.org\/abs\/2310.01852"}],"event":{"name":"SenSys '25: 23rd ACM Conference on Embedded Networked Sensor Systems","location":"UC Irvine Student Center. Irvine CA USA","acronym":"SenSys '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","SIGMETRICS ACM Special Interest Group on Measurement and Evaluation","SIGOPS ACM Special Interest Group on Operating Systems","SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 23rd ACM Conference on Embedded Networked Sensor Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3715014.3722068","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:56:51Z","timestamp":1750298211000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3715014.3722068"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,6]]},"references-count":76,"alternative-id":["10.1145\/3715014.3722068","10.1145\/3715014"],"URL":"https:\/\/doi.org\/10.1145\/3715014.3722068","relation":{},"subject":[],"published":{"date-parts":[[2025,5,6]]},"assertion":[{"value":"2025-05-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}