{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T13:00:54Z","timestamp":1785502854746,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","funder":[{"name":"National Key R&#x5c;&#x5c;&amp;D Program of China","award":["2022ZD0119100"],"award-info":[{"award-number":["2022ZD0119100"]}]},{"name":"China NSF grant","award":["62472278"],"award-info":[{"award-number":["62472278"]}]},{"name":"China NSF grant","award":["62432007"],"award-info":[{"award-number":["62432007"]}]},{"name":"China NSF grant","award":["62441236"],"award-info":[{"award-number":["62441236"]}]},{"name":"China NSF grant","award":["62025204"],"award-info":[{"award-number":["62025204"]}]},{"name":"China NSF grant","award":["62332014"],"award-info":[{"award-number":["62332014"]}]},{"name":"China NSF grant","award":["62332013"],"award-info":[{"award-number":["62332013"]}]},{"name":"China NSF grant","award":["U25A6024"],"award-info":[{"award-number":["U25A6024"]}]},{"name":"China NSF grant","award":["62225202"],"award-info":[{"award-number":["62225202"]}]},{"name":"Fundamental and Interdisciplinary Disciplines Breakthrough Plan of the Ministry of Education of China","award":["JYB2025XDXM103"],"award-info":[{"award-number":["JYB2025XDXM103"]}]},{"name":"Tencent Rhino Bird Key Research Project","award":["-"],"award-info":[{"award-number":["-"]}]},{"name":"Shanghai QiYuan Innovation Foundation","award":["-"],"award-info":[{"award-number":["-"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,9]]},"DOI":"10.1145\/3770854.3780181","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:07:40Z","timestamp":1785499660000},"page":"128-139","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Vistar: Enhancing the Perception Capability of LLMs under Imprecise IMU-Text Alignment"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8954-9109","authenticated-orcid":false,"given":"Yatong","family":"Chen","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0713-4393","authenticated-orcid":false,"given":"Chenzhi","family":"Hu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3334-5901","authenticated-orcid":false,"given":"Bowen","family":"He","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1368-1688","authenticated-orcid":false,"given":"Ruijie","family":"Wang","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0710-0963","authenticated-orcid":false,"given":"Xiaomin","family":"Ouyang","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7643-7239","authenticated-orcid":false,"given":"Shengzhong","family":"Liu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5152-0055","authenticated-orcid":false,"given":"Jianxin","family":"Li","sequence":"additional","affiliation":[{"name":"Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0965-9058","authenticated-orcid":false,"given":"Fan","family":"Wu","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6934-1685","authenticated-orcid":false,"given":"Guihai","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,20]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.922"},{"key":"e_1_3_2_2_2_1","volume-title":"arXiv preprint arXiv:2502.13923","author":"Bai Shuai","year":"2025","unstructured":"Shuai Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Sibo Song, Kai Dang, Peng Wang, Shijie Wang, Jun Tang, Humen Zhong, Yuanzhi Zhu, Mingkun Yang, Zhaohai Li, Jianqiang Wan, Pengfei Wang, Wei Ding, Zheren Fu, Yiheng Xu, Jiabo Ye, Xi Zhang, Tianbao Xie, Zesen Cheng, Hang Zhang, Zhibo Yang, Haiyang Xu, and Junyang Lin. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_2_3_1","volume-title":"Language and myth","author":"Cassirer Ernst","unstructured":"Ernst Cassirer and Ernst Alfred Cassirer. 1946. Language and myth. Vol. 51. Courier Corporation."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3699779"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599533"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671470"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2023.ACL-LONG.99"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01834"},{"key":"e_1_3_2_2_11_1","volume-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 2: Short Papers). 595-599","author":"Guo Quan","year":"2025","unstructured":"Quan Guo and Xin Liang. 2025. Transform Retrieval for Textual Entailment in RAG. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 2: Short Papers). 595-599."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i1.32004"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3729485"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02238"},{"key":"e_1_3_2_2_16_1","volume-title":"Why is there so much more research on vision than on any other sensory modality? Frontiers in psychology","author":"Hutmacher Fabian","year":"2019","unstructured":"Fabian Hutmacher. 2019. Why is there so much more research on vision than on any other sensory modality? Frontiers in psychology, Vol. 10 (2019), 481030."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714853"},{"key":"e_1_3_2_2_18_1","volume-title":"MMAct: A Large-Scale Dataset for Cross Modal Human Action Understanding. In The IEEE International Conference on Computer Vision (ICCV).","author":"Kong Quan","year":"2019","unstructured":"Quan Kong, Ziming Wu, Ziwei Deng, Martin Klinkigt, Bin Tong, and Tomokazu Murakami. 2019. MMAct: A Large-Scale Dataset for Cross Modal Human Action Understanding. In The IEEE International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3594738.3611361"},{"key":"e_1_3_2_2_20_1","unstructured":"Patrick Lewis Ethan Perez Aleksandra Piktus Fabio Petroni Vladimir Karpukhin Naman Goyal Heinrich K\u00fcttler Mike Lewis Wen-tau Yih Tim Rockt\u00e4schel et al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. Advances in neural information processing systems Vol. 33 (2020) 9459-9474."},{"key":"e_1_3_2_2_21_1","unstructured":"Bo Li Yuanhan Zhang Dong Guo Renrui Zhang Feng Li Hao Zhang Kaichen Zhang Peiyuan Zhang Yanwei Li Ziwei Liu and Chunyuan Li. 2024b. LLaVA-OneVision: Easy Visual Task Transfer. arXiv:2408.03326 [cs.CV] https:\/\/arxiv.org\/abs\/2408.03326"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00693"},{"key":"e_1_3_2_2_24_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Liu Shengzhong","year":"2023","unstructured":"Shengzhong Liu, Tomoyoshi Kimura, Dongxin Liu, Ruijie Wang, Jinyang Li, Suhas Diggavi, Mani Srivastava, and Tarek Abdelzaher. 2023. FOCAL: Contrastive learning for multimodal time-series sensing signals in factorized orthogonal latent space. Advances in Neural Information Processing Systems, Vol. 36 (2023)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW65960.2025.00111"},{"key":"e_1_3_2_2_27_1","volume-title":"Iot-lm: Large multisensory language models for the internet of things. arXiv preprint arXiv:2407.09801","author":"Mo Shentong","year":"2024","unstructured":"Shentong Mo, Russ Salakhutdinov, Louis-Philippe Morency, and Paul Pu Liang. 2024. Iot-lm: Large multisensory language models for the internet of things. arXiv preprint arXiv:2407.09801 (2024)."},{"key":"e_1_3_2_2_28_1","volume-title":"IMU2CLIP: Multimodal Contrastive Learning for IMU Motion Sensors from Egocentric Videos and Text. arXiv preprint arXiv:2210.14395","author":"Moon Seungwhan","year":"2022","unstructured":"Seungwhan Moon, Andrea Madotto, Zhaojiang Lin, Alireza Dirafzoon, Aparajita Saraf, Amy Bearman, and Babak Damavandi. 2022. IMU2CLIP: Multimodal Contrastive Learning for IMU Motion Sensors from Egocentric Videos and Text. arXiv preprint arXiv:2210.14395 (2022)."},{"key":"e_1_3_2_2_29_1","unstructured":"Girish Narayanswamy Xin Liu Kumar Ayush Yuzhe Yang Xuhai Xu Shun Liao Jake Garrison Shyam Tailor Jake Sunshine Yun Liu et al. 2024. Scaling Wearable Foundation Models. arXiv preprint arXiv:2410.13638 (2024)."},{"key":"e_1_3_2_2_30_1","volume-title":"LLMSense: Harnessing LLMs for High-level Reasoning Over Spatiotemporal Sensor Traces. In 2024 IEEE 3rd Workshop on Machine Learning on Edge in Sensor Systems (SenSys-ML). IEEE, 9-14","author":"Ouyang Xiaomin","year":"2024","unstructured":"Xiaomin Ouyang and Mani Srivastava. 2024. LLMSense: Harnessing LLMs for High-level Reasoning Over Spatiotemporal Sensor Traces. In 2024 IEEE 3rd Workshop on Machine Learning on Edge in Sensor Systems (SenSys-ML). IEEE, 9-14."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715014.3722053"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.adp7821"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.57"},{"key":"e_1_3_2_2_34_1","volume-title":"International Conference on Machine Learning (ICML).","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01850"},{"key":"e_1_3_2_2_36_1","volume-title":"Dynamic programming algorithm optimization for spoken word recognition","author":"Sakoe Hiroaki","year":"2003","unstructured":"Hiroaki Sakoe and Seibi Chiba. 2003. Dynamic programming algorithm optimization for spoken word recognition. IEEE transactions on acoustics, speech, and signal processing, Vol. 26, 1 (2003), 43-49."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/MIPR62202.2024.00031"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671673"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671633"},{"key":"e_1_3_2_2_40_1","unstructured":"Qwen Team. 2025. Qwen3 Technical Report. arXiv:2505.09388 [cs.CL] https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_2_2_41_1","volume-title":"Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems","author":"Tong Zhan","year":"2022","unstructured":"Zhan Tong, Yibing Song, Jue Wang, and Limin Wang. 2022. Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in neural information processing systems, Vol. 35 (2022), 10078-10093."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3690640"},{"key":"e_1_3_2_2_43_1","volume-title":"Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. Advances in neural information processing systems","author":"Wang Wenhui","year":"2020","unstructured":"Wenhui Wang, Furu Wei, Li Dong, Hangbo Bao, Nan Yang, and Ming Zhou. 2020. Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. Advances in neural information processing systems, Vol. 33 (2020), 5776-5788."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.3389\/fcomm.2024.1352252"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671974"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611830"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539388"},{"key":"e_1_3_2_2_48_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Zhang Renrui","year":"2024","unstructured":"Renrui Zhang, Jiaming Han, Chris Liu, Aojun Zhou, Pan Lu, Yu Qiao, Hongsheng Li, and Peng Gao. 2024a. LLaMA-adapter: Efficient fine-tuning of large language models with zero-initialized attention. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3414"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"crossref","unstructured":"Xiang Zhang Ziyuan Zhao Theodoros Tsiligkaridis and Marinka Zitnik. 2022b. Self-Supervised Contrastive Pre-Training For Time Series via Time-Frequency Consistency. In Neural Information Processing Systems (NeurIPS).","DOI":"10.52202\/068431-0288"},{"key":"e_1_3_2_2_51_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Zheng Huaixiu Steven","year":"2024","unstructured":"Huaixiu Steven Zheng, Swaroop Mishra, Xinyun Chen, Heng-Tze Cheng, Ed H. Chi, Quoc V. Le, and Denny Zhou. 2024. Take a Step Back: Evoking Reasoning via Abstraction in Large Language Models. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=3bq3jsvcQ1"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3711896.3737169"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.783"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3679582"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.590"}],"event":{"name":"KDD '26: The 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Jeju Island Republic of Korea","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770854.3780181","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:10:19Z","timestamp":1785499819000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770854.3780181"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":55,"alternative-id":["10.1145\/3770854.3780181","10.1145\/3770854"],"URL":"https:\/\/doi.org\/10.1145\/3770854.3780181","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]},"assertion":[{"value":"2026-04-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}