{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T03:22:40Z","timestamp":1783740160812,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":102,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100006374","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CAREER IIS-2338878"],"award-info":[{"award-number":["CAREER IIS-2338878"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]},{"name":"NEC Labs America Inc."},{"name":"Morgan Stanley"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3736567","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T20:52:41Z","timestamp":1754254361000},"page":"6043-6053","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":36,"title":["Multi-modal Time Series Analysis: A Tutorial and Survey"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4226-7534","authenticated-orcid":false,"given":"Yushan","family":"Jiang","sequence":"first","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7390-5874","authenticated-orcid":false,"given":"Kanghui","family":"Ning","sequence":"additional","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4893-586X","authenticated-orcid":false,"given":"Zijie","family":"Pan","sequence":"additional","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2102-796X","authenticated-orcid":false,"given":"Xuyang","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2986-6612","authenticated-orcid":false,"given":"Jingchao","family":"Ni","sequence":"additional","affiliation":[{"name":"University of Houston, Houston, TX, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2480-448X","authenticated-orcid":false,"given":"Wenchao","family":"Yu","sequence":"additional","affiliation":[{"name":"NEC Laboratories America, Princeton, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6539-6882","authenticated-orcid":false,"given":"Anderson","family":"Schneider","sequence":"additional","affiliation":[{"name":"Morgan Stanley, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9363-738X","authenticated-orcid":false,"given":"Haifeng","family":"Chen","sequence":"additional","affiliation":[{"name":"NEC Laboratories America, Princeton, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3484-7483","authenticated-orcid":false,"given":"Yuriy","family":"Nevmyvaka","sequence":"additional","affiliation":[{"name":"Morgan Stanley, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7027-7916","authenticated-orcid":false,"given":"Dongjin","family":"Song","sequence":"additional","affiliation":[{"name":"University of Connecticut, Storrs, CT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Joon Son Chung, and Andrew Zisserman","author":"Afouras Triantafyllos","year":"2018","unstructured":"Triantafyllos Afouras, Joon Son Chung, and Andrew Zisserman. 2018. LRS3-TED: a large-scale dataset for visual speech recognition. arxiv:1809.00496 [cs.CV] https:\/\/arxiv.org\/abs\/1809.00496"},{"key":"e_1_3_2_1_2_1","volume-title":"Multimodal machine learning: A survey and taxonomy","author":"Baltru\u0161aitis Tadas","year":"2018","unstructured":"Tadas Baltru\u0161aitis, Chaitanya Ahuja, and Louis-Philippe Morency. 2018. Multimodal machine learning: A survey and taxonomy. IEEE transactions on pattern analysis and machine intelligence, Vol. 41, 2 (2018), 423-443."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3604237.3626901"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-023-02266-0"},{"key":"e_1_3_2_1_5_1","volume-title":"James Fackler, and Kimia Ghobadi.","author":"Chan Nimeesha","year":"2024","unstructured":"Nimeesha Chan, Felix Parker, William Bennett, Tianyi Wu, Mung Yao Jia, James Fackler, and Kimia Ghobadi. 2024. Medtsllm: Leveraging llms for multimodal medical time series analysis. arXiv preprint arXiv:2408.07773(2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"Shubhankar Agarwal, and Sandeep P. Chinchali.","author":"Chattopadhyay Sameep","year":"2025","unstructured":"Sameep Chattopadhyay, Pulkit Paliwal, Sai Shankar Narasimhan, Shubhankar Agarwal, and Sandeep P. Chinchali. 2025. Context Matters: Leveraging Contextual Features for Time Series Forecasting. arxiv:2410.12672 [cs.LG] https:\/\/arxiv.org\/abs\/2410.12672"},{"key":"e_1_3_2_1_7_1","unstructured":"Jialin Chen Aosong Feng Ziyu Zhao Juan Garza Gaukhar Nurbek Cheng Qin Ali Maatouk Leandros Tassiulas Yifeng Gao and Rex Ying. 2025. MTBench: A Multimodal Time Series Benchmark for Temporal Reasoning and Question Answering. arXiv preprint arXiv:2503.16858(2025)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611977653.ch25"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539115"},{"key":"e_1_3_2_1_10_1","volume-title":"Nathanael Christian Yoder, and Tomas Pfister","author":"Chen Si-An","year":"2023","unstructured":"Si-An Chen, Chun-Liang Li, Sercan O Arik, Nathanael Christian Yoder, and Tomas Pfister. 2023a. TSMixer: An All-MLP Architecture for Time Series Forecast-ing. Transactions on Machine Learning Research(2023)."},{"key":"e_1_3_2_1_11_1","first-page":"66329","volume-title":"Zhang(Eds.)","volume":"37","author":"Chen Wei","year":"2024","unstructured":"Wei Chen, Xixuan Hao, Yuankai Wu, and Yuxuan Liang. 2024. Terra: A Multimodal Spatio-Temporal Dataset Spanning the Earth. In Advances in Neural Information Processing Systems, A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang(Eds.), Vol. 37. Curran Associates, Inc., 66329-66356. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/file\/7a6a7fbd1ee0c9684b3f919f79d129ef-Paper-Datasets_and_Benchmarks_Track.pdf"},{"key":"e_1_3_2_1_12_1","volume-title":"Cheng Lu, Jialu Yuan, and Di Zhu.","author":"Chen Zihan","year":"2023","unstructured":"Zihan Chen, Lei Nico Zheng, Cheng Lu, Jialu Yuan, and Di Zhu. 2023c. ChatGPT Informed Graph Neural Network for Stock Movement Prediction. Available at SSRN 4464002(2023)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701551.3703499"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-019-0031-8"},{"key":"e_1_3_2_1_16_1","volume-title":"Marc Wilson, Desislav Ivanov, Mikhail Papkov, Eva Schnider, Jing Tang, Kay Lamerigts, Gabriela Botea, Michael A Sanchez, et al.","author":"Daswani Mayank","year":"2024","unstructured":"Mayank Daswani, Mathias MJ Bellaiche, Marc Wilson, Desislav Ivanov, Mikhail Papkov, Eva Schnider, Jing Tang, Kay Lamerigts, Gabriela Botea, Michael A Sanchez, et al., 2024. Plots Unlock Time-Series Understanding in Multimodal Models. arXiv preprint arXiv:2410.02637(2024)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-acl.352"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671629"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403362"},{"key":"e_1_3_2_1_20_1","volume-title":"Citygpt: Empowering urban spatial cognition of large language models. arXiv preprint arXiv:2406.13948(2024).","author":"Feng Jie","year":"2024","unstructured":"Jie Feng, Yuwei Du, Tianhui Liu, Siqi Guo, Yuming Lin, and Yong Li. 2024. Citygpt: Empowering urban spatial cognition of large language models. arXiv preprint arXiv:2406.13948(2024)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41591-024-02902-1"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.dib.2021.106913"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuroimage.2022.119754"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.3301922"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.commtr.2024.100150"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02291"},{"key":"e_1_3_2_1_27_1","unstructured":"Nora Hollenstein Marius Troendle Ce Zhang and Nicolas Langer. 2020. ZuCo 2.0: A Dataset of Physiological Recordings During Natural Reading and Annotation. In Proceedings of the Twelfth Language Resources and Evaluation Conference Nicoletta Calzolari Fr\u00e9d\u00e9ric B\u00e9chet Philippe Blache Khalid Choukri Christopher Cieri Thierry Declerck Sara Goggi Hitoshi Isahara Bente Maegaard Joseph Mariani H\u00e9l\u00e8ne Mazo Asuncion Moreno Jan Odijk and Stelios Piperidis(Eds.). European Language Resources Association Marseille France 138-146. https:\/\/aclanthology.org\/2020.lrec-1.18\/"},{"key":"e_1_3_2_1_28_1","unstructured":"Shi Bin Hoo Samuel M\u00fcller David Salinas and Frank Hutter. 2025. The tabular foundation model TabPFN outperforms specialized time series forecasting models based on simple features. arXiv preprint arXiv:2501.02945(2025)."},{"key":"e_1_3_2_1_29_1","unstructured":"Kexin Huang Jaan Altosaar and Rajesh Ranganath. 2020. ClinicalBERT: Modeling Clinical Notes and Predicting Hospital Readmission. arxiv:1904.05342 [cs.CL] https:\/\/arxiv.org\/abs\/1904.05342"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i4.25555"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i21.30383"},{"key":"e_1_3_2_1_32_1","unstructured":"Yushan Jiang Wenchao Yu Geon Lee Dongjin Song Kijung Shin Wei Cheng Yanchi Liu and Haifeng Chen. 2025. Explainable Multi-modal Time Series Prediction with LLM-in-the-Loop. arxiv:2503.01013 [cs.LG] https:\/\/arxiv.org\/abs\/2503.01013"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220095"},{"key":"e_1_3_2_1_34_1","volume-title":"The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Unb5CVPtae","author":"Jin Ming","year":"2024","unstructured":"Ming Jin, Shiyu Wang, Lintao Ma, Zhixuan Chu, James Y. Zhang, Xiaoming Shi, Pin-Yu Chen, Yuxuan Liang, Yuan-Fang Li, Shirui Pan, and Qingsong Wen. 2024. Time-LLM: Time Series Forecasting by Reprogramming Large Language Models. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Unb5CVPtae"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.13026\/s6n6-xd98"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1038\/sdata.2016.35"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1678"},{"key":"e_1_3_2_1_38_1","unstructured":"Kai Kim Howard Tsai Rajat Sen Abhimanyu Das Zihao Zhou Abhishek Tanpure Mathew Luo and Rose Yu. 2024. Multi-Modal Forecaster: Jointly Predicting Time Series and Textual Data. arxiv:2411.06735 [cs.AI] https:\/\/arxiv.org\/abs\/2411.06735"},{"key":"e_1_3_2_1_39_1","unstructured":"Yaxuan Kong Yiyuan Yang Yoontae Hwang Wenjie Du Stefan Zohren Zhangyang Wang Ming Jin and Qingsong Wen. 2025 a. Time-MQA: Time Series Multi-Task Question Answering with Context Enhancement. arxiv:2503.01875 [cs.CL] https:\/\/arxiv.org\/abs\/2503.01875"},{"key":"e_1_3_2_1_40_1","unstructured":"Yaxuan Kong Yiyuan Yang Shiyu Wang Chenghao Liu Yuxuan Liang Ming Jin Stefan Zohren Dan Pei Yan Liu and Qingsong Wen. 2025 b. Position: Empowering Time Series Reasoning with Multimodal LLMs. arXiv preprint arXiv:2502.01477(2025)."},{"key":"e_1_3_2_1_41_1","volume-title":"Zhengzhang Chen and Haifeng Chen","author":"Chengyuan Deng Reon Matsuoka Dongjie Wang","year":"2024","unstructured":"Dongjie Wang Chengyuan Deng Reon Matsuoka Lecheng Zheng, Zhengzhang Chen and Haifeng Chen. 2024. LEMMA-RCA: A Large Multi-modal Multi-domain Dataset for Root Cause Analysis. arxiv:2406.05375 [cs.AI]"},{"key":"e_1_3_2_1_42_1","unstructured":"Geon Lee Wenchao Yu Wei Cheng and Haifeng Chen. 2024. MoAT: Multi-Modal Augmented Time Series Forecasting. https:\/\/openreview.net\/forum?id=uRXxnoqDHH"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Geon Lee Wenchao Yu Kijung Shin Wei Cheng and Haifeng Chen. 2025. TimeCAP: Learning to Contextualize Augment and Predict Time Series Events with Large Language Model Agents. In AAAI.","DOI":"10.1609\/aaai.v39i17.33989"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btz682"},{"key":"e_1_3_2_1_45_1","volume-title":"LITE: Modeling Environmental Ecosystems with Multimodal Large Language Models. arXiv preprint arXiv:2404.01165(2024).","author":"Li Haoran","year":"2024","unstructured":"Haoran Li, Junqi Liu, Zexian Wang, Shiyuan Luo, Xiaowei Jia, and Huaxiu Yao. 2024b. LITE: Modeling Environmental Ecosystems with Multimodal Large Language Models. arXiv preprint arXiv:2404.01165(2024)."},{"key":"e_1_3_2_1_46_1","unstructured":"Jun Li Che Liu Sibo Cheng Rossella Arcucci and Shenda Hong. 2023b. Frozen Language Model Helps ECG Zero-Shot Learning. In Medical Imaging with Deep Learning."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.63317\/5ncmjfm6wkbo"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2139"},{"key":"e_1_3_2_1_49_1","unstructured":"Zihao Li Xiao Lin Zhining Liu Jiaru Zou Ziwei Wu Lecheng Zheng Dongqi Fu Yada Zhu Hendrik Hamann Hanghang Tong and Jingrui He. 2025. Language in the Flow of Time: Time-Series-Paired Texts Weaved into a Unified Temporal Narrative. arxiv:2502.08942 [cs.LG] https:\/\/arxiv.org\/abs\/2502.08942"},{"key":"e_1_3_2_1_50_1","volume-title":"UrbanGPT: Spatio-Temporal Large Language Models. ArXiv","author":"Li Zhonghang","year":"2024","unstructured":"Zhonghang Li, Lianghao Xia, Jiabin Tang, Yong Xu, Lei Shi, Long Xia, Dawei Yin, and Chao Huang. 2024c. UrbanGPT: Spatio-Temporal Large Language Models. ArXiv, Vol. abs\/2403.00813 (2024). https:\/\/api.semanticscholar.org\/CorpusID:268230972"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3656580"},{"key":"e_1_3_2_1_52_1","unstructured":"Minhua Lin Zhengzhang Chen Yanchi Liu Xujiang Zhao Zongyu Wu Junxiang Wang Xiang Zhang Suhang Wang and Haifeng Chen. 2024. Decoding Time Series with LLMs: A Multi-Agent Framework for Cross-Domain Annotation. arxiv:2410.17462 [cs.AI] https:\/\/arxiv.org\/abs\/2410.17462"},{"key":"e_1_3_2_1_53_1","unstructured":"Chenxi Liu Qianxiong Xu Hao Miao Sun Yang Lingzheng Zhang Cheng Long Ziyue Li and Rui Zhao. 2025. TimeCMA: Towards LLM-Empowered Multivariate Time Series Forecasting via Cross-Modality Alignment. In AAAI."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigData62323.2024.10825980"},{"key":"e_1_3_2_1_55_1","unstructured":"Haoxin Liu Shangqing Xu Zhiyuan Zhao Lingkai Kong Harshavardhan Kamarthi Aditya B. Sasanur Megha Sharma Jiaming Cui Qingsong Wen Chao Zhang and B. Aditya Prakash. 2024c. Time-MMD: Multi-Domain Multimodal Dataset for Time Series Analysis. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track. https:\/\/openreview.net\/forum?id=fuD0h4R1IL"},{"key":"e_1_3_2_1_56_1","unstructured":"Lei Liu Shuo Yu Runze Wang Zhenxun Ma and Yanming Shen. 2024d. How can large language models understand spatial-temporal data? arXiv preprint arXiv:2401.14192(2024)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645434"},{"key":"e_1_3_2_1_58_1","volume-title":"Paolo Di Achille, and Shwetak Patel","author":"Liu Xin","year":"2023","unstructured":"Xin Liu, Daniel McDuff, Geza Kovacs, Isaac Galatzer-Levy, Jacob Sunshine, Jiening Zhan, Ming-Zher Poh, Shun Liao, Paolo Di Achille, and Shwetak Patel. 2023. Large Language Models are Few-Shot Health Learners. arXiv preprint arXiv:2305.15525(2023). https:\/\/arxiv.org\/abs\/2305.15525"},{"key":"e_1_3_2_1_59_1","unstructured":"Jingchao Ni Ziming Zhao ChengAo Shen Hanghang Tong Dongjin Song Wei Cheng Dongsheng Luo and Haifeng Chen. 2025. Harnessing Vision Models for Time Series Analysis: A Survey. arXiv preprint arXiv:2502.08869(2025)."},{"key":"e_1_3_2_1_60_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Nie Yuqi","year":"2023","unstructured":"Yuqi Nie, Nam H Nguyen, Phanwadee Sinthong, and Jayant Kalagnanam. 2023. A Time Series is Worth 64 Words: Long-term Forecasting with Transformers. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_61_1","unstructured":"Kanghui Ning Zijie Pan Yu Liu Yushan Jiang James Y. Zhang Kashif Rasul Anderson Schneider Lintao Ma Yuriy Nevmyvaka and Dongjin Song. 2025. TS-RAG: Retrieval-Augmented Generation based Time Series Foundation Models are Stronger Zero-Shot Forecaster. arxiv:2503.07649 [cs.LG] https:\/\/arxiv.org\/abs\/2503.07649"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.3389\/fmolb.2023.1136071"},{"key":"e_1_3_2_1_63_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Pan Zijie","year":"2024","unstructured":"Zijie Pan, Yushan Jiang, Sahil Garg, Anderson Schneider, Yuriy Nevmyvaka, and Dongjin Song. 2024. S^2IP-LLM: Semantic Space Informed Prompt Learning with LLM for Time Series Forecasting. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_64_1","unstructured":"Vinay Prithyani Mohsin Mohammed Richa Gadgil Ricardo Buitrago Vinija Jain and Aman Chadha. 2024. On the Feasibility of Vision-Language Models for Time-Series Classification. arXiv preprint arXiv:2412.17304(2024)."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/366"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2020.114332"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1088\/1361-6579\/ab03ea"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00985"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.676"},{"key":"e_1_3_2_1_70_1","unstructured":"ChengAo Shen Zhengzhang Chen Dongsheng Luo Dongkuan Xu Haifeng Chen and Jingchao Ni. 2024. Exploring Multi-Modal Integration with Tool-Augmented LLM Agents for Precise Causal Discovery. arxiv:2412.13667 [cs.LG] https:\/\/arxiv.org\/abs\/2412.13667"},{"key":"e_1_3_2_1_71_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Z1Qlm11uOM","author":"Shi Bowen","year":"2022","unstructured":"Bowen Shi, Wei-Ning Hsu, Kushal Lakhotia, and Abdelrahman Mohamed. 2022. Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Z1Qlm11uOM"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1002\/for.3104"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-020-0495-6"},{"key":"e_1_3_2_1_74_1","unstructured":"Jiahao Wang Mingyue Cheng Qingyang Mao Yitong Zhou Feiyang Xu and Xin Li. 2025. TableTime: Reformulating Time Series Classification as Training-Free Table Understanding with Large Language Models. arxiv:2411.15737 [cs.AI] https:\/\/arxiv.org\/abs\/2411.15737"},{"key":"e_1_3_2_1_75_1","volume-title":"Advances in Neural Information Processing Systems","author":"Wang Xinlei","year":"2024","unstructured":"Xinlei Wang, Maike Feng, Jing Qiu, JINJIN GU, and Junhua Zhao. 2024. From News to Forecast: Integrating Event Analysis in LLM-Based Time Series Forecasting with Reflection. In Advances in Neural Information Processing Systems, A. Globerson, L. Mackey, D. Belgrave, A. Fan, U. Paquet, J. Tomczak, and C. Zhang(Eds.), Vol. 37. Curran Associates, Inc., 58118-58153. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/file\/6aef8bffb372096ee73d98da30119f89-Paper-Conference.pdf"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.171"},{"key":"e_1_3_2_1_77_1","volume-title":"Open Vocabulary Electroencephalography-To-Text Decoding and Zero-shot Sentiment Classification. In AAAI Conference on Artificial Intelligence. https:\/\/api.semanticscholar.org\/CorpusID:244909027","author":"Wang Zhenhailong","year":"2021","unstructured":"Zhenhailong Wang and Heng Ji. 2021. Open Vocabulary Electroencephalography-To-Text Decoding and Zero-shot Sentiment Classification. In AAAI Conference on Artificial Intelligence. https:\/\/api.semanticscholar.org\/CorpusID:244909027"},{"key":"e_1_3_2_1_78_1","unstructured":"Andrew Robert Williams Arjun Ashok \u00c9tienne Marcotte Valentina Zantedeschi Jithendaraa Subramanian Roland Riachi James Requeima Alexandre Lacoste Irina Rish Nicolas Chapados et al. 2024. Context is key: A benchmark for forecasting with essential textual information. arXiv preprint arXiv:2410.18959(2024)."},{"key":"e_1_3_2_1_79_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Wu Haixu","year":"2023","unstructured":"Haixu Wu, Tengge Hu, Yong Liu, Hang Zhou, Jianmin Wang, and Mingsheng Long. 2023. TimesNet: Temporal 2D-Variation Modeling for General Time Series Analysis. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3269290"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403118"},{"key":"e_1_3_2_1_82_1","unstructured":"Qianqian Xie Weiguang Han Yanzhao Lai Min Peng and Jimin Huang. 2023. The wall street neophyte: A zero-shot analysis of chatgpt over multimodal stock movement prediction challenges. arXiv preprint arXiv:2304.05351(2023)."},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijepes.2022.108567"},{"key":"e_1_3_2_1_84_1","unstructured":"Haojun Xu Yan Gao Zheng Hui Jie Li and Xinbo Gao. 2023. Language Knowledge-Assisted Representation Learning for Skeleton-Based Action Recognition. arxiv:2305.12398 [cs.CV] https:\/\/arxiv.org\/abs\/2305.12398"},{"key":"e_1_3_2_1_85_1","unstructured":"Wenyan Xu Dawei Xiang Yue Liu Xiyu Wang Yanxiang Ma Liang Zhang Chang Xu and Jiaheng Zhang. 2025. FinMultiTime: A Four-Modal Bilingual Dataset for Financial Time-Series Analysis. arxiv:2506.05019 [cs.CE] https:\/\/arxiv.org\/abs\/2506.05019"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1183"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.329"},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1186\/s13326-021-00235-3"},{"key":"e_1_3_2_1_89_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Yi Kun","year":"2024","unstructured":"Kun Yi, Qi Zhang, Wei Fan, Shoujin Wang, Pengyang Wang, Hui He, Ning An, Defu Lian, Longbing Cao, and Zhendong Niu. 2024. Frequency-domain MLPs are more effective learners in time series forecasting. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-industry.69"},{"key":"e_1_3_2_1_91_1","unstructured":"Dong Zhang Shimin Li Xin Zhang Jun Zhan Pengyu Wang Yaqian Zhou and Xipeng Qiu. 2023. SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities. arxiv:2305.11000 [cs.CL] https:\/\/arxiv.org\/abs\/2305.11000"},{"key":"e_1_3_2_1_92_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3369699"},{"key":"e_1_3_2_1_93_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098117"},{"key":"e_1_3_2_1_94_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i17.17761"},{"key":"e_1_3_2_1_95_1","doi-asserted-by":"publisher","DOI":"10.1109\/JBHI.2020.2971610"},{"key":"e_1_3_2_1_96_1","unstructured":"Yuwei Zhang Tong Xia Aaqib Saeed and Cecilia Mascolo. 2024b. RespLLM: Unifying Audio and Text with Multimodal LLMs for Generalized Respiratory Health Prediction. arxiv:2410.05361 [cs.LG] https:\/\/arxiv.org\/abs\/2410.05361"},{"key":"e_1_3_2_1_97_1","volume-title":"VIMTS: Variational-based Imputation for Multi-modal Time Series. In 2022 IEEE International Conference on Big Data (Big Data). IEEE, 349-358","author":"Zhao Xiaohu","year":"2022","unstructured":"Xiaohu Zhao, Kebin Jia, Benjamin Letcher, Jennifer Fair, Yiqun Xie, and Xiaowei Jia. 2022. VIMTS: Variational-based Imputation for Multi-modal Time Series. In 2022 IEEE International Conference on Big Data (Big Data). IEEE, 349-358."},{"key":"e_1_3_2_1_98_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645442"},{"key":"e_1_3_2_1_99_1","unstructured":"Siru Zhong Weilin Ruan Ming Jin Huan Li Qingsong Wen and Yuxuan Liang. 2025. Time-VLM: Exploring Multimodal Vision-Language Models for Augmented Time Series Forecasting. arxiv:2502.04395 [cs.CV] https:\/\/arxiv.org\/abs\/2502.04395"},{"key":"e_1_3_2_1_100_1","unstructured":"Xin Zhou Weiqing Wang Francisco J Bald\u00e1n Wray Buntine and Christoph Bergmeir. 2025. MoTime: A Dataset Suite for Multimodal Time Series Forecasting. arXiv preprint arXiv:2505.15072(2025)."},{"key":"e_1_3_2_1_101_1","unstructured":"Zihao Zhou and Rose Yu. 2024. Can LLMs Understand Time Series Anomalies? arXiv preprint arXiv:2410.05440(2024)."},{"key":"e_1_3_2_1_102_1","volume-title":"Think it","author":"Zhuang Jiaxin","year":"2024","unstructured":"Jiaxin Zhuang, Leon Yan, Zhenwei Zhang, Ruiqi Wang, Jiawei Zhang, and Yuantao Gu. 2024. See it, Think it, Sorted: Large Multimodal Models are Few-shot Time Series Anomaly Analyzers. arXiv preprint arXiv:2411.02465(2024)."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3736567","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:59:03Z","timestamp":1777571943000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3736567"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":102,"alternative-id":["10.1145\/3711896.3736567","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3736567","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}