{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T15:54:11Z","timestamp":1784390051456,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T00:00:00Z","timestamp":1732233600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,22]]},"DOI":"10.1145\/3698587.3701364","type":"proceedings-article","created":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T10:05:08Z","timestamp":1734343508000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["TimelyGPT: Extrapolatable Transformer Pre-training for Long-term Time-Series Forecasting in Healthcare"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-2450-3271","authenticated-orcid":false,"given":"Ziyang","family":"Song","sequence":"first","affiliation":[{"name":"School of Computer Science, McGill University, Montreal, QC, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6151-6687","authenticated-orcid":false,"given":"Qincheng","family":"Lu","sequence":"additional","affiliation":[{"name":"School of Computer Science, McGill University, Montreal, QC, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9699-384X","authenticated-orcid":false,"given":"Hao","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Computer Science, McGill University, Montreal, QC, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0617-7889","authenticated-orcid":false,"given":"He","family":"Zhu","sequence":"additional","affiliation":[{"name":"School of Computer Science, McGill University, Montreal, QC, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1817-5047","authenticated-orcid":false,"given":"David","family":"Buckeridge","sequence":"additional","affiliation":[{"name":"School of Population and Global Health, McGill University, Montreal, Quebec, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1163-3634","authenticated-orcid":false,"given":"Yue","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer Science, McGill University, Mila - Quebec AI Institute, Montreal, QC, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,12,16]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbi.2022.104190"},{"key":"e_1_3_2_2_2_1","volume-title":"Xception: Deep Learning with Depthwise Separable Convolutions. 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","author":"Chollet Fran\u00e7ois","year":"2016","unstructured":"Fran\u00e7ois Chollet. 2016. Xception: Deep Learning with Depthwise Separable Convolutions. 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016), 1800--1807. https:\/\/api.semanticscholar.org\/CorpusID:2375110"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1038\/nbt.2749"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btq126"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/324"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1161\/01.CIR.101.23.e215"},{"key":"e_1_3_2_2_8_1","volume-title":"Efficiently Modeling Long Sequences with Structured State Spaces. In The International Conference on Learning Representations (ICLR).","author":"Gu Albert","year":"2022","unstructured":"Albert Gu, Karan Goel, and Christopher R\u00e9. 2022. Efficiently Modeling Long Sequences with Structured State Spaces. In The International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"e_1_3_2_2_10_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research","author":"Horn Max","year":"2020","unstructured":"Max Horn, Michael Moor, Christian Bock, Bastian Rieck, and Karsten Borgwardt. 2020. Set Functions for Time Series. In Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 119), Hal Daum\u00e9 III and Aarti Singh (Eds.). PMLR, 4353--4363."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i4.25556"},{"key":"e_1_3_2_2_12_1","unstructured":"Jared Kaplan Sam McCandlish Tom Henighan Tom B. Brown Benjamin Chess Rewon Child Scott Gray Alec Radford Jeffrey Wu and Dario Amodei. 2020. Scaling Laws for Neural Language Models. arXiv:2001.08361 [cs.LG]"},{"key":"e_1_3_2_2_13_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning (ICML'20)","author":"Katharopoulos Angelos","year":"2020","unstructured":"Angelos Katharopoulos, Apoorv Vyas, Nikolaos Pappas, and Fran\u00e7ois Fleuret. 2020. Transformers are RNNs: fast autoregressive transformers with linear attention. In Proceedings of the 37th International Conference on Machine Learning (ICML'20). JMLR.org, Article 478, 10 pages."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/10.867928"},{"key":"e_1_3_2_2_15_1","series-title":"and Time Series","volume-title":"Convolutional Networks for Images, Speech","author":"LeCun Yann","unstructured":"Yann LeCun and Yoshua Bengio. 1998. Convolutional Networks for Images, Speech, and Time Series. MIT Press, Cambridge, MA, USA, 255--258."},{"key":"e_1_3_2_2_16_1","unstructured":"Zhe Li Shiyi Qi Yiduo Li and Zenglin Xu. 2023. Revisiting Long-term Time Series Forecasting: An Investigation on Linear Mapping. arXiv:2305.10721 [cs.LG]"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2024.3475809"},{"key":"e_1_3_2_2_18_1","volume-title":"International Conference on Learning Representations.","author":"Nie Yuqi","year":"2023","unstructured":"Yuqi Nie, Nam H. Nguyen, Phanwadee Sinthong, and Jayant Kalagnanam. 2023. A Time Series is Worth 64 Words: Long-term Forecasting with Transformers. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414142"},{"key":"e_1_3_2_2_20_1","unstructured":"Guilherme Penedo Quentin Malartic Daniel Hesslow Ruxandra Cojocaru Alessandro Cappelli Hamza Alobeidli Baptiste Pannier Ebtesam Almazrouei and Julien Launay. 2023. The RefinedWeb Dataset for Falcon LLM: Outperforming Curated Corpora with Web Data and Web Data Only. arXiv:2306.01116 [cs.CL]"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"crossref","unstructured":"Bo Peng Eric Alcaide Quentin Anthony Alon Albalak Samuel Arcadinho Huanqi Cao Xin Cheng Michael Chung Matteo Grella Kranthi Kiran GV Xuzheng He Haowen Hou Przemyslaw Kazienko Jan Kocon Jiaming Kong Bartlomiej Koptyra Hayden Lau Krishna Sri Ipsit Mantri Ferdinand Mom Atsushi Saito Xiangru Tang Bolun Wang Johan S. Wind Stansilaw Wozniak Ruichong Zhang Zhenyuan Zhang Qihang Zhao Peng Zhou Jian Zhu and Rui-Jie Zhu. 2023. RWKV: Reinventing RNNs for the Transformer Era. arXiv:2305.13048 [cs.CL]","DOI":"10.18653\/v1\/2023.findings-emnlp.936"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2021.3070057"},{"key":"e_1_3_2_2_23_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Poli Michael","year":"2023","unstructured":"Michael Poli, Stefano Massaroli, Eric Nguyen, Daniel Y. Fu, Tri Dao, Stephen Baccus, Yoshua Bengio, Stefano Ermon, and Christopher R\u00e9. 2023. Hyena hierarchy: towards larger convolutional language models. In Proceedings of the 40th International Conference on Machine Learning (Honolulu, Hawaii, USA) (ICML'23). JMLR.org, Article 1164, 36 pages."},{"key":"e_1_3_2_2_24_1","volume-title":"Test Long: Attention with Linear Biases Enables Input Length Extrapolation. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=R8sQPpGCv0","author":"Press Ofir","year":"2022","unstructured":"Ofir Press, Noah Smith, and Mike Lewis. 2022. Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=R8sQPpGCv0"},{"key":"e_1_3_2_2_25_1","volume-title":"Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever.","author":"Radford Alec","year":"2022","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2022. Robust Speech Recognition via Large-Scale Weak Supervision. arXiv:2212.04356 [eess.AS]"},{"key":"e_1_3_2_2_26_1","unstructured":"Alec Radford Jeff Wu Rewon Child David Luan Dario Amodei and Ilya Sutskever. 2019. Language Models are Unsupervised Multitask Learners. https:\/\/api.semanticscholar.org\/CorpusID:160025533"},{"key":"e_1_3_2_2_27_1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J. Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21, 1, Article 140 (Jan. 2020), 67 pages.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_2_28_1","unstructured":"Kashif Rasul Arjun Ashok Andrew Robert Williams Hena Ghonia Rishika Bhagwatkar Arian Khorasani Mohammad Javad Darvishi Bayazi George Adamopoulos Roland Riachi Nadhir Hassen Marin Bilo\u0161 Sahil Garg Anderson Schneider Nicolas Chapados Alexandre Drouin Valentina Zantedeschi Yuriy Nevmyvaka and Irina Rish. 2024. Lag-Llama: Towards Foundation Models for Probabilistic Time Series Forecasting. arXiv:2310.08278 [cs.LG] https:\/\/arxiv.org\/abs\/2310.08278"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.3390\/s19143079"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","unstructured":"Arash Shaban-Nejad Maxime Lavigne Anya Okhmatovskaia and David Buckeridge. 2016. PopHR: a knowledge-based platform to support integration analysis and visualization of population health data: The Population Health Record (PopHR). Annals of the New York Academy of Sciences 1387 (10 2016). https:\/\/doi.org\/10.1111\/nyas.13271","DOI":"10.1111\/nyas.13271"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/N18-2074"},{"key":"e_1_3_2_2_32_1","volume-title":"Multi-Time Attention Networks for Irregularly Sampled Time Series. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=4c0J6lwQ4_","author":"Shukla Satya Narayan","year":"2021","unstructured":"Satya Narayan Shukla and Benjamin Marlin. 2021. Multi-Time Attention Networks for Irregularly Sampled Time Series. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=4c0J6lwQ4_"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNSRE.2022.3230250"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3542675"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1111\/epi.16541"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"e_1_3_2_2_37_1","volume-title":"Retentive Network: A Successor to Transformer for Large Language Models. arXiv:2307.08621 [cs.CL]","author":"Sun Yutao","year":"2023","unstructured":"Yutao Sun, Li Dong, Shaohan Huang, Shuming Ma, Yuqing Xia, Jilong Xue, Jianyong Wang, and Furu Wei. 2023. Retentive Network: A Successor to Transformer for Large Language Models. arXiv:2307.08621 [cs.CL]"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.816"},{"key":"e_1_3_2_2_39_1","volume-title":"International Conference on Learning Representations.","author":"Tang Wensi","year":"2021","unstructured":"Wensi Tang, Guodong Long, Lu Liu, Tianyi Zhou, Michael Blumenstein, and Jing Jiang. 2021. Omni-Scale CNNs: a simple and effective kernel size configuration for time series classification. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_40_1","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems (NIPS '21)","author":"Tolstikhin Ilya","year":"2024","unstructured":"Ilya Tolstikhin, Neil Houlsby, Alexander Kolesnikov, Lucas Beyer, Xiaohua Zhai, Thomas Unterthiner, Jessica Yung, Andreas Steiner, Daniel Keysers, Jakob Uszkoreit, Mario Lucic, and Alexey Dosovitskiy. 2024. MLP-mixer: an all-MLP architecture for vision. In Proceedings of the 35th International Conference on Neural Information Processing Systems (NIPS '21). Curran Associates Inc., Red Hook, NY, USA, Article 1857, 12 pages."},{"key":"e_1_3_2_2_41_1","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arXiv:2302.13971 [cs.CL]"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_2_43_1","volume-title":"TimeMixer: Decomposable Multiscale Mixing for Time Series Forecasting. In International Conference on Learning Representations (ICLR).","author":"Wang Shiyu","year":"2024","unstructured":"Shiyu Wang, Haixu Wu, Xiaoming Shi, Tengge Hu, Huakun Luo, Lintao Ma, James Y Zhang, and JUN ZHOU. 2024. TimeMixer: Decomposable Multiscale Mixing for Time Series Forecasting. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_44_1","volume-title":"International Conference on Learning Representations.","author":"Wu Haixu","year":"2023","unstructured":"Haixu Wu, Tengge Hu, Yong Liu, Hang Zhou, Jianmin Wang, and Mingsheng Long. 2023. TimesNet: Temporal 2D-Variation Modeling for General Time Series Analysis. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_45_1","volume-title":"Autoformer: Decomposition Transformers with Auto-Correlation for Long-Term Series Forecasting. In Advances in Neural Information Processing Systems.","author":"Wu Haixu","year":"2021","unstructured":"Haixu Wu, Jiehui Xu, Jianmin Wang, and Mingsheng Long. 2021. Autoformer: Decomposition Transformers with Auto-Correlation for Long-Term Series Forecasting. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_46_1","volume-title":"Lite Transformer with Long-Short Range Attention. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=ByeMPlHKPH","author":"Zhanghao","year":"2020","unstructured":"Zhanghao Wu*, Zhijian Liu*, Ji Lin, Yujun Lin, and Song Han. 2020. Lite Transformer with Long-Short Range Attention. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=ByeMPlHKPH"},{"key":"e_1_3_2_2_47_1","volume-title":"Annual Symposium proceedings. AMIA Symposium 2017 (04 2018)","author":"Yuan Mengru","year":"2018","unstructured":"Mengru Yuan, Guido Powell, Maxime Lavigne, Anya Okhmatovskaia, and David Buckeridge. 2018. Initial Usability Evaluation of a Knowledge-Based Population Health Information System: The Population Health Record (PopHR). AMIA ... Annual Symposium proceedings. AMIA Symposium 2017 (04 2018), 1878--1884."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20881"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","unstructured":"Ailing Zeng Muxi Chen Lei Zhang and Qiang Xu. 2023. Are transformers effective for time series forecasting?. In Proceedings of the Thirty-Seventh AAAI Conference on Artificial Intelligence and Thirty-Fifth Conference on Innovative Applications of Artificial Intelligence and Thirteenth Symposium on Educational Advances in Artificial Intelligence (AAAI'23\/IAAI'23\/EAAI'23). AAAI Press Article 1248 8 pages. https:\/\/doi.org\/10.1609\/aaai.v37i9.26317","DOI":"10.1609\/aaai.v37i9.26317"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","unstructured":"George Zerveas Srideepika Jayaraman Dhaval Patel Anuradha Bhamidipaty and Carsten Eickhoff. 2021. A Transformer-based Framework for Multivariate Time Series Representation Learning (KDD '21). https:\/\/doi.org\/10.1145\/3447548.3467401","DOI":"10.1145\/3447548.3467401"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01179"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3292066"},{"key":"e_1_3_2_2_53_1","volume-title":"Graph-Guided Network For Irregularly Sampled Multivariate Time Series. In International Conference on Learning Representations, ICLR.","author":"Zhang Xiang","year":"2022","unstructured":"Xiang Zhang, Marko Zeman, Theodoros Tsiligkaridis, and Marinka Zitnik. 2022. Graph-Guided Network For Irregularly Sampled Multivariate Time Series. In International Conference on Learning Representations, ICLR."},{"key":"e_1_3_2_2_54_1","volume-title":"Informer: Beyond Efficient Transformer for Long Sequence Time-Series Forecasting. In The Thirty-Fifth AAAI Conference on Artificial Intelligence, AAAI 2021, Virtual Conference","volume":"35","author":"Zhou Haoyi","year":"2021","unstructured":"Haoyi Zhou, Shanghang Zhang, Jieqi Peng, Shuai Zhang, Jianxin Li, Hui Xiong, and Wancai Zhang. 2021. Informer: Beyond Efficient Transformer for Long Sequence Time-Series Forecasting. In The Thirty-Fifth AAAI Conference on Artificial Intelligence, AAAI 2021, Virtual Conference, Vol. 35. AAAI Press, 11106--11115."},{"key":"e_1_3_2_2_55_1","volume-title":"Proc. 39th International Conference on Machine Learning (ICML 2022)","author":"Zhou Tian","year":"2022","unstructured":"Tian Zhou, Ziqing Ma, Qingsong Wen, Xue Wang, Liang Sun, and Rong Jin. 2022. FEDformer: Frequency enhanced decomposed transformer for long-term series forecasting. In Proc. 39th International Conference on Machine Learning (ICML 2022) (Baltimore, Maryland)."}],"event":{"name":"BCB '24: 15th ACM International Conference on Bioinformatics, Computational Biology and Health Informatics","location":"Shenzhen China","acronym":"BCB '24","sponsor":["SIGBio ACM Special Interest Group on Bioinformatics"]},"container-title":["Proceedings of the 15th ACM International Conference on Bioinformatics, Computational Biology and Health Informatics"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3698587.3701364","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3698587.3701364","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T11:26:47Z","timestamp":1755862007000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3698587.3701364"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,22]]},"references-count":55,"alternative-id":["10.1145\/3698587.3701364","10.1145\/3698587"],"URL":"https:\/\/doi.org\/10.1145\/3698587.3701364","relation":{},"subject":[],"published":{"date-parts":[[2024,11,22]]},"assertion":[{"value":"2024-12-16","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}