{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T00:07:37Z","timestamp":1782518857353,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":94,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100006374","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["#2013502, #1726550, #1660878, #1651909, and #1522107"],"award-info":[{"award-number":["#2013502, #1726550, #1660878, #1651909, and #1522107"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737154","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T21:03:27Z","timestamp":1754255007000},"page":"3565-3576","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["THEMES: An Offline Apprenticeship Learning Framework for Evolving Reward Functions"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0026-9096","authenticated-orcid":false,"given":"Xi","family":"Yang","sequence":"first","affiliation":[{"name":"IBM Research, Yorktown Heights, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5086-7399","authenticated-orcid":false,"given":"Md Mirajul","family":"Islam","sequence":"additional","affiliation":[{"name":"North Carolina State University, Raleigh, NC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3474-8637","authenticated-orcid":false,"given":"Ge","family":"Gao","sequence":"additional","affiliation":[{"name":"Stanford University, Stanford, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1765-7837","authenticated-orcid":false,"given":"Min","family":"Chi","sequence":"additional","affiliation":[{"name":"North Carolina State University, Raleigh, NC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"e_1_3_2_2_3_1","volume-title":"Min-Max Entropy Inverse RL of Multiple Tasks. In 2021 IEEE International Conference on Robotics and Automation (ICRA). IEEE.","author":"Arora Saurabh","year":"2021","unstructured":"Saurabh Arora, Prashant Doshi, and Bikramjit Banerjee. 2021. Min-Max Entropy Inverse RL of Multiple Tasks. In 2021 IEEE International Conference on Robotics and Automation (ICRA). IEEE."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2151"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2018.06.007"},{"key":"e_1_3_2_2_6_1","volume-title":"Proceedings of the 28th International Conference on Machine Learning. 897-904","author":"Babes Monica","year":"2011","unstructured":"Monica Babes, Vukosi Marivate, Kaushik Subramanian, and Michael L Littman. 2011. Apprenticeship learning about multiple intentions. In Proceedings of the 28th International Conference on Machine Learning. 897-904."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"Inci M Baytas Cao Xiao Xi Zhang Fei Wang Anil K Jain and Jiayu Zhou. 2017. Paient subtyping via time-aware LSTM networks. In SIGKDD. ACM.","DOI":"10.1145\/3097983.3097997"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"crossref","unstructured":"Stephen Boyd Neal Parikh Eric Chu Borja Peleato Jonathan Eckstein et al. 2011. Distributed optimization and statistical learning via the alternating direction method of multipliers. Foundations and Trends\u00ae in Machine learning 3 1 (2011) 1-122.","DOI":"10.1561\/2200000016"},{"key":"e_1_3_2_2_9_1","volume-title":"Openai gym. arXiv preprint arXiv:1606.01540","author":"Brockman Greg","year":"2016","unstructured":"Greg Brockman, Vicki Cheung, Ludwig Pettersson, Jonas Schneider, John Schulman, Jie Tang, and Wojciech Zaremba. 2016. Openai gym. arXiv preprint arXiv:1606.01540 (2016)."},{"key":"e_1_3_2_2_10_1","volume-title":"International conference on machine learning. PMLR, 783-792","author":"Brown Daniel","year":"2019","unstructured":"Daniel Brown, Wonjoon Goo, Prabhat Nagarajan, and Scott Niekum. 2019. Extrapolating beyond suboptimal demonstrations via inverse reinforcement learning from observations. In International conference on machine learning. PMLR, 783-792."},{"key":"e_1_3_2_2_11_1","volume-title":"Conference on robot learning. PMLR, 330-359","author":"Brown Daniel S","year":"2020","unstructured":"Daniel S Brown, Wonjoon Goo, and Scott Niekum. 2020. Better-thandemonstrator imitation learning via automatically-ranked demonstrations. In Conference on robot learning. PMLR, 330-359."},{"key":"e_1_3_2_2_12_1","volume-title":"Learning from Imperfect Demonstrations through Dynamics Evaluation. arXiv preprint arXiv:2312.11194","author":"Bu Xizhou","year":"2023","unstructured":"Xizhou Bu, Zhiqiang Ma, Zhengxiong Liu, Wenjuan Li, and Panfeng Huang. 2023. Learning from Imperfect Demonstrations through Dynamics Evaluation. arXiv preprint arXiv:2312.11194 (2023)."},{"key":"e_1_3_2_2_13_1","volume-title":"Scalable bayesian inverse reinforcement learning. arXiv preprint arXiv:2102.06483","author":"Chan Alex J","year":"2021","unstructured":"Alex J Chan and Mihaela van der Schaar. 2021. Scalable bayesian inverse reinforcement learning. arXiv preprint arXiv:2102.06483 (2021)."},{"key":"e_1_3_2_2_14_1","first-page":"305","article-title":"Nonparametric Bayesian inverse reinforcement learning for multiple reward functions","author":"Choi Jaedeug","year":"2012","unstructured":"Jaedeug Choi and Kee-Eung Kim. 2012. Nonparametric Bayesian inverse reinforcement learning for multiple reward functions. In Advances in Neural Information Processing Systems. 305-313.","journal-title":"Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"crossref","unstructured":"Victor Coba Melissa Whitmill et al. 2011. Resuscitation bundle compliance in severe sepsis and septic shock: improves survival is better late than never. Journal of intensive care medicine (2011).","DOI":"10.1177\/0885066610392499"},{"key":"e_1_3_2_2_16_1","first-page":"929","volume-title":"Proceedings of the 28th international conference on machine learning (ICML-11)","author":"Cuturi Marco","year":"2011","unstructured":"Marco Cuturi. 2011. Fast global alignment kernels. In Proceedings of the 28th international conference on machine learning (ICML-11). 929-936."},{"key":"e_1_3_2_2_17_1","volume-title":"European workshop on reinforcement learning. Springer, 273-284","author":"Dimitrakakis Christos","year":"2011","unstructured":"Christos Dimitrakakis and Constantin A Rothkopf. 2011. Bayesian multitask inverse reinforcement learning. In European workshop on reinforcement learning. Springer, 273-284."},{"key":"e_1_3_2_2_18_1","first-page":"486","article-title":"A theoretical analysis of deep Q-learning","author":"Fan Jianqing","year":"2020","unstructured":"Jianqing Fan, Zhaoran Wang, Yuchen Xie, and Zhuoran Yang. 2020. A theoretical analysis of deep Q-learning. In Learning for Dynamics and Control. PMLR, 486-489.","journal-title":"Learning for Dynamics and Control. PMLR"},{"key":"e_1_3_2_2_19_1","volume-title":"Triple-GAIL: a multi-modal imitation learning framework with generative adversarial nets. arXiv preprint arXiv:2005.10622","author":"Fei Cong","year":"2020","unstructured":"Cong Fei, Bin Wang, Yuzheng Zhuang, Zongzhang Zhang, Jianye Hao, Hongbo Zhang, Xuewu Ji, and Wulong Liu. 2020. Triple-GAIL: a multi-modal imitation learning framework with generative adversarial nets. arXiv preprint arXiv:2005.10622 (2020)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599844"},{"key":"e_1_3_2_2_21_1","volume-title":"International conference on machine learning. PMLR, 49-58","author":"Finn Chelsea","year":"2016","unstructured":"Chelsea Finn, Sergey Levine, and Pieter Abbeel. 2016. Guided cost learning: Deep inverse optimal control via policy optimization. In International conference on machine learning. PMLR, 49-58."},{"key":"e_1_3_2_2_22_1","volume-title":"Conference on robot learning. PMLR, 357-368","author":"Finn Chelsea","year":"2017","unstructured":"Chelsea Finn, Tianhe Yu, Tianhao Zhang, Pieter Abbeel, and Sergey Levine. 2017. One-shot visual imitation learning via meta-learning. In Conference on robot learning. PMLR, 357-368."},{"key":"e_1_3_2_2_23_1","volume-title":"The elements of statistical learning","author":"Friedman Jerome","unstructured":"Jerome Friedman, Trevor Hastie, and Robert Tibshirani. 2001. The elements of statistical learning. Vol. 1. Springer series in statistics New York."},{"key":"e_1_3_2_2_24_1","volume-title":"Sparse inverse covariance estimation with the graphical lasso. Biostatistics","author":"Friedman Jerome","year":"2008","unstructured":"Jerome Friedman, Trevor Hastie, and Robert Tibshirani. 2008. Sparse inverse covariance estimation with the graphical lasso. Biostatistics (2008)."},{"key":"e_1_3_2_2_25_1","volume-title":"Ess-infogail: Semi-supervised imitation learning from imbalanced demonstrations. Advances in Neural Information Processing Systems 36","author":"Fu Huiqiao","year":"2024","unstructured":"Huiqiao Fu, Kaiqiang Tang, Yuanyang Lu, Yiming Qi, Guizhou Deng, Flood Sung, and Chunlin Chen. 2024. Ess-infogail: Semi-supervised imitation learning from imbalanced demonstrations. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_2_26_1","unstructured":"Zipeng Fu Minghuan Liu Ming Zhou and Weinan Zhang. 2020. Multi-Modal Imitation Learning in Partially Observable Environments. (2020)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Ge Gao Qitong Gao Xi Yang Miroslav Pajic and Min Chi. 2022. A Reinforcement Learning-Informed Pattern Mining Framework for Multivariate Time Series Classification. (2022).","DOI":"10.24963\/ijcai.2022\/415"},{"key":"e_1_3_2_2_28_1","volume-title":"Early recognition and management of sepsis in adults: the first six hours. American family physician 88, 1","author":"Gauer Robert","year":"2013","unstructured":"Robert Gauer. 2013. Early recognition and management of sepsis in adults: the first six hours. American family physician 88, 1 (2013), 44-53."},{"key":"e_1_3_2_2_29_1","unstructured":"Adam Gleave Mohammad Taufeeque Juan Rocamonde Erik Jenner Steven H. Wang Sam Toyer Maximilian Ernestus Nora Belrose Scott Emmons and Stuart Russell. 2022. imitation: Clean Imitation Learning Implementations. arXiv:2211.11972v1 [cs.LG]. arXiv:2211.11972 [cs.LG] https:\/\/arxiv.org\/abs\/2211.11972"},{"key":"e_1_3_2_2_30_1","volume-title":"Your classifier is secretly an energy based model and you should treat it like one. arXiv preprint arXiv:1912.03263","author":"Grathwohl Will","year":"2019","unstructured":"Will Grathwohl, Kuan-Chieh Wang, J\u00f6rn-Henrik Jacobsen, David Duvenaud, Mohammad Norouzi, and Kevin Swersky. 2019. Your classifier is secretly an energy based model and you should treat it like one. arXiv preprint arXiv:1912.03263 (2019)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106251"},{"key":"e_1_3_2_2_32_1","first-page":"215","article-title":"Toeplitz inverse covariance-based clustering of multivariate time series data","author":"Hallac David","year":"2017","unstructured":"David Hallac, Sagar Vare, Stephen Boyd, and Jure Leskovec. 2017. Toeplitz inverse covariance-based clustering of multivariate time series data. In SIGKDD. 215-223.","journal-title":"SIGKDD."},{"key":"e_1_3_2_2_33_1","volume-title":"Multi-modal imitation learning from unstructured demonstrations using generative adversarial nets. Advances in neural information processing systems 30","author":"Hausman Karol","year":"2017","unstructured":"Karol Hausman, Yevgen Chebotar, Stefan Schaal, Gaurav Sukhatme, and Joseph J Lim. 2017. Multi-modal imitation learning from unstructured demonstrations using generative adversarial nets. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_2_34_1","volume-title":"A targeted real-time early warning score (TREWScore) for septic shock. STM","author":"Henry Katharine E","year":"2015","unstructured":"Katharine E Henry, David N Hager, Peter J Pronovost, and Suchi Saria. 2015. A targeted real-time early warning score (TREWScore) for septic shock. STM (2015)."},{"key":"e_1_3_2_2_35_1","volume-title":"Generative adversarial imitation learning. Advances in neural information processing systems 29","author":"Ho Jonathan","year":"2016","unstructured":"Jonathan Ho and Stefano Ermon. 2016. Generative adversarial imitation learning. Advances in neural information processing systems 29 (2016)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013951"},{"key":"e_1_3_2_2_37_1","first-page":"7354","article-title":"Strictly batch imitation learning by energy-based distribution matching","volume":"33","author":"Jarrett Daniel","year":"2020","unstructured":"Daniel Jarrett, Ioana Bica, and Mihaela van der Schaar. 2020. Strictly batch imitation learning by energy-based distribution matching. Advances in Neural Information Processing Systems 33 (2020), 7354-7365.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_38_1","volume-title":"Leo Anthony Celi, and Roger G Mark","author":"Johnson Alistair EW","year":"2016","unstructured":"Alistair EW Johnson, Tom J Pollard, Lu Shen, Li-wei H Lehman, Mengling Feng, Mohammad Ghassemi, Benjamin Moody, Peter Szolovits, Leo Anthony Celi, and Roger G Mark. 2016. MIMIC-III, a freely accessible critical care database. Scientific data 3, 1 (2016), 1-9."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/322"},{"key":"e_1_3_2_2_40_1","volume-title":"European Workshop on Reinforcement Learning. Springer, 285-296","author":"Klein Edouard","year":"2011","unstructured":"Edouard Klein, Matthieu Geist, and Olivier Pietquin. 2011. Batch, off-policy and model-free apprenticeship learning. In European Workshop on Reinforcement Learning. Springer, 285-296."},{"key":"e_1_3_2_2_41_1","volume-title":"Situated GAIL: Multitask imitation using task-conditioned adversarial inverse reinforcement learning. arXiv preprint arXiv:1911.00238","author":"Kobayashi Kyoichiro","year":"2019","unstructured":"Kyoichiro Kobayashi, Takato Horii, Ryo Iwaki, Yukie Nagai, and Minoru Asada. 2019. Situated GAIL: Multitask imitation using task-conditioned adversarial inverse reinforcement learning. arXiv preprint arXiv:1911.00238 (2019)."},{"key":"e_1_3_2_2_42_1","volume-title":"The artificial intelligence clinician learns optimal treatment strategies for sepsis in intensive care. Nature medicine 24, 11","author":"Komorowski Matthieu","year":"2018","unstructured":"Matthieu Komorowski, Leo A Celi, Omar Badawi, Anthony C Gordon, and A Aldo Faisal. 2018. The artificial intelligence clinician learns optimal treatment strategies for sepsis in intensive care. Nature medicine 24, 11 (2018), 1716-1720."},{"key":"e_1_3_2_2_43_1","volume-title":"Debidatta Dwibedi, Sergey Levine, and Jonathan Tompson.","author":"Kostrikov Ilya","year":"2018","unstructured":"Ilya Kostrikov, Kumar Krishna Agrawal, Debidatta Dwibedi, Sergey Levine, and Jonathan Tompson. 2018. Discriminator-actor-critic: Addressing sample inefficiency and reward bias in adversarial imitation learning. arXiv preprint:1809.02925 (2018)."},{"key":"e_1_3_2_2_44_1","volume-title":"Imitation learning via off-policy distribution matching. arXiv preprint arXiv:1912.05032","author":"Kostrikov Ilya","year":"2019","unstructured":"Ilya Kostrikov, Ofir Nachum, and Jonathan Tompson. 2019. Imitation learning via off-policy distribution matching. arXiv preprint arXiv:1912.05032 (2019)."},{"key":"e_1_3_2_2_45_1","volume-title":"Hirl: Hierarchical inverse reinforcement learning for long-horizon tasks with delayed rewards. arXiv preprint arXiv:1604.06508","author":"Krishnan Sanjay","year":"2016","unstructured":"Sanjay Krishnan, Animesh Garg, Richard Liaw, Lauren Miller, Florian T Pokorny, and Ken Goldberg. 2016. Hirl: Hierarchical inverse reinforcement learning for long-horizon tasks with delayed rewards. arXiv preprint arXiv:1604.06508 (2016)."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Anand Kumar Daniel Roberts Kenneth E Wood et al. 2006. Duration of hypotension before initiation of effective antimicrobial therapy is the critical determinant of survival in human septic shock. Critical care medicine (2006).","DOI":"10.1097\/01.CCM.0000217961.75225.E9"},{"key":"e_1_3_2_2_47_1","volume-title":"Truly batch apprenticeship learning with deep successor features. arXiv preprint arXiv:1903.10077","author":"Lee Donghun","year":"2019","unstructured":"Donghun Lee, Srivatsan Srinivasan, and Finale Doshi-Velez. 2019. Truly batch apprenticeship learning with deep successor features. arXiv preprint arXiv:1903.10077 (2019)."},{"key":"e_1_3_2_2_48_1","volume-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643","author":"Levine Sergey","year":"2020","unstructured":"Sergey Levine, Aviral Kumar, George Tucker, and Justin Fu. 2020. Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643 (2020)."},{"key":"e_1_3_2_2_49_1","volume-title":"International Conference on Learning Representations.","author":"Li Jiayi","year":"2021","unstructured":"Jiayi Li, Tao Lu, Xiaoge Cao, Yinghao Cai, and Shuo Wang. 2021. Meta-imitation learning by watching video demonstrations. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219831"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611972825.77"},{"key":"e_1_3_2_2_52_1","volume-title":"Infogail: Interpretable imitation learning from visual demonstrations. Advances in neural information processing systems 30","author":"Li Yunzhu","year":"2017","unstructured":"Yunzhu Li, Jiaming Song, and Stefano Ermon. 2017. Infogail: Interpretable imitation learning from visual demonstrations. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_2_53_1","volume-title":"RNN and LSTM for Learning Gain Prediction. In Artificial Intelligence in Education, Elisabeth Andr\u00e9, Ryan Baker, Xiangen Hu, Ma. Mercedes T. Rodrigo, and Benedict du Boulay (Eds.)","author":"Lin Chen","unstructured":"Chen Lin and Min Chi. 2017. A Comparisons of BKT, RNN and LSTM for Learning Gain Prediction. In Artificial Intelligence in Education, Elisabeth Andr\u00e9, Ryan Baker, Xiangen Hu, Ma. Mercedes T. Rodrigo, and Benedict du Boulay (Eds.). Springer, 536-539."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-97304-3_25"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.65109\/TNVS8310"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20735"},{"key":"e_1_3_2_2_57_1","volume-title":"Deep Learning vs. Bayesian Knowledge Tracing: Student Models for Interventions. Journal of Educational Data Mining","author":"Mao Ye","year":"2018","unstructured":"Ye Mao, Chen Lin, and Min Chi. 2018. Deep Learning vs. Bayesian Knowledge Tracing: Student Models for Interventions. Journal of Educational Data Mining (2018)."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1056\/NEJMoa022139"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"crossref","unstructured":"Nicolai Meinshausen and Peter B\u00fchlmann. 2006. High-dimensional graphs and variable selection with the Lasso. Ann. Statist. (2006).","DOI":"10.1214\/009053606000000281"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/Humanoids.2011.6100830"},{"key":"e_1_3_2_2_61_1","volume-title":"Bryan Kian Hsiang Low, and Patrick Jaillet","author":"Nguyen Quoc Phong","year":"2015","unstructured":"Quoc Phong Nguyen, Bryan Kian Hsiang Low, and Patrick Jaillet. 2015. Inverse reinforcement learning with locally consistent reward functions. Advances in neural information processing systems 28 (2015)."},{"key":"e_1_3_2_2_62_1","volume-title":"Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery & data mining. 1480-1490","author":"Garud Iyengar Oh","year":"2019","unstructured":"Min-hwan Oh and Garud Iyengar. 2019. Sequential anomaly detection using inverse reinforcement learning. In Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery & data mining. 1480-1490."},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.05.076"},{"key":"e_1_3_2_2_64_1","volume-title":"Efficient training of artificial neural networks for autonomous navigation. Neural computation 3, 1","author":"Pomerleau Dean A","year":"1991","unstructured":"Dean A Pomerleau. 1991. Efficient training of artificial neural networks for autonomous navigation. Neural computation 3, 1 (1991), 88-97."},{"key":"e_1_3_2_2_65_1","volume-title":"Multi-modal inverse constrained reinforcement learning from a mixture of demonstrations. Advances in Neural Information Processing Systems 36","author":"Qiao Guanren","year":"2024","unstructured":"Guanren Qiao, Guiliang Liu, Pascal Poupart, and Zhiqiang Xu. 2024. Multi-modal inverse constrained reinforcement learning from a mixture of demonstrations. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_2_66_1","volume-title":"Machine Learning for Healthcare Conference. PMLR, 147-163","author":"Raghu Aniruddh","year":"2017","unstructured":"Aniruddh Raghu, Matthieu Komorowski, Leo Anthony Celi, Peter Szolovits, and Marzyeh Ghassemi. 2017. Continuous state-space models for optimal sepsis treatment: a deep reinforcement learning approach. In Machine Learning for Healthcare Conference. PMLR, 147-163."},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROBIO.2012.6491170"},{"key":"e_1_3_2_2_68_1","volume-title":"Gaussian Mixture Models. Encyclopedia of biometrics 741","author":"Reynolds Douglas A","year":"2009","unstructured":"Douglas A Reynolds. 2009. Gaussian Mixture Models. Encyclopedia of biometrics 741 (2009)."},{"key":"e_1_3_2_2_69_1","volume-title":"Proceedings of the fourteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings, 627-635","author":"Ross St\u00e9phane","year":"2011","unstructured":"St\u00e9phane Ross, Geoffrey Gordon, and Drew Bagnell. 2011. A reduction of imitation learning and structured prediction to no-regret online learning. In Proceedings of the fourteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings, 627-635."},{"key":"e_1_3_2_2_70_1","volume-title":"Personalized Input-Output Hidden Markov Models for Disease Progression Modeling. In Machine Learning for Healthcare Conference.","author":"Severson Kristen A","year":"2020","unstructured":"Kristen A Severson, Lana M Chahine, et al. 2020. Personalized Input-Output Hidden Markov Models for Disease Progression Modeling. In Machine Learning for Healthcare Conference."},{"key":"e_1_3_2_2_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3672059"},{"key":"e_1_3_2_2_72_1","volume-title":"Directed-info gail: Learning hierarchical policies from unsegmented demonstrations using directed information. arXiv preprint arXiv:1810.01266","author":"Sharma Arjun","year":"2018","unstructured":"Arjun Sharma, Mohit Sharma, Nicholas Rhinehart, and Kris M Kitani. 2018. Directed-info gail: Learning hierarchical policies from unsegmented demonstrations using directed information. arXiv preprint arXiv:1810.01266 (2018)."},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"crossref","unstructured":"Mervyn Singer Clifford S Deutschman et al. 2016. The third international consensus definitions for sepsis and septic shock (Sepsis-3). Jama 315 (2016).","DOI":"10.1001\/jama.2016.0287"},{"key":"e_1_3_2_2_74_1","unstructured":"Padhraic Smyth. 1997. Clustering sequences with hidden Markov models. In Advances in neural information processing systems. 648-654."},{"key":"e_1_3_2_2_75_1","volume-title":"Imitation learning with a value-based prior. arXiv preprint arXiv:1206.5290","author":"Syed Umar","year":"2012","unstructured":"Umar Syed and Robert E Schapire. 2012. Imitation learning with a value-based prior. arXiv preprint arXiv:1206.5290 (2012)."},{"key":"e_1_3_2_2_76_1","volume-title":"International Conference on Machine Learning. PMLR, 9407-9417","author":"Tangkaratt Voot","year":"2020","unstructured":"Voot Tangkaratt, Bo Han, Mohammad Emtiyaz Khan, and Masashi Sugiyama. 2020. Variational imitation learning with diverse-quality demonstrations. In International Conference on Machine Learning. PMLR, 9407-9417."},{"key":"e_1_3_2_2_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1967.1054010"},{"key":"e_1_3_2_2_78_1","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498535"},{"key":"e_1_3_2_2_79_1","volume-title":"Hierarchical Imitation Learning with Contextual Bandits for Dynamic Treatment Regimes. In Reinforcement Learning for Real Life Workshop in the 38th International Conference on Machine Learning.","author":"Wang Lu","year":"2021","unstructured":"Lu Wang, Wenchao Yu, Wei Cheng, Bo Zong, and Haifeng Chen. 2021. Hierarchical Imitation Learning with Contextual Bandits for Dynamic Treatment Regimes. In Reinforcement Learning for Real Life Workshop in the 38th International Conference on Machine Learning."},{"key":"e_1_3_2_2_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3366423.3380248"},{"key":"e_1_3_2_2_81_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i8.26222"},{"key":"e_1_3_2_2_82_1","first-page":"3155","article-title":"Robust Adversarial Imitation Learning via Adaptively-Selected Demonstrations","author":"Wang Yunke","year":"2021","unstructured":"Yunke Wang, Chang Xu, and Bo Du. 2021. Robust Adversarial Imitation Learning via Adaptively-Selected Demonstrations.. In IJCAI. 3155-3161.","journal-title":"IJCAI."},{"key":"e_1_3_2_2_83_1","volume-title":"International Conference on Machine Learning. PMLR, 10961-10970","author":"Wang Yunke","year":"2021","unstructured":"Yunke Wang, Chang Xu, Bo Du, and Honglak Lee. 2021. Learning to weight imperfect demonstrations. In International Conference on Machine Learning. PMLR, 10961-10970."},{"key":"e_1_3_2_2_84_1","volume-title":"Robust imitation of diverse behaviors. Advances in Neural Information Processing Systems 30","author":"Wang Ziyu","year":"2017","unstructured":"Ziyu Wang, Josh S Merel, Scott E Reed, Nando de Freitas, Gregory Wayne, and Nicolas Heess. 2017. Robust imitation of diverse behaviors. Advances in Neural Information Processing Systems 30 (2017)."},{"key":"e_1_3_2_2_85_1","volume-title":"Proceedings of the 28th international conference on machine learning. Citeseer, 681-688","author":"Welling Max","year":"2011","unstructured":"Max Welling and Yee W Teh. 2011. Bayesian learning via stochastic gradient Langevin dynamics. In Proceedings of the 28th international conference on machine learning. Citeseer, 681-688."},{"key":"e_1_3_2_2_86_1","volume-title":"International Conference on Machine Learning. PMLR, 6818-6827","author":"Wu Yueh-Hua","year":"2019","unstructured":"Yueh-Hua Wu, Nontawat Charoenphakdee, Han Bao, Voot Tangkaratt, and Masashi Sugiyama. 2019. Imitation learning from imperfect demonstration. In International Conference on Machine Learning. PMLR, 6818-6827."},{"key":"e_1_3_2_2_87_1","volume-title":"International Conference on Machine Learning. PMLR, 24725-24742","author":"Xu Haoran","year":"2022","unstructured":"Haoran Xu, Xianyuan Zhan, Honglei Yin, and Huiling Qin. 2022. Discriminatorweighted offline imitation learning from suboptimal demonstrations. In International Conference on Machine Learning. PMLR, 24725-24742."},{"key":"e_1_3_2_2_88_1","volume-title":"Trail: Near-optimal imitation learning with suboptimal data. arXiv preprint arXiv:2110.14770","author":"Yang Mengjiao","year":"2021","unstructured":"Mengjiao Yang, Sergey Levine, and Ofir Nachum. 2021. Trail: Near-optimal imitation learning with suboptimal data. arXiv preprint arXiv:2110.14770 (2021)."},{"key":"e_1_3_2_2_89_1","doi-asserted-by":"crossref","unstructured":"Xi Yang Yuan Zhang and Min Chi. 2021. Multi-series time-aware sequence partitioning for disease progression modeling. In IJCAI.","DOI":"10.24963\/ijcai.2021\/493"},{"key":"e_1_3_2_2_90_1","volume-title":"Student Subtyping via EM-Inverse Reinforcement Learning","author":"Yang Xi","year":"2020","unstructured":"Xi Yang, Guojing Zhou, Michelle Taub, Roger Azevedo, and Min Chi. 2020. Student Subtyping via EM-Inverse Reinforcement Learning. International Educational Data Mining Society (2020)."},{"key":"e_1_3_2_2_91_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/607"},{"key":"e_1_3_2_2_92_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599506"},{"key":"e_1_3_2_2_93_1","volume-title":"TAC-GAIL: A Multi-modal Imitation Learning Method. In International Conference on Neural Information Processing. Springer, 688-699","author":"Zhu Jiacheng","year":"2020","unstructured":"Jiacheng Zhu and Chong Jiang. 2020. TAC-GAIL: A Multi-modal Imitation Learning Method. In International Conference on Neural Information Processing. Springer, 688-699."},{"key":"e_1_3_2_2_94_1","unstructured":"Brian D Ziebart Andrew Maas J Andrew Bagnell and Anind K Dey. 2008. Maximum entropy inverse reinforcement learning. (2008)."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737154","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T18:02:07Z","timestamp":1777572127000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737154"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":94,"alternative-id":["10.1145\/3711896.3737154","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737154","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}