{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,27]],"date-time":"2026-07-27T12:59:27Z","timestamp":1785157167007,"version":"3.55.0"},"reference-count":260,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,7,27]],"date-time":"2026-07-27T00:00:00Z","timestamp":1785110400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,27]],"date-time":"2026-07-27T00:00:00Z","timestamp":1785110400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Data Min Knowl Disc"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1007\/s10618-026-01246-3","type":"journal-article","created":{"date-parts":[[2026,7,27]],"date-time":"2026-07-27T10:14:03Z","timestamp":1785147243000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A survey on explainable reinforcement learning: state of the art, challenges and opportunities"],"prefix":"10.1007","volume":"40","author":[{"given":"Daniele","family":"Melloni","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andrea","family":"Zingoni","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,27]]},"reference":[{"key":"1246_CR1","doi-asserted-by":"publisher","first-page":"52138","DOI":"10.1109\/ACCESS.2018.2870052","volume":"6","author":"A Adadi","year":"2018","unstructured":"Adadi A, Berrada M (2018) Peeking inside the black-box: a survey on explainable artificial intelligence (XAI). IEEE Access 6:52138\u201352160","journal-title":"IEEE Access"},{"key":"1246_CR2","unstructured":"Akrour R, Tateo D, Peters J (2019) Towards reinforcement learning of human readable policies. In: The European conference on machine learning and principles and practice of knowledge discovery in databases: the 1st workshop on deep continuous-discrete machine learning. p 37"},{"key":"1246_CR5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-04110-6","volume-title":"Artificial intelligence in IoT","author":"F Al-Turjman","year":"2019","unstructured":"Al-Turjman F (2019) Artificial intelligence in IoT. Springer, Heidelberg, Germany"},{"key":"1246_CR3","doi-asserted-by":"publisher","first-page":"171058","DOI":"10.1109\/ACCESS.2020.3023394","volume":"8","author":"A Alharin","year":"2020","unstructured":"Alharin A, Doan T-N, Sartipi M (2020) Reinforcement learning interpretation methods: a survey. IEEE Access 8:171058\u2013171077","journal-title":"IEEE Access"},{"key":"1246_CR4","doi-asserted-by":"crossref","unstructured":"Altmann P, Davignon C, Zorn M, Ritz F, Linnhoff-Popien C, Gabor T (2025) Surrogate fitness metrics for interpretable reinforcement learning. arXiv preprint arXiv:2504.14645","DOI":"10.1007\/s42979-026-04884-y"},{"key":"1246_CR6","doi-asserted-by":"crossref","unstructured":"Amir D, Amir O (2018) Highlights: Summarizing agent behavior to people. In: Proceedings of the 17th international conference on autonomous agents and MultiAgent systems, pp 1168\u20131176","DOI":"10.65109\/WIZC1785"},{"key":"1246_CR7","doi-asserted-by":"publisher","first-page":"628","DOI":"10.1007\/s10458-019-09418-w","volume":"33","author":"O Amir","year":"2019","unstructured":"Amir O, Doshi-Velez F, Sarne D (2019) Summarizing agent strategies. Auton Agent Multi-Agent Syst 33:628\u2013644","journal-title":"Auton Agent Multi-Agent Syst"},{"key":"1246_CR8","doi-asserted-by":"crossref","unstructured":"Anderson A, Dodge J, Sadarangani A, Juozapaitis Z, Newman E, Irvine J, Chattopadhyay S, Fern A, Burnett M (2019) Explaining reinforcement learning to mere mortals: an empirical study. arXiv preprint arXiv: 1903.09708","DOI":"10.24963\/ijcai.2019\/184"},{"key":"1246_CR9","doi-asserted-by":"crossref","unstructured":"Anjomshoae S, Najjar A, Calvaresi D, Fr\u00e4mling K (2019) Explainable agents and robots: results from a systematic literature review. In: 18th international conference on autonomous agents and multiagent systems (AAMAS 2019), Montreal, Canada, May 13\u201317. International Foundation for Autonomous Agents and Multiagent Systems, pp 1078\u20131088","DOI":"10.65109\/KCZB5817"},{"key":"1246_CR10","doi-asserted-by":"publisher","first-page":"4561","DOI":"10.1609\/aaai.v33i01.33014561","volume":"33","author":"RM Annasamy","year":"2019","unstructured":"Annasamy RM, Sycara K (2019) Towards better interpretability in deep q-networks. Proceedings of the AAAI conference on artificial intelligence 33:4561\u20134569","journal-title":"Proceedings of the AAAI conference on artificial intelligence"},{"key":"1246_CR11","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1016\/j.inffus.2019.12.012","volume":"58","author":"AB Arrieta","year":"2020","unstructured":"Arrieta AB, D\u00edaz-Rodr\u00edguez N, Del Ser J, Bennetot A, Tabik S, Barbado A, Garc\u00eda S, Gil-L\u00f3pez S, Molina D, Benjamins R et al (2020) Explainable artificial intelligence (xai): concepts, taxonomies, opportunities and challenges toward responsible ai. Inform Fusion 58:82\u2013115","journal-title":"Inform Fusion"},{"issue":"1791","key":"1246_CR12","doi-asserted-by":"publisher","first-page":"20190307","DOI":"10.1098\/rstb.2019.0307","volume":"375","author":"M Baroni","year":"2020","unstructured":"Baroni M (2020) Linguistic generalization and compositionality in modern artificial neural networks. Philos Trans R Soc B 375(1791):20190307","journal-title":"Philos Trans R Soc B"},{"issue":"1\u20132","key":"1246_CR13","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1023\/A:1022140919877","volume":"13","author":"AG Barto","year":"2003","unstructured":"Barto AG, Mahadevan S (2003) Recent advances in hierarchical reinforcement learning. Discrete Event Dyn Syst 13(1\u20132):41\u201377","journal-title":"Discrete Event Dyn Syst"},{"key":"1246_CR14","doi-asserted-by":"publisher","first-page":"993","DOI":"10.1016\/j.future.2019.07.059","volume":"101","author":"G Baryannis","year":"2019","unstructured":"Baryannis G, Dani S, Antoniou G (2019) Predicting supply chain risks using machine learning: the trade-off between performance and interpretability. Futur Gener Comput Syst 101:993\u20131004","journal-title":"Futur Gener Comput Syst"},{"key":"1246_CR15","unstructured":"Bastani O, Pu Y, Solar-Lezama A (2018) Verifiable reinforcement learning via policy extraction. Adv Neural Inf Process Syst 31"},{"key":"1246_CR16","unstructured":"Beechey D, Smith TM, \u015eim\u015fek \u00d6 (2023) Explaining reinforcement learning with shapley values. In: International conference on machine learning. PMLR, pp 2003\u20132014"},{"key":"1246_CR17","doi-asserted-by":"publisher","first-page":"688969","DOI":"10.3389\/fdata.2021.688969","volume":"39","author":"V Belle","year":"2021","unstructured":"Belle V, Papantonis I (2021) Principles and practice of explainable machine learning. Front Big Data 39:688969","journal-title":"Front Big Data"},{"key":"1246_CR18","doi-asserted-by":"crossref","unstructured":"Bertino E, Kantarcioglu M, Akcora CG, Samtani S, Mittal S, Gupta M (2021) Ai for security and security for AI. In: Proceedings of the eleventh ACM conference on data and application security and privacy. pp 333\u2013334","DOI":"10.1145\/3422337.3450357"},{"key":"1246_CR19","unstructured":"Bertsimas D, Delarue A, Jaillet P, Martin S (2019) The price of interpretability. arXiv preprint arXiv:1907.03419"},{"key":"1246_CR20","doi-asserted-by":"crossref","unstructured":"Boggess K, Kraus S, Feng L (2022) Toward policy explanations for multi-agent reinforcement learning. arXiv preprint arXiv: 2204.12568","DOI":"10.24963\/ijcai.2022\/16"},{"key":"1246_CR21","doi-asserted-by":"publisher","DOI":"10.1201\/9781315139470","volume-title":"Classification and regression trees","author":"L Breiman","year":"2017","unstructured":"Breiman L (2017) Classification and regression trees. Routledge, London, UK"},{"key":"1246_CR22","unstructured":"Brockman G, Cheung V, Pettersson L, Schneider J, Schulman J, Tang J, Zaremba W (2016) Openai gym. arXiv preprint arXiv: 1606.01540"},{"key":"1246_CR23","unstructured":"Brooks D, Shultz A, Desai M, Kovac P, Yanco HA (2010) Towards state summarization for autonomous robots. AAAI Fall Symposium Series"},{"key":"1246_CR24","unstructured":"Brown A, Petrik M (2018) Interpretable reinforcement learning with ensemble methods. arXiv preprint arXiv: 1809.06995"},{"issue":"1","key":"1246_CR25","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1177\/1745691610393980","volume":"6","author":"M Buhrmester","year":"2011","unstructured":"Buhrmester M, Kwang T, Gosling SD (2011) Amazon\u2019s mechanical turk: A new source of inexpensive, yet high-quality, data? Perspect Psychol Sci 6(1):3\u20135","journal-title":"Perspect Psychol Sci"},{"key":"1246_CR26","doi-asserted-by":"publisher","first-page":"103923","DOI":"10.1016\/j.artint.2023.103923","volume":"320","author":"E Candela","year":"2023","unstructured":"Candela E, Doustaly O, Parada L, Feng F, Demiris Y, Angeloudis P (2023) Risk-aware controller for autonomous vehicles using model-based collision prediction and reinforcement learning. Artif Intell 320:103923","journal-title":"Artif Intell"},{"issue":"3","key":"1246_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3502289","volume":"55","author":"L Cao","year":"2022","unstructured":"Cao L (2022) Ai in finance: challenges, techniques, and opportunities. ACM Comput Surv (CSUR) 55(3):1\u201338","journal-title":"ACM Comput Surv (CSUR)"},{"key":"1246_CR28","doi-asserted-by":"crossref","unstructured":"Capasso C, Zingoni A, Calabr\u2018o G, Sterpa A (2023) Legal and technical answers to privacy issues raised by AI-based facial recognition algorithms. In: 2023 IEEE international conference on metrology for extended reality, artificial intelligence and neural engineering (MetroXRAINE). IEEE, pp 575-580.","DOI":"10.1109\/MetroXRAINE58569.2023.10405669"},{"key":"1246_CR29","unstructured":"Cederborg T, Grover I, Isbell\u00a0Jr CL, Thomaz AL (2015) Policy shaping with human teachers. In: IJCAI, pp 3366\u20133372"},{"key":"1246_CR30","doi-asserted-by":"crossref","unstructured":"Chakraborti T, Sreedharan S, Kambhampati S (2020) The emerging landscape of explainable AI planning and decision making. arXiv preprint arXiv: 2002.11697","DOI":"10.24963\/ijcai.2020\/669"},{"key":"1246_CR31","doi-asserted-by":"crossref","unstructured":"Chen C, Zhang M, Liu Y, Ma S (2018) Neural attentional rating regression with review-level explanations. In: Proceedings of the 2018 World Wide Web Conference. pp 1583\u20131592","DOI":"10.1145\/3178876.3186070"},{"key":"1246_CR32","unstructured":"Chen V, Gupta A, Marino K (2020) Ask your humans: using human instructions to improve generalization in reinforcement learning. arXiv preprint arXiv: 2011.00517"},{"key":"1246_CR33","doi-asserted-by":"publisher","first-page":"62457","DOI":"10.52202\/075280-2728","volume":"36","author":"Z Cheng","year":"2023","unstructured":"Cheng Z, Wu X, Yu J, Sun W, Guo W, Xing X (2023) Statemask: explaining deep reinforcement learning through state mask. Adv Neural Inf Process Syst 36:62457\u201362487","journal-title":"Adv Neural Inf Process Syst"},{"key":"1246_CR34","unstructured":"Chen Z, Silvestri F, Tolomei G, Zhu H, Wang J, Ahn H (2021) Relace: Reinforcement learning agent for counterfactual explanations of arbitrary predictive models. arXiv preprint arXiv: 2110.11960"},{"issue":"4","key":"1246_CR35","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1109\/MCG.2018.042731661","volume":"38","author":"J Choo","year":"2018","unstructured":"Choo J, Liu S (2018) Visual analytics for explainable deep learning. IEEE Comput Graph Appl 38(4):84\u201392","journal-title":"IEEE Comput Graphics Appl"},{"key":"1246_CR36","doi-asserted-by":"publisher","first-page":"841","DOI":"10.1016\/S1574-6526(07)03022-2","volume":"3","author":"A Cimatti","year":"2008","unstructured":"Cimatti A, Pistore M, Traverso P (2008) Automated planning. Found Artif Intell 3:841\u2013867","journal-title":"Found Artif Intell"},{"key":"1246_CR37","unstructured":"Coppens Y, Efthymiadis K, Lenaerts T, Now\u00e9 A, Miller T, Weber R, Magazzeni D (2019) Distilling deep reinforcement learning policies in soft decision trees. In: Proceedings of the IJCAI 2019 workshop on explainable artificial intelligence. pp 1\u20136"},{"key":"1246_CR38","doi-asserted-by":"crossref","unstructured":"Cruz F, Dazeley R, Vamplew P (2019) Memory-based explainable reinforcement learning. In: AI 2019: Advances in Artificial Intelligence: 32nd Australasian Joint Conference, Adelaide, SA, Australia, December 2\u20135, 2019, Proceedings 32. Springer, pp 66\u201377","DOI":"10.1007\/978-3-030-35288-2_6"},{"issue":"6","key":"1246_CR39","doi-asserted-by":"publisher","first-page":"6936","DOI":"10.1007\/s10489-022-03788-7","volume":"53","author":"Y Dai","year":"2023","unstructured":"Dai Y, Ouyang H, Zheng H, Long H, Duan X (2023) Interpreting a deep reinforcement learning model with conceptual embedding and performance analysis. Appl Intell 53(6):6936\u20136952","journal-title":"Appl Intell"},{"issue":"23","key":"1246_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s00521-023-08423-1","volume":"35","author":"R Dazeley","year":"2023","unstructured":"Dazeley R, Vamplew P, Cruz F (2023) Explainable reinforcement learning for broad-Xai: a conceptual framework and survey. Neural Comput Appl 35(23):1\u201324","journal-title":"Neural Comput Appl"},{"issue":"4","key":"1246_CR41","doi-asserted-by":"publisher","first-page":"1389","DOI":"10.1007\/s10994-022-06293-7","volume":"112","author":"G De Toni","year":"2023","unstructured":"De Toni G, Lepri B, Passerini A (2023) Synthesizing explainable counterfactual policies for algorithmic recourse with program synthesis. Mach Learn 112(4):1389\u20131409","journal-title":"Mach Learn"},{"key":"1246_CR42","first-page":"362","volume-title":"International advanced computing conference","author":"S Deshpande","year":"2021","unstructured":"Deshpande S, Walambe R, Kotecha K, Jakovljevi\u0107 MM (2021) Post-hoc explainable reinforcement learning using probabilistic graphical models. In: International advanced computing conference. Springer, Cham, pp 362\u2013376"},{"key":"1246_CR43","doi-asserted-by":"crossref","unstructured":"Dethise A, Canini M, Kandula S (2019) Cracking open the black box: What observations can tell us about reinforcement learning agents. In: Proceedings of the 2019 Workshop on Network Meets AI & ML. pp 29\u201336","DOI":"10.1145\/3341216.3342210"},{"key":"1246_CR44","doi-asserted-by":"publisher","first-page":"169","DOI":"10.1023\/A:1007355226281","volume":"28","author":"TG Dietterich","year":"1997","unstructured":"Dietterich TG, Flann NS (1997) Explanation-based learning and reinforcement learning: a unified view. Mach Learn 28:169\u2013210","journal-title":"Mach Learn"},{"issue":"4","key":"1246_CR45","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1002\/ail2.36","volume":"2","author":"J Dodge","year":"2021","unstructured":"Dodge J, Anderson A, Khanna R, Irvine J, Dikkala R, Lam K-H, Tabatabai D, Ruangrotsakun A, Shureih Z, Kahng M et al (2021) From \u201cno clear winner\u2019\u2019 to an effective explainable artificial intelligence process: an empirical journey. Appl AI Lett 2(4):36","journal-title":"Appl AI Lett"},{"key":"1246_CR46","doi-asserted-by":"crossref","unstructured":"Dodson T, Mattei N, Goldsmith J (2011) A natural language argumentation interface for explanation generation in markov decision processes. In: Algorithmic decision theory: second international conference, ADT 2011, Piscataway, NJ, USA, October 26\u201328, 2011. Proceedings 2. Springer, pp 42\u201355","DOI":"10.1007\/978-3-642-24873-3_4"},{"key":"1246_CR47","unstructured":"Dong Y, Su H, Zhu J, Bao F (2017) Towards interpretable deep neural networks by leveraging adversarial examples. arXiv preprint arXiv: 1708.05493"},{"key":"1246_CR48","doi-asserted-by":"crossref","unstructured":"Do\u0161ilovi\u0107 FK, Br\u010di\u0107 M, Hlupi\u0107 N (2018) Explainable artificial intelligence: a survey. In: 2018 41st International convention on information and communication technology, electronics and microelectronics (MIPRO). IEEE, pp 0210\u20130215","DOI":"10.23919\/MIPRO.2018.8400040"},{"key":"1246_CR49","doi-asserted-by":"crossref","unstructured":"Ehsan U, Tambwekar P, Chan L, Harrison B, Riedl MO (2019) Automated rationale generation: a technique for explainable ai and its effects on human perceptions. In: Proceedings of the 24th international conference on intelligent user interfaces. pp 263\u2013274","DOI":"10.1145\/3301275.3302316"},{"key":"1246_CR50","unstructured":"Elizalde F, Sucar LE, Luque M, Diez J, Reyes A (2008) Policy explanation in factored Markov decision processes. In: Proceedings of the 4th European workshop on probabilistic graphical models (PGM 2008). pp 97\u2013104"},{"key":"1246_CR51","first-page":"51","volume-title":"MICAI 2009: Advances in artificial intelligence: 8th Mexican international conference on artificial intelligence, Guanajuato, M\u00e9xico, November 9\u201313, 2009. Proceedings 8","author":"F Elizalde","year":"2009","unstructured":"Elizalde F, Sucar E, Noguez J, Reyes A (2009) Generating explanations based on Markov decision processes. In: MICAI 2009: Advances in artificial intelligence: 8th Mexican international conference on artificial intelligence, Guanajuato, M\u00e9xico, November 9\u201313, 2009. Proceedings 8. Springer, Cham, pp 51\u201362"},{"key":"1246_CR52","first-page":"330","volume-title":"International conference on machine learning, optimization, and data science","author":"RC Engelhardt","year":"2022","unstructured":"Engelhardt RC, Lange M, Wiskott L, Konen W (2022) Sample-based rule extraction for explainable reinforcement learning.In: International conference on machine learning, optimization, and data science. Springer, Cham, pp 330\u2013345"},{"key":"1246_CR53","unstructured":"Erwig M, Fern A, Murali M, Koul A (2018) Explaining deep adaptive programs via reward decomposition. In: IJCAI\/ECAI workshop on explainable artificial intelligence"},{"key":"1246_CR54","unstructured":"Espeholt L, Soyer H, Munos R, Simonyan K, Mnih V, Ward T, Doron Y, Firoiu V, Harley T, Dunning I (2018) Impala: Scalable distributed deep-rl with importance weighted actor-learner architectures. In: International conference on machine learning. PMLR, pp 1407\u20131416"},{"key":"1246_CR55","doi-asserted-by":"publisher","unstructured":"Feit F, Metzger A, Pohl K (2022) Explaining online reinforcement learning decisions of self-adaptive systems. In: 2022 IEEE international conference on autonomic computing and self-organizing systems (ACSOS). pp 51\u201360. https:\/\/doi.org\/10.1109\/ACSOS55765.2022.00023","DOI":"10.1109\/ACSOS55765.2022.00023"},{"issue":"3\u20134","key":"1246_CR56","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1016\/0004-3702(71)90010-5","volume":"2","author":"RE Fikes","year":"1971","unstructured":"Fikes RE, Nilsson NJ (1971) Strips: A new approach to the application of theorem proving to problem solving. Artif Intell 2(3\u20134):189\u2013208","journal-title":"Artif Intell"},{"key":"1246_CR57","unstructured":"Finn C, Tan XY, Duan Y, Darrell T, Levine S, Abbeel P (2015) Learning visual feature spaces for robotic manipulation with deep spatial autoencoders. CoRR abs\/1509.06113 arXiv: 1509.06113"},{"key":"1246_CR58","doi-asserted-by":"crossref","unstructured":"Foerster J, Farquhar G, Afouras T, Nardelli N, Whiteson S (2018) Counterfactual multi-agent policy gradients. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"1246_CR59","unstructured":"Frosst N, Hinton G (2017) Distilling a neural network into a soft decision tree. arXiv preprint arXiv: 1711.09784"},{"key":"1246_CR60","doi-asserted-by":"crossref","unstructured":"Fukuchi Y, Osawa M, Yamakawa H, Imai M (2017) Autonomous self-explanation of behavior for interactive reinforcement learning agents. In: Proceedings of the 5th international conference on human agent interaction. pp 97\u2013101","DOI":"10.1145\/3125739.3125746"},{"key":"1246_CR61","doi-asserted-by":"crossref","unstructured":"Fulton N, Platzer A (2018) Safe reinforcement learning via formal methods: Toward safe control through proof and learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.12107"},{"issue":"1","key":"1246_CR62","first-page":"1437","volume":"16","author":"J Garc\u0131a","year":"2015","unstructured":"Garc\u0131a J, Fern\u00e1ndez F (2015) A comprehensive survey on safe reinforcement learning. J Mach Learn Res 16(1):1437\u20131480","journal-title":"J Mach Learn Res"},{"key":"1246_CR63","unstructured":"Garnelo M, Arulkumaran K, Shanahan M (2016) Towards deep symbolic reinforcement learning. arXiv preprint arXiv:1609.05518"},{"issue":"11","key":"1246_CR64","doi-asserted-by":"publisher","first-page":"745","DOI":"10.1016\/S2589-7500(21)00208-9","volume":"3","author":"M Ghassemi","year":"2021","unstructured":"Ghassemi M, Oakden-Rayner L, Beam AL (2021) The false hope of current approaches to explainable artificial intelligence in health care. The Lancet Digital Health 3(11):745\u2013750","journal-title":"The Lancet Digital Health"},{"key":"1246_CR65","doi-asserted-by":"crossref","unstructured":"Gilpin LH, Bau D, Yuan BZ, Bajwa A, Specter M, Kagal L (2018) Explaining explanations: an overview of interpretability of machine learning. In: IEEE 5th International Conference on Data Science and Advanced Analytics (DSAA). IEEE, pp 80\u201389","DOI":"10.1109\/DSAA.2018.00018"},{"key":"1246_CR66","unstructured":"Glanois C, Weng P, Zimmer M, Li D, Yang T, Hao J, Liu W (2021) A survey on interpretable reinforcement learning. arXiv preprint arXiv: 2112.13112"},{"key":"1246_CR67","unstructured":"Gorji R, Granmo S, O-C (2023) Off-policy and on-policy reinforcement learning with the tsetlin machine. Appl Intell 1\u201318"},{"key":"1246_CR68","doi-asserted-by":"publisher","unstructured":"Gosiewska A, Kozak A, Biecek P (2021) Simpler is better: lifting interpretability-performance trade-off via automated feature engineering. Decis Support Syst 150:113556. https:\/\/doi.org\/10.1016\/j.dss.2021.113556","DOI":"10.1016\/j.dss.2021.113556"},{"key":"1246_CR69","unstructured":"Granmo O-C (2018) The tsetlin machine-a game theoretic bandit driven approach to optimal pattern recognition with propositional logic. arXiv preprint arXiv: 1804.01508"},{"key":"1246_CR70","unstructured":"Greydanus S, Koul A, Dodge J, Fern A (2018) Visualizing and understanding atari agents. In: International Conference on Machine Learning. PMLR, pp 1792\u20131801"},{"key":"1246_CR71","unstructured":"Griffith S, Subramanian K, Scholz J, Isbell CL, Thomaz AL (2013) Policy shaping: Integrating human feedback with reinforcement learning. Adv Neural Inf Process Syst 26:"},{"key":"1246_CR72","unstructured":"Guan L, Verma M, Kambhampati S (2020) Explanation augmented feedback in human-in-the-loop reinforcement learning. arXiv preprint arXiv: 2006.14804"},{"key":"1246_CR73","first-page":"21885","volume":"34","author":"L Guan","year":"2021","unstructured":"Guan L, Verma M, Guo SS, Zhang R, Kambhampati S (2021) Widening the pipeline in human-guided reinforcement learning with explanation and context-aware data augmentation. Adv Neural Inf Process Syst 34:21885\u201321897","journal-title":"Adv Neural Inf Process Syst"},{"issue":"2","key":"1246_CR74","doi-asserted-by":"publisher","first-page":"859","DOI":"10.1109\/TCYB.2022.3163816","volume":"53","author":"Y Guan","year":"2023","unstructured":"Guan Y, Ren Y, Sun Q, Li SE, Ma H, Duan J, Dai Y, Cheng B (2023) Integrated decision and control: toward interpretable and computationally efficient driving intelligence. IEEE Trans Cybernetics 53(2):859\u2013873. https:\/\/doi.org\/10.1109\/TCYB.2022.3163816","journal-title":"IEEE Trans Cybernetics"},{"key":"1246_CR75","first-page":"12222","volume":"34","author":"W Guo","year":"2021","unstructured":"Guo W, Wu X, Khan U, Xing X (2021) Edge: Explaining deep reinforcement learning policies. Adv Neural Inf Process Syst 34:12222\u201312236","journal-title":"Adv Neural Inf Process Syst"},{"key":"1246_CR76","doi-asserted-by":"publisher","unstructured":"Hailu G, Sommer G (1998) Integrating symbolic knowledge in reinforcement learning. In: SMC\u201998 conference proceedings. 1998 IEEE international conference on systems, man, and cybernetics (Cat. No.98CH36218), vol 2. pp 1491\u201314962. https:\/\/doi.org\/10.1109\/ICSMC.1998.728096","DOI":"10.1109\/ICSMC.1998.728096"},{"key":"1246_CR77","first-page":"75","volume-title":"The cma evolution strategy: a comparing review","author":"N Hansen","year":"2006","unstructured":"Hansen N (2006) The cma evolution strategy: a comparing review. In: Advances in the estimation of distribution algorithms, Towards a new evolutionary computation, pp 75\u2013102"},{"key":"1246_CR78","doi-asserted-by":"crossref","unstructured":"Hayes B, Shah JA (2017) Improving robot controller transparency through autonomous policy explanation. In: Proceedings of the 2017 ACM\/IEEE International Conference on Human-robot Interaction. pp 303\u2013312","DOI":"10.1145\/2909824.3020233"},{"key":"1246_CR79","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1016\/j.engappai.2017.07.005","volume":"65","author":"D Hein","year":"2017","unstructured":"Hein D, Hentschel A, Runkler T, Udluft S (2017) Particle swarm optimization for generating interpretable fuzzy reinforcement learning policies. Eng Appl Artif Intell 65:87\u201398","journal-title":"Eng Appl Artif Intell"},{"key":"1246_CR80","doi-asserted-by":"publisher","first-page":"158","DOI":"10.1016\/j.engappai.2018.09.007","volume":"76","author":"D Hein","year":"2018","unstructured":"Hein D, Udluft S, Runkler TA (2018) Interpretable policies for reinforcement learning by genetic programming. Eng Appl Artif Intell 76:158\u2013169","journal-title":"Eng Appl Artif Intell"},{"key":"1246_CR81","doi-asserted-by":"crossref","unstructured":"Hein D, Udluft S, Runkler TA (2019) Generating interpretable reinforcement learning policies using genetic programming. In: Proceedings of the Genetic and Evolutionary Computation Conference Companion. pp 23\u201324","DOI":"10.1145\/3319619.3326755"},{"key":"1246_CR82","doi-asserted-by":"crossref","unstructured":"Hessel M, Modayil J, Van Hasselt H, Schaul T, Ostrovski G, Dabney W, Horgan D, Piot B, Azar M, Silver D (2018) Rainbow: combining improvements in deep reinforcement learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"1246_CR83","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2020.106685","volume":"214","author":"A Heuillet","year":"2021","unstructured":"Heuillet A, Couthouis F, D\u00edaz-Rodr\u00edguez N (2021) Explainability in deep reinforcement learning. Knowl-Based Syst 214:106685","journal-title":"Knowl-Based Syst"},{"issue":"1","key":"1246_CR84","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1109\/MCI.2021.3129959","volume":"17","author":"A Heuillet","year":"2022","unstructured":"Heuillet A, Couthouis F, D\u00edaz-Rodr\u00edguez N (2022) Collective explainable ai: explaining cooperative strategies and agent contribution in multiagent reinforcement learning with shapley values. IEEE Comput Intell Mag 17(1):59\u201371","journal-title":"IEEE Comput Intell Mag"},{"key":"1246_CR85","unstructured":"Hoffman RR, Mueller ST, Klein G, Litman J (2018) Metrics for explainable AI: challenges and prospects. arXiv preprint arXiv: 1812.04608"},{"key":"1246_CR86","doi-asserted-by":"crossref","unstructured":"Holzinger A (2018) From machine learning to explainable AI. In: World symposium on digital intelligence for systems and machines (DISA). IEEE. pp 55\u201366","DOI":"10.1109\/DISA.2018.8490530"},{"key":"1246_CR87","doi-asserted-by":"crossref","unstructured":"Holzinger A, Goebel R, Fong R, Moon T, M\u00fcller K-R, Samek W (2020) XXAI-beyond explainable artificial intelligence. In: International workshop on extending explainable AI beyond deep models and classifiers. Springer, Cham, pp 3\u201310","DOI":"10.1007\/978-3-031-04083-2_1"},{"key":"1246_CR88","doi-asserted-by":"crossref","unstructured":"Hostetter JW, Abdelshiheed M, Barnes T, CM (2023) Leveraging fuzzy logic towards more explainable reinforcement learning-induced pedagogical policies on intelligent tutoring systems. In: 2023 IEEE international conference on fuzzy systems","DOI":"10.1109\/FUZZ52849.2023.10309741"},{"key":"1246_CR89","doi-asserted-by":"crossref","unstructured":"Huang SH, Bhatia K, Abbeel P, Dragan AD (2018) Establishing appropriate trust via critical states. In: IEEE\/RSJ international conference on intelligent robots and systems (IROS). IEEE. pp 3929\u20133936","DOI":"10.1109\/IROS.2018.8593649"},{"key":"1246_CR90","doi-asserted-by":"publisher","first-page":"309","DOI":"10.1007\/s10514-018-9771-0","volume":"43","author":"SH Huang","year":"2019","unstructured":"Huang SH, Held D, Abbeel P, Dragan AD (2019) Enabling robots to communicate their objectives. Auton Robot 43:309\u2013326","journal-title":"Auton Robot"},{"key":"1246_CR91","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.101834","volume":"98","author":"Z Huang","year":"2023","unstructured":"Huang Z, Sun S, Zhao J, Mao L (2023) Multi-modal policy fusion for end-to-end autonomous driving. Inform Fusion 98:101834","journal-title":"Inform Fusion"},{"issue":"7","key":"1246_CR92","doi-asserted-by":"publisher","first-page":"12655","DOI":"10.1109\/TNNLS.2024.3443082","volume":"36","author":"Z Huang","year":"2024","unstructured":"Huang Z, Zhao J, Sun S (2024) De-pessimism offline reinforcement learning via value compensation. IEEE Trans Neural Netw Learn Syst 36(7):12655\u201312667","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"1246_CR93","doi-asserted-by":"crossref","unstructured":"Huber T, Schiller D, Andr\u00e9 E (2019) Enhancing explainability of deep reinforcement learning through selective layer-wise relevance propagation. In: KI 2019: advances in artificial intelligence: 42nd German conference on AI, Kassel, Germany, September 23\u201326, 2019, Proceedings 42. Springer, pp 188\u2013202","DOI":"10.1007\/978-3-030-30179-8_16"},{"key":"1246_CR94","doi-asserted-by":"publisher","DOI":"10.3389\/frai.2022.903875","volume":"5","author":"T Huber","year":"2022","unstructured":"Huber T, Limmer B, Andr\u00e9 E (2022) Benchmarking perturbation-based saliency maps for explaining atari agents. Front Artif Intell 5:903875","journal-title":"Frontiers Artif Intell"},{"issue":"2","key":"1246_CR95","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3054912","volume":"50","author":"A Hussein","year":"2017","unstructured":"Hussein A, Gaber MM, Elyan E, Jayne C (2017) Imitation learning: a survey of learning methods. ACM Comput Surv (CSUR) 50(2):1\u201335","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"1246_CR96","doi-asserted-by":"crossref","unstructured":"Iyer R, Li Y, Li H, Lewis M, Sundar R, Sycara K (2018) Transparency and explanation in deep reinforcement learning neural networks. In: Proceedings of the 2018 AAAI\/ACM Conference on AI, Ethics, and Society, pp 144\u2013150","DOI":"10.1145\/3278721.3278776"},{"key":"1246_CR97","doi-asserted-by":"crossref","unstructured":"Jaunet T, Vuillemot R, Wolf C (2020) Drlviz: Understanding decisions and memory in deep reinforcement learning. Comput Graph Forum 39:49\u201361","DOI":"10.1111\/cgf.13962"},{"key":"1246_CR98","doi-asserted-by":"crossref","unstructured":"Jin P, Tian J, Zhi D, Wen X, Zhang M (2022) Trainify: a cegar-driven training and verification framework for safe deep reinforcement learning. In: International conference on computer aided verification. Springer, Cham, pp 193\u2013218","DOI":"10.1007\/978-3-031-13185-1_10"},{"key":"1246_CR99","unstructured":"Juozapaitis Z, Koul A, Fern A, Erwig M, Doshi-Velez F (2019) Explainable reinforcement learning via reward decomposition. In: IJCAI\/ECAI workshop on explainable artificial intelligence"},{"key":"1246_CR100","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"LP Kaelbling","year":"1996","unstructured":"Kaelbling LP, Littman ML, Moore AW (1996) Reinforcement learning: a survey. J Artif Intell Res 4:237\u2013285","journal-title":"J Artif Intell Res"},{"key":"1246_CR101","doi-asserted-by":"crossref","unstructured":"Kaptein F, Broekens J, Hindriks K, Neerincx M (2017) The role of emotion in self-explanations by cognitive agents. In: 2017 seventh international conference on affective computing and intelligent interaction workshops and demos (ACIIW). IEEE, pp 88\u201393","DOI":"10.1109\/ACIIW.2017.8272595"},{"key":"1246_CR102","doi-asserted-by":"crossref","unstructured":"Katz G, Barrett C, Dill DL, Julian K, Kochenderfer MJ (2017) Reluplex: An efficient smt solver for verifying deep neural networks. In: Computer aided verification: 29th international conference, CAV 2017, Heidelberg, Germany, July 24\u201328, 2017, Proceedings, Part I 30. Springer, pp 97\u2013117","DOI":"10.1007\/978-3-319-63387-9_5"},{"key":"1246_CR103","doi-asserted-by":"crossref","unstructured":"Katz G, Huang DA, Ibeling D, Julian K, Lazarus C, Lim R, Shah P, Thakoor S, Wu H, Zelji\u0107 A, et\u00a0al (2019) The marabou framework for verification and analysis of deep neural networks. In: Computer aided verification: 31st international conference, CAV 2019, New York City, NY, USA, July 15\u201318, 2019, Proceedings, Part I 31. Springer, pp. 443\u2013452","DOI":"10.1007\/978-3-030-25540-4_26"},{"key":"1246_CR104","doi-asserted-by":"crossref","unstructured":"Kazak Y, Barrett C, Katz G, Schapira M (2019) Verifying deep-rl-driven systems. In: Proceedings of the 2019 workshop on network meets AI & ML, pp 83\u201389","DOI":"10.1145\/3341216.3342218"},{"key":"1246_CR105","doi-asserted-by":"crossref","unstructured":"Kazhdan D, Shams Z, Li\u00f2 P (2020) Marleme: A multi-agent reinforcement learning model extraction library. In: 2020 international joint conference on neural networks (IJCNN). IEEE, pp 1\u20138","DOI":"10.1109\/IJCNN48605.2020.9207564"},{"key":"1246_CR106","doi-asserted-by":"publisher","first-page":"194","DOI":"10.1609\/icaps.v19i1.13365","volume":"19","author":"O Khan","year":"2009","unstructured":"Khan O, Poupart P, Black J (2009) Minimal sufficient explanations for factored markov decision processes. In: Proceedings of the international conference on automated planning and scheduling 19:194\u2013200","journal-title":"Proceedings of the International Conference on Automated Planning and Scheduling"},{"key":"1246_CR107","unstructured":"Kim B, Khanna R, Koyejo OO (2016) Examples are not enough, learn to criticize! criticism for interpretability. Adv Neural Inf Process Syst 29"},{"key":"1246_CR108","doi-asserted-by":"crossref","unstructured":"Kim J, Rohrbach A, Darrell T, Canny J, Akata Z (2018) Textual explanations for self-driving vehicles. In: Proceedings of the European conference on computer vision (ECCV). pp 563\u2013578","DOI":"10.1007\/978-3-030-01216-8_35"},{"key":"1246_CR109","doi-asserted-by":"crossref","unstructured":"Kim J, Moon S, Rohrbach A, Darrell T, Canny J (2020) Advisable learning for self-driving vehicles by internalizing observation-to-action rules. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 9661\u20139670","DOI":"10.1109\/CVPR42600.2020.00968"},{"issue":"2","key":"1246_CR110","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1109\/MRA.2010.936952","volume":"17","author":"J Kober","year":"2010","unstructured":"Kober J, Peters J (2010) Imitation and reinforcement learning. IEEE Robot Autom Mag 17(2):55\u201362","journal-title":"IEEE Robotics & Automation Magazine"},{"key":"1246_CR111","unstructured":"Koh PW, Liang P (2017) Understanding black-box predictions via influence functions. In: International conference on machine learning. PMLR, pp 1885\u20131894"},{"key":"1246_CR112","unstructured":"Koul A, Greydanus S, Fern A (2018) Learning finite state representations of recurrent policy networks. arXiv preprint arXiv:1811.12530"},{"key":"1246_CR113","unstructured":"Kulkarni TD, Gupta A, Ionescu C, Borgeaud S, Reynolds M, Zisserman A, Mnih V (2019) Unsupervised learning of object keypoints for perception and control. CoRR abs\/1906.11883 arXiv: 1906.11883"},{"key":"1246_CR114","unstructured":"Kumar S, Vishal M, Ravi V (2022) Explainable reinforcement learning on financial stock trading using shap. arXiv preprint arXiv:2208.08790"},{"key":"1246_CR115","doi-asserted-by":"crossref","unstructured":"Lachman R, Joffe M (2021) Analyzing Future Applications of AI. Sensors, and Robotics in Society. IGI Global, pp 201\u2013220 (Applications of artificial intelligence in media and entertainment)","DOI":"10.4018\/978-1-7998-3499-1.ch012"},{"key":"1246_CR116","doi-asserted-by":"crossref","unstructured":"Lage I, Lifschitz D, Doshi-Velez F, Amir O (2019) Exploring computational user models for agent policy summarization. In: IJCAI: Proceedings of the Conference, vol. 28, p 1401. NIH Public Access","DOI":"10.24963\/ijcai.2019\/194"},{"key":"1246_CR117","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1017\/S0140525X16001837","volume":"40","author":"BM Lake","year":"2017","unstructured":"Lake BM, Ullman TD, Tenenbaum JB, Gershman SJ (2017) Building machines that learn and think like people. Behavioral and Brain Sciences 40:253","journal-title":"Behavioral and Brain Sciences"},{"key":"1246_CR118","unstructured":"Landajuela M, Petersen BK, Kim S, Santiago CP, Glatt R, Mundhenk N, Pettit JF Faissol D (2021) Discovering symbolic policies with deep reinforcement learning. International Conference on Machine Learning. PMLR, pp 5979\u20135989"},{"key":"1246_CR119","doi-asserted-by":"crossref","unstructured":"Li J, Kuang K, Wang B, Liu F, Chen L, Wu F, Xiao J (2021) Shapley counterfactual credits for multi-agent reinforcement learning. In: Proceedings of the 27th ACM SIGKDD Conference on Knowledge Discovery & Data Mining, pp 934\u2013942","DOI":"10.1145\/3447548.3467420"},{"issue":"12","key":"1246_CR120","doi-asserted-by":"publisher","first-page":"3197","DOI":"10.1007\/s10115-022-01756-8","volume":"64","author":"X Li","year":"2022","unstructured":"Li X, Xiong H, Li X, Wu X, Zhang X, Liu J, Bian J, Dou D (2022) Interpretable deep learning: Interpretation, interpretability, trustworthiness, and beyond. Knowl Inf Syst 64(12):3197\u20133234","journal-title":"Knowl Inf Syst"},{"key":"1246_CR121","unstructured":"Lillicrap TP, Hunt JJ, Pritzel A, Heess N, Erez T, Tassa Y, Silver D, Wierstra D (2015) Continuous control with deep reinforcement learning. arXiv preprint arXiv: 1509.02971"},{"key":"1246_CR123","unstructured":"Lin Z, Sun S (2025) Revealing the challenges of sim-to-real transfer in model-based reinforcement learning via latent space modeling. arXiv preprint arXiv:2506.12735"},{"key":"1246_CR122","unstructured":"Lin Z, Lam K-H, Fern A (2020) Contrastive explanations for reinforcement learning via embedded self predictions. arXiv preprint arXiv: 2010.05180"},{"key":"1246_CR124","doi-asserted-by":"publisher","unstructured":"Lin B, Cecchi G, Bouneffouf D (2023) Psychotherapy ai companion with reinforcement learning recommendations and interpretable policy dynamics. In: WWW \u201923 Companion. Association for Computing Machinery, New York, NY, USA, pp. 932\u2013939. https:\/\/doi.org\/10.1145\/3543873.3587623","DOI":"10.1145\/3543873.3587623"},{"issue":"3","key":"1246_CR125","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1145\/3236386.3241340","volume":"16","author":"ZC Lipton","year":"2018","unstructured":"Lipton ZC (2018) The mythos of model interpretability: in machine learning, the concept of interpretability is both important and slippery. Queue 16(3):31\u201357","journal-title":"Queue"},{"issue":"1","key":"1246_CR126","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1016\/j.visinf.2017.01.006","volume":"1","author":"S Liu","year":"2017","unstructured":"Liu S, Wang X, Liu M, Zhu J (2017) Towards better analysis of machine learning models: a visual analytics perspective. Visual Inform 1(1):48\u201356","journal-title":"Visual Inform"},{"key":"1246_CR127","doi-asserted-by":"crossref","unstructured":"Liu G, Schulte O, Zhu W, Li Q (2019) Toward interpretable deep reinforcement learning with linear model u-trees. In: Machine learning and knowledge discovery in databases: European conference, ECML PKDD 2018, Dublin, Ireland, September 10\u201314, 2018, Proceedings, Part II 18. Springer, pp. 414\u2013429","DOI":"10.1007\/978-3-030-10928-8_25"},{"key":"1246_CR128","doi-asserted-by":"publisher","unstructured":"Liu X, Liu S, An B, Gao Y, Yang S, Li W (2023) Effective interpretable policy distillation via critical experience point identification. IEEE Intell Syst 38:1\u201310. https:\/\/doi.org\/10.1109\/MIS.2023.3265868","DOI":"10.1109\/MIS.2023.3265868"},{"key":"1246_CR129","doi-asserted-by":"crossref","unstructured":"Lomas M, Chevalier R, Cross EV, Garrett RC, Hoare J, Kopack M (2012) Explaining robot actions. In: Proceedings of the seventh annual ACM\/IEEE international conference on human-robot interaction. pp 187\u2013188","DOI":"10.1145\/2157689.2157748"},{"key":"1246_CR130","unstructured":"Lowe R, Wu YI, Tamar A, Harb J, Pieter Abbeel O, Mordatch I (2017) Multi-agent actor-critic for mixed cooperative-competitive environments. Adv Neural Inform Process Syst 30"},{"key":"1246_CR131","unstructured":"Lundberg SM, Lee S-I (2017) A unified approach to interpreting model predictions. Adv Neural Inform Process Syst 30"},{"key":"1246_CR132","doi-asserted-by":"crossref","unstructured":"L\u00fctjens B, Everett M, How JP (2019) Safe reinforcement learning with model uncertainty estimates. In: International conference on robotics and automation (ICRA). IEEE. pp 8662\u20138668","DOI":"10.1109\/ICRA.2019.8793611"},{"key":"1246_CR133","doi-asserted-by":"publisher","first-page":"2970","DOI":"10.1609\/aaai.v33i01.33012970","volume":"33","author":"D Lyu","year":"2019","unstructured":"Lyu D, Yang F, Liu B, Gustafson S (2019) Sdrl: interpretable and data-efficient deep reinforcement learning leveraging symbolic planning. Proc AAAI Conf Artif Intell 33:2970\u20132977","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR134","unstructured":"Maaten L, Hinton G (2008) Visualizing data using t-sne. J Mach Learn Res 9(11):2579\u20132605"},{"key":"1246_CR135","doi-asserted-by":"publisher","first-page":"2493","DOI":"10.1609\/aaai.v34i03.5631","volume":"34","author":"P Madumal","year":"2020","unstructured":"Madumal P, Miller T, Sonenberg L, Vetere F (2020) Explainable reinforcement learning through a causal lens. Proc AAAI Conf Artif Intell 34:2493\u20132500","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR136","doi-asserted-by":"crossref","unstructured":"Maes F, Fonteneau R, Wehenkel L, Ernst D (2012) Policy search in a space of simple closed-form formulas: towards interpretability of reinforcement learning. In: Discovery science: 15th international conference, DS 2012, Lyon, France, October 29\u201331, 2012. Proceedings 15. Springer, pp 37\u201351","DOI":"10.1007\/978-3-642-33492-4_6"},{"key":"1246_CR137","unstructured":"Mahmood AR, Korenkevych D, Vasan G, Ma W, Bergstra J (2018) Benchmarking reinforcement learning algorithms on real-world robots. In: Conference on robot learning. PMLR, pp 561\u2013591"},{"key":"1246_CR138","unstructured":"Mania H, Guy A, Recht B (2018) Simple random search of static linear policies is competitive for reinforcement learning. Adv Neural Inf Process Syst 31"},{"key":"1246_CR139","unstructured":"Mansour Y, Moshkovitz M, Rudin C (2022) There is no accuracy-interpretability tradeoff in reinforcement learning for mazes. arXiv preprint arXiv:2206.04266"},{"key":"1246_CR140","doi-asserted-by":"crossref","unstructured":"Mao H, Netravali R, Alizadeh M (2017) Neural adaptive video streaming with pensieve. In: Proceedings of the conference of the ACM special interest group on data communication. pp 197\u2013210","DOI":"10.1145\/3098822.3098843"},{"key":"1246_CR141","doi-asserted-by":"crossref","unstructured":"Mash M, Lin R, Sarne D (2014) Peer-design agents for reliably evaluating distribution of outcomes in environments involving people. In: Proceedings of the 2014 international conference on autonomous agents and multi-agent systems. pp 949\u2013956","DOI":"10.65109\/PKBW5907"},{"key":"1246_CR142","unstructured":"Matarese M, Rossi S, Sciutti A, Rea F (2020) Towards transparency of TD-RL robotic systems with a human teacher. arXiv preprint arXiv: 2005.05926"},{"key":"1246_CR143","doi-asserted-by":"crossref","unstructured":"Melloni D, Zingoni A (2024) Interpreting type 1 diabetes management via contrastive explanations. In: 2024 IEEE international conference on metrology for extended reality, artificial intelligence and neural engineering (MetroXRAINE). IEEE, pp 692\u2013697","DOI":"10.1109\/MetroXRAINE62247.2024.10796610"},{"key":"1246_CR144","doi-asserted-by":"crossref","unstructured":"Melloni D, Zingoni A (2025) Adopting post-hoc explainable reinforcement learning in healthcare scenarios. In: 2025 IEEE international conference on metrology for extended reality, artificial intelligence and neural engineering (MetroXRAINE). IEEE, pp 536\u2013541","DOI":"10.1109\/MetroXRAINE66377.2025.11340444"},{"key":"1246_CR145","unstructured":"Milani S, Topin N, Veloso M, Fang F (2022a) A survey of explainable reinforcement learning. arXiv preprint arXiv: 2202.08434"},{"key":"1246_CR146","doi-asserted-by":"crossref","unstructured":"Milani S, Zhang Z, Topin N, Shi ZR, Kamhoua C, Papalexakis EE, Fang F (2022b) Maviper: Learning decision tree policies for interpretable multi-agent reinforcement learning. In: Joint European conference on machine learning and knowledge discovery in databases. Springer, pp 251\u2013266","DOI":"10.1007\/978-3-031-26412-2_16"},{"key":"1246_CR147","doi-asserted-by":"crossref","unstructured":"Milani R, Moll M, Leone D, Pickl R, S (2023) A bayesian network approach to explainable reinforcement learning with distal information. Sensors 23(4):2013","DOI":"10.3390\/s23042013"},{"key":"1246_CR148","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.artint.2018.07.007","volume":"267","author":"T Miller","year":"2019","unstructured":"Miller T (2019) Explanation in artificial intelligence: insights from the social sciences. Artif Intell 267:1\u201338","journal-title":"Artif Intell"},{"key":"1246_CR149","doi-asserted-by":"crossref","unstructured":"Mishra I, Dao G, Lee M (2018) Visual sparse bayesian reinforcement learning: a framework for interpreting what an agent has learned. In: IEEE symposium series on computational intelligence (SSCI). IEEE, pp 1427\u20131434","DOI":"10.1109\/SSCI.2018.8628887"},{"key":"1246_CR152","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller M (2013) Playing atari with deep reinforcement learning. arXiv preprint arXiv: 1312.5602"},{"issue":"7540","key":"1246_CR150","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"1246_CR151","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap T, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. In: International conference on machine learning. PMLR, pp 1928\u20131937"},{"key":"1246_CR153","doi-asserted-by":"crossref","unstructured":"Montavon G, Binder A, Lapuschkin S, Samek W, M\u00fcller K-R (2019) Layer-wise relevance propagation: an overview. In: Explainable AI: interpreting, explaining and visualizing deep learning. pp 193\u2013209","DOI":"10.1007\/978-3-030-28954-6_10"},{"key":"1246_CR154","unstructured":"Mott A, Zoran D, Chrzanowski M, Wierstra D, Jimenez Rezende D (2019) Towards interpretable reinforcement learning using attention augmented agents. Adv Neural Inf Process Syst 32:"},{"key":"1246_CR155","doi-asserted-by":"crossref","unstructured":"Movin M, Junior GD, Hollm\u00e9n J, Papapetrou P (2023) Explaining black box reinforcement learning agents through counterfactual policies. In: International symposium on intelligent data analysis. Springer, pp 314\u2013326","DOI":"10.1007\/978-3-031-30047-9_25"},{"key":"1246_CR156","doi-asserted-by":"crossref","unstructured":"Mujtaba DF, Mahapatra NR (2019) Ethical considerations in ai-based recruitment. In: 2019 IEEE international symposium on technology and society (ISTAS). IEEE, pp 1\u20137","DOI":"10.1109\/ISTAS48451.2019.8937920"},{"key":"1246_CR157","unstructured":"Nachum O, Norouzi M, Xu K, Schuurmans D (2017) Bridging the gap between value and policy based reinforcement learning. Adv Neural Inform Process Syst 30"},{"key":"1246_CR158","first-page":"9497","volume":"33","author":"G Nangue Tasse","year":"2020","unstructured":"Nangue Tasse G, James S, Rosman B (2020) A boolean task algebra for reinforcement learning. Adv Neural Inf Process Syst 33:9497\u20139507","journal-title":"Adv Neural Inf Process Syst"},{"key":"1246_CR159","doi-asserted-by":"crossref","unstructured":"Neerincx MA, Waa J, Kaptein F, Diggelen J (2018) Using perceptual and cognitive explanations for enhanced human-agent team performance. In: Engineering psychology and cognitive ergonomics: 15th international conference, EPCE 2018, Held as Part of HCI International 2018, Las Vegas, NV, USA, July 15\u201320, 2018, Proceedings 15. Springer, pp 204\u2013214","DOI":"10.1007\/978-3-319-91122-9_18"},{"key":"1246_CR160","unstructured":"Ng AY, Russell S, et\u00a0al (2000) Algorithms for inverse reinforcement learning. In: ICML, vol. 1, p 2"},{"key":"1246_CR161","doi-asserted-by":"crossref","unstructured":"Nikulin D, Ianina A, Aliev V, Nikolenko S (2019) Free-lunch saliency via attention in ATARI agents. In: IEEE\/CVF international conference on computer vision workshop (ICCVW). IEEE. pp 4240\u20134249","DOI":"10.1109\/ICCVW.2019.00522"},{"key":"1246_CR162","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1007\/s11187-019-00202-4","volume":"55","author":"M Obschonka","year":"2020","unstructured":"Obschonka M, Audretsch DB (2020) Artificial intelligence and big data in entrepreneurship: a new era has begun. Small Bus Econ 55:529\u2013539","journal-title":"Small Bus Econ"},{"key":"1246_CR163","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103455","volume":"295","author":"ML Olson","year":"2021","unstructured":"Olson ML, Khanna R, Neal L, Li F, Wong W-K (2021) Counterfactual state explanations for reinforcement learning agents via generative deep learning. Artif Intell 295:103455","journal-title":"Artif Intell"},{"key":"1246_CR164","doi-asserted-by":"crossref","unstructured":"O\u2019Sullivan S, Nevejans N, Allen C, Blyth A, Leonard S, Pagallo U, Holzinger K, Holzinger A, Sajid MI, Ashrafian H (2019) Legal, regulatory, and ethical frameworks for development of standards in artificial intelligence (ai) and autonomous robotic surgery. Int J Med Robot Comput Assist Surg 15(1):1968","DOI":"10.1002\/rcs.1968"},{"key":"1246_CR165","doi-asserted-by":"crossref","unstructured":"Pan X, Chen X, Cai Q, Canny J, Yu F (2019) Semantic predictive control for explainable and efficient policy learning. In: International conference on robotics and automation (ICRA). IEEE, pp 3203\u20133209","DOI":"10.1109\/ICRA.2019.8794437"},{"key":"1246_CR166","doi-asserted-by":"crossref","unstructured":"Pan, M., Huang W, Li Y, Zhou X, Luo J (2020) XGAIL: Explainable generative adversarial imitation learning for explainable human decision analysis. In: Proceedings of the 26th ACM SIGKDD international conference on knowledge discovery & data mining. pp 1334\u20131343","DOI":"10.1145\/3394486.3403186"},{"key":"1246_CR167","unstructured":"Petrik M, Luss R (2016) Interpretable policies for dynamic product recommendations. In: UAI"},{"key":"1246_CR168","unstructured":"Petsiuk V, Das A, Saenko K (2018) Rise: Randomized input sampling for explanation of black-box models. arXiv preprint arXiv: 1806.07421"},{"key":"1246_CR169","doi-asserted-by":"crossref","unstructured":"Pirsiavash H, Ramanan D (2014) Parsing videos of actions with segmental grammars. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp 612\u2013619","DOI":"10.1109\/CVPR.2014.85"},{"key":"1246_CR170","first-page":"10526","volume":"33","author":"G Plumb","year":"2020","unstructured":"Plumb G, Al-Shedivat M, Cabrera \u00c1A, Perer A, Xing E, Talwalkar A (2020) Regularizing black-box models for improved interpretability. Adv Neural Inf Process Syst 33:10526\u201310536","journal-title":"Adv Neural Inf Process Syst"},{"key":"1246_CR171","doi-asserted-by":"publisher","first-page":"10007","DOI":"10.1609\/aaai.v33i01.330110007","volume":"33","author":"R Pocius","year":"2019","unstructured":"Pocius R, Neal L, Fern A (2019) Strategic tasks for explainable reinforcement learning. Proc AAAI Conf Artif Intell 33:10007\u201310008","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR172","doi-asserted-by":"crossref","unstructured":"Pu P, Chen L (2006) Trust building with explanation interfaces. In: Proceedings of the 11th international conference on intelligent user interfaces. pp 93\u2013100","DOI":"10.1145\/1111449.1111475"},{"key":"1246_CR173","doi-asserted-by":"crossref","unstructured":"Puiutta E, Veith EM (2020) Explainable reinforcement learning: a survey. In: Machine learning and knowledge extraction: 4th IFIP TC 5, TC 12, WG 8.4, WG 8.9, WG 12.9 international cross-domain conference, CD-MAKE 2020, Dublin, Ireland, August 25\u201328, 2020, Proceedings 4. Springer, pp 77\u201395","DOI":"10.1007\/978-3-030-57321-8_5"},{"key":"1246_CR174","unstructured":"Pyeatt LD, Howe AE (2001) Decision tree function approximation in reinforcement learning. In: Proceedings of the third international symposium on adaptive systems: evolutionary computation and probabilistic graphical models, vol 2. Cuba, pp 70\u201377"},{"key":"1246_CR175","unstructured":"Qing Y, Liu S, Song J, Song M (2022) A survey on explainable reinforcement learning: concepts, algorithms, challenges. arXiv preprint arXiv:2211.06665"},{"key":"1246_CR176","doi-asserted-by":"crossref","unstructured":"Ribeiro MT, Singh, S, Guestrin C (2016) \u201cwhy should i trust you?\u201d explaining the predictions of any classifier. In: Proceedings of the 22nd ACM SIGKDD international conference on knowledge discovery and data mining, pp 1135\u20131144","DOI":"10.1145\/2939672.2939778"},{"key":"1246_CR177","unstructured":"Ross S, Gordon G, Bagnell D (2011) A reduction of imitation learning and structured prediction to no-regret online learning. In: Proceedings of the fourteenth international conference on artificial intelligence and statistics (JMLR Workshop and Conference Proceedings). pp 627\u2013635"},{"key":"1246_CR178","unstructured":"Roth AM, Topin N, Jamshidi P, Veloso M (2019) Conservative q-improvement: reinforcement learning for an interpretable decision-tree policy. arXiv preprint arXiv: 1907.01180"},{"key":"1246_CR179","doi-asserted-by":"crossref","unstructured":"Rudin C (2019) Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead. Nat Mach Intell 1(5):206\u2013215","DOI":"10.1038\/s42256-019-0048-x"},{"issue":"6","key":"1246_CR180","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1016\/S1364-6613(99)01327-3","volume":"3","author":"S Schaal","year":"1999","unstructured":"Schaal S (1999) Is imitation learning the route to humanoid robots? Trends Cogn Sci 3(6):233\u2013242","journal-title":"Trends Cogn Sci"},{"key":"1246_CR181","first-page":"298","volume":"298","author":"A Schwartz","year":"1993","unstructured":"Schwartz A (1993) A reinforcement learning method for maximizing undiscounted rewards. In: Proceedings of the tenth international conference on machine learning 298:298\u2013305","journal-title":"Proceedings of the Tenth International Conference on Machine Learning"},{"key":"1246_CR182","doi-asserted-by":"crossref","unstructured":"Selvaraju RR, Cogswell M, Das A, Vedantam R, Parikh D, Batra D (2017) Grad-cam: Visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE international conference on computer vision. pp 618\u2013626","DOI":"10.1109\/ICCV.2017.74"},{"key":"1246_CR183","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2020.103367","volume":"288","author":"P Sequeira","year":"2020","unstructured":"Sequeira P, Gervasio M (2020) Interestingness elements for explainable reinforcement learning: understanding agents\u2019 capabilities and limitations. Artif Intell 288:103367","journal-title":"Artif Intell"},{"issue":"1","key":"1246_CR184","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1207\/s15516709cog2901_5","volume":"29","author":"M Shanahan","year":"2005","unstructured":"Shanahan M (2005) Perception as abduction: turning sensor data into meaningful representation. Cogn Sci 29(1):103\u2013134","journal-title":"Cogn Sci"},{"key":"1246_CR185","unstructured":"Sheh RK (2017) Different XAI for different HRI. In: AAAI fall symposium series. pp 114\u2013117"},{"key":"1246_CR186","unstructured":"Shi W, Wang Z, Song S, Huang G (2020) Self-supervised discovering of causal features: towards interpretable reinforcement learning. arXiv preprint arXiv: 2003.07069"},{"key":"1246_CR187","doi-asserted-by":"crossref","unstructured":"Shi X, Zhang J, Liang Z, Seng D (2023) Maddpgviz: a visual analytics approach to understand multi-agent deep reinforcement learning. J Vis 26:1\u201317","DOI":"10.1007\/s12650-023-00928-0"},{"key":"1246_CR188","unstructured":"Shu T, Xiong C, Socher R (2017) Hierarchical and interpretable skill acquisition in multi-task reinforcement learning. arXiv preprint arXiv: 1712.07294"},{"key":"1246_CR189","doi-asserted-by":"crossref","unstructured":"Si Z, Pei M, Yao B, Zhu S-C (2011) Unsupervised learning of event and-or grammar and semantics from video. In: 2011 international conference on computer vision. IEEE, pp 41\u201348","DOI":"10.1109\/ICCV.2011.6126223"},{"key":"1246_CR190","unstructured":"Silva A, Killian T, Rodriguez IDJ, Son S-H, Gombolay M (2019) Optimization methods for interpretable differentiable decision trees in reinforcement learning. arXiv preprint arXiv: 1903.09338"},{"key":"1246_CR191","doi-asserted-by":"publisher","first-page":"10251","DOI":"10.1609\/aaai.v34i06.6587","volume":"34","author":"T Silver","year":"2020","unstructured":"Silver T, Allen KR, Lew AK, Kaelbling LP, Tenenbaum J (2020) Few-shot bayesian imitation learning with logical program policies. Proc AAAI Conf Artif Intell 34:10251\u201310258","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR192","unstructured":"Simonyan K, Vedaldi A, Zisserman A (2013) Deep inside convolutional networks: visualising image classification models and saliency maps. arXiv preprint arXiv:1312.6034"},{"key":"1246_CR193","doi-asserted-by":"publisher","first-page":"2641","DOI":"10.1007\/s10994-021-05963-2","volume":"110","author":"J Skirzy\u0144ski","year":"2021","unstructured":"Skirzy\u0144ski J, Becker F, Lieder F (2021) Automatic discovery of interpretable planning strategies. Mach Learn 110:2641\u20132683","journal-title":"Mach Learn"},{"key":"1246_CR194","doi-asserted-by":"crossref","unstructured":"Smith B, Khojandi A, Vasudevan R (2023) Bias in reinforcement learning: a review in healthcare applications. ACM Comput Surv 56(2):1\u201317.","DOI":"10.1145\/3609502"},{"key":"1246_CR195","unstructured":"Sodhani S, Zhang A, Pineau J (2021) Multi-task reinforcement learning with context-based representations. In: International conference on machine learning. PMLR, pp 9767\u20139779"},{"issue":"2","key":"1246_CR196","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1007\/s13218-020-00637-y","volume":"34","author":"K Sokol","year":"2020","unstructured":"Sokol K, Flach P (2020) One explanation does not fit all: the promise of interactive explanations for machine learning transparency. KI-K\u00fcnstliche Intell 34(2):235\u2013250","journal-title":"KI-K\u00fcnstliche Intelligenz"},{"key":"1246_CR197","doi-asserted-by":"crossref","unstructured":"Song Z, Wang Y, Qian P, Song S, Coenen F, Jiang Z, Su J (2022) From deterministic to stochastic: an interpretable stochastic model-free reinforcement learning framework for portfolio optimization. Appl Intell 1\u201316","DOI":"10.1007\/s10489-022-04217-5"},{"key":"1246_CR198","unstructured":"Sorokin I, Seleznev A, Pavlov M, Fedorov A, Ignateva A (2015) Deep attention recurrent q-network. arXiv preprint arXiv: 1512.01693"},{"key":"1246_CR199","doi-asserted-by":"crossref","unstructured":"Sreedharan S, Srivastava S, Kambhampati S (2018) Hierarchical expertise level modeling for user specific contrastive explanations. IJCAI 4829\u20134836","DOI":"10.24963\/ijcai.2018\/671"},{"key":"1246_CR200","first-page":"17493","volume":"34","author":"G Stein","year":"2021","unstructured":"Stein G (2021) Generating high-quality explanations for navigation in partially-revealed environments. Adv Neural Inf Process Syst 34:17493\u201317506","journal-title":"Adv Neural Inf Process Syst"},{"key":"1246_CR201","doi-asserted-by":"crossref","unstructured":"Su P-H, Budzianowski P, Ultes S, Gasic M, Young S (2017) Sample-efficient actor-critic reinforcement learning with supervised data for dialogue management. arXiv preprint arXiv: 1707.00130","DOI":"10.18653\/v1\/W17-5518"},{"key":"1246_CR202","unstructured":"Sukhbaatar S, Fergus R (2016) Learning multiagent communication with backpropagation. Adv Neural Inf Process Syst 29"},{"issue":"2","key":"1246_CR203","doi-asserted-by":"publisher","first-page":"1035","DOI":"10.1109\/TNNLS.2021.3107375","volume":"34","author":"Y Sun","year":"2023","unstructured":"Sun Y, Zhang K, Sun C (2023a) Model-based transfer reinforcement learning based on graphical model representations. IEEE Trans Neural Netw Learn Syst 34(2):1035\u20131048. https:\/\/doi.org\/10.1109\/TNNLS.2021.3107375","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"1246_CR204","doi-asserted-by":"publisher","unstructured":"Sun Y-W, Liu W-Z, Sun C-Y (2023b) Causality in reinforcement learning control: The state of the art and prospects. Zidonghua Xuebao\/Acta Automatica Sinica 49(3), 661\u2013677 https:\/\/doi.org\/10.16383\/j.aas.c220823. Cited by: 0","DOI":"10.16383\/j.aas.c220823"},{"key":"1246_CR205","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton RS, Barto AG (2018) Reinforcement learning: an introduction. MIT press, Cambridge, MA"},{"key":"1246_CR206","doi-asserted-by":"crossref","unstructured":"Sutton RS, Modayil J, Delp M, Degris T, Pilarski PM, White A, Precup D (2011) Horde: A scalable real-time architecture for learning knowledge from unsupervised sensorimotor interaction. In: The 10th international conference on autonomous agents and multiagent systems-volume 2. pp 761\u2013768","DOI":"10.65109\/QDHN7183"},{"key":"1246_CR207","doi-asserted-by":"crossref","unstructured":"Szymanski M, Millecamp M, Verbert K (2021) Visual, textual or hybrid: the effect of user expertise on different explanations. In: 26th international conference on intelligent user interfaces. pp 109\u2013119","DOI":"10.1145\/3397481.3450662"},{"key":"1246_CR208","doi-asserted-by":"crossref","unstructured":"Tabrez A, Hayes B (2019) Improving human-robot interaction through explainable reinforcement learning. in: 14th ACM\/IEEE international conference on human-robot interaction (HRI). IEEE, pp 751\u2013753","DOI":"10.1109\/HRI.2019.8673198"},{"key":"1246_CR209","unstructured":"Tacchetti A, Song HF, Mediano PA, Zambaldi V, Rabinowitz NC, Graepel T, Botvinick M, Battaglia PW (2018) Relational forward models for multi-agent learning. arXiv preprint arXiv: 1809.11044"},{"key":"1246_CR210","doi-asserted-by":"crossref","unstructured":"Tang Y, Nguyen D, Ha D (2020) Neuroevolution of self-interpretable agents. In: Proceedings of the 2020 genetic and evolutionary computation conference. pp 414\u2013424","DOI":"10.1145\/3377930.3389847"},{"key":"1246_CR211","unstructured":"Thomaz AL, Hoffman G, Breazeal C (2005) Real-time interactive reinforcement learning for robots. In: AAAI 2005 workshop on human comprehensible machine learning, vol 3"},{"issue":"1","key":"1246_CR212","doi-asserted-by":"publisher","first-page":"1638","DOI":"10.1038\/s41598-023-28804-9","volume":"13","author":"E Tjoa","year":"2023","unstructured":"Tjoa E, Guan C (2023) Self reward design with fine-grained interpretability. Sci Rep 13(1):1638","journal-title":"Sci Rep"},{"key":"1246_CR213","doi-asserted-by":"publisher","first-page":"2514","DOI":"10.1609\/aaai.v33i01.33012514","volume":"33","author":"N Topin","year":"2019","unstructured":"Topin N, Veloso M (2019) Generation of policy-level explanations for reinforcement learning. Proc AAAI Conf Artif Intell 33:2514\u20132521","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR214","doi-asserted-by":"crossref","unstructured":"Topin N, Milani S, Fang F, Veloso M (2021) Iterative bounding mdps: learning interpretable policies via non-interpretable method. In: Proceedings of the AAAI conference on artificial intelligence, vol 35. pp 9923\u20139931","DOI":"10.1609\/aaai.v35i11.17192"},{"key":"1246_CR215","first-page":"25146","volume":"34","author":"D Trivedi","year":"2021","unstructured":"Trivedi D, Zhang J, Sun S-H, Lim JJ (2021) Learning to synthesize programs as interpretable and generalizable policies. Adv Neural Inf Process Syst 34:25146\u201325163","journal-title":"Adv Neural Inf Process Syst"},{"issue":"3","key":"1246_CR216","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1080\/0952813X.2015.1020517","volume":"28","author":"O Tutsoy","year":"2016","unstructured":"Tutsoy O, Brown M (2016) An analysis of value function learning with piecewise linear control. J Exp Theor Artif Intell 28(3):529\u2013545","journal-title":"J Exp Theor Artif Intell"},{"key":"1246_CR217","first-page":"769","volume":"98","author":"WT Uther","year":"1998","unstructured":"Uther WT, Veloso MM (1998) Tree based discretization for continuous state space reinforcement learning. AAAI\/IAAI 98:769\u2013774","journal-title":"Aaai\/iaai"},{"key":"1246_CR218","unstructured":"Utomo CP, Li X, Chen W (2018) Treatment recommendation in critical care: a scalable and interpretable approach in partially observable health states. Cited by: 4. https:\/\/www.scopus.com\/inward\/record.uri?eid=2-s2.0-85062544443&partnerID=40&md5=58c90cc08abc34c0f6d2377052ede3f1"},{"key":"1246_CR219","doi-asserted-by":"crossref","unstructured":"Van Hasselt H, Guez A, Silver D (2016) Deep reinforcement learning with double q-learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 30","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"1246_CR220","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inform Process Syst 30"},{"key":"1246_CR221","doi-asserted-by":"crossref","unstructured":"Veith EM, Fischer L, Tr\u00f6schel M, Nie\u00dfe A (2019) Analyzing cyber-physical systems from the perspective of artificial intelligence. In: Proceedings of the 2019 international conference on artificial intelligence, robotics and control. pp 85\u201395","DOI":"10.1145\/3388218.3388222"},{"key":"1246_CR222","unstructured":"Verma A, Murali V, Singh R, Kohli P, Chaudhuri S (2018) Programmatically interpretable reinforcement learning. In: International conference on machine learning. PMLR, pp 5045\u20135054"},{"key":"1246_CR223","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1016\/j.inffus.2021.05.009","volume":"76","author":"G Vilone","year":"2021","unstructured":"Vilone G, Longo L (2021) Notions of explainability and evaluation approaches for explainable artificial intelligence. Inform Fusion 76:89\u2013106","journal-title":"Inform Fusion"},{"key":"1246_CR224","unstructured":"Vinyals O, Ewalds T, Bartunov S, Georgiev P, Vezhnevets AS, Yeo M, Makhzani A, K\u00fcttler H, Agapiou J, Schrittwieser J (2017) Starcraft II: a new challenge for reinforcement learning. arXiv preprint arXiv:1708.04782"},{"key":"1246_CR225","unstructured":"Vinyals O, Babuschkin I, Chung J, Mathieu M, Jaderberg M, Czarnecki W, Dudzik A, Huang A, Georgiev P, Powell R, Ewalds T, Horgan D, Kroiss M, Danihelka I, Agapiou J, Oh J, Dalibard V, Choi D, Sifre L, Sulsky Y, Vezhnevets S, Molloy J, Cai T, Budden D, Paine T, Gulcehre C, Wang Z, Pfaff T, Pohlen T, Yogatama D, Cohen J, McKinney K, Smith O, Schaul T, Lillicrap T, Apps C, Kavukcuoglu K, Hassabis D, Silver D (2019) AlphaStar: mastering the real-time strategy game StarCraft II. https:\/\/deepmind.com\/blog\/alphastar-mastering-real-time-strategy-game-starcraft-ii\/"},{"issue":"3\u20134","key":"1246_CR226","doi-asserted-by":"publisher","first-page":"169","DOI":"10.1016\/S0378-4754(99)00115-9","volume":"51","author":"D Vogiatzis","year":"2000","unstructured":"Vogiatzis D, Stafylopatis A (2000) Reinforcement learning for symbolic expression induction. Math Comput Simul 51(3\u20134):169\u2013179","journal-title":"Math Comput Simul"},{"issue":"5","key":"1246_CR227","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3527448","volume":"55","author":"GA Vouros","year":"2022","unstructured":"Vouros GA (2022) Explainable deep reinforcement learning: state of the art and challenges. ACM Comput Surv 55(5):1\u201339","journal-title":"ACM Comput Surv"},{"key":"1246_CR228","unstructured":"Waa J, Diggelen J, Bosch K, Neerincx M (2018) Contrastive explanations for reinforcement learning in terms of expected consequences. arXiv preprint arXiv:1807.08706"},{"key":"1246_CR229","unstructured":"W\u00e4ldchen S, Pokutta S, Huber F (2022) Training characteristic functions with reinforcement learning: Xai-methods play connect four. In: International conference on machine learning. PMLR, pp 22457\u201322474"},{"issue":"3","key":"1246_CR230","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3457188","volume":"10","author":"S Wallk\u00f6tter","year":"2021","unstructured":"Wallk\u00f6tter S, Tulli S, Castellano G, Paiva A, Chetouani M (2021) Explainable embodied agents through social cues: a review. ACM Trans Human-Robot Interact (THRI) 10(3):1\u201324","journal-title":"ACM Trans Human-Robot Interaction (THRI)"},{"key":"1246_CR231","unstructured":"Wang Z, Schaul T, Hessel M, Hasselt H, Lanctot M, Freitas N (2016a) Dueling network architectures for deep reinforcement learning. In: International conference on machine learning. PMLR, pp 1995\u20132003"},{"key":"1246_CR232","doi-asserted-by":"crossref","unstructured":"Wang N, Pynadath DV, Hill SG (2016b) Trust calibration within a human-robot team: Comparing automatically generated explanations. In: 11th ACM\/IEEE international conference on human-robot interaction (HRI). IEEE, pp 109\u2013116","DOI":"10.1109\/HRI.2016.7451741"},{"key":"1246_CR238","doi-asserted-by":"crossref","unstructured":"Wang N, Pynadath DV, Hill SG (2016c) The impact of pomdp-generated explanations on trust and performance in human-robot teams. In: Proceedings of the 2016 international conference on autonomous agents & multiagent systems. pp 997\u20131005","DOI":"10.65109\/MJOE4434"},{"key":"1246_CR233","doi-asserted-by":"crossref","unstructured":"Wang X, Chen Y, Yang J, Wu L, Wu Z, Xie X (2018) A reinforcement learning framework for explainable recommendation. In: 2018 IEEE international conference on data mining (ICDM), pp. 587\u2013596. IEEE","DOI":"10.1109\/ICDM.2018.00074"},{"issue":"1","key":"1246_CR234","doi-asserted-by":"publisher","first-page":"288","DOI":"10.1109\/TVCG.2018.2864504","volume":"25","author":"J Wang","year":"2018","unstructured":"Wang J, Gou L, Shen H-W, Yang H (2018) Dqnviz: a visual analytics approach to understand deep q-networks. IEEE Trans Vis Comput Graph 25(1):288\u2013298","journal-title":"IEEE Trans Visual Comput Graphics"},{"key":"1246_CR235","doi-asserted-by":"crossref","unstructured":"Wang J, Zhang Y, Tang K, Wu J, Xiong Z (2019) Alphastock: A buying-winners-and-selling-losers investment strategy using interpretable deep reinforcement attention networks. In: Proceedings of the 25th ACM SIGKDD international conference on knowledge discovery & data mining. pp 1900\u20131908","DOI":"10.1145\/3292500.3330647"},{"key":"1246_CR236","doi-asserted-by":"publisher","first-page":"7285","DOI":"10.1609\/aaai.v34i05.6220","volume":"34","author":"J Wang","year":"2020","unstructured":"Wang J, Zhang Y, Kim T-K, Gu Y (2020) Shapley q-value: a local reward approach to solve global reward games. In: Proceedings of the AAAI conference on artificial intelligence 34:7285\u20137292","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"issue":"2","key":"1246_CR237","doi-asserted-by":"publisher","first-page":"2297","DOI":"10.1109\/TPAMI.2022.3170302","volume":"45","author":"X Wang","year":"2023","unstructured":"Wang X, Wu Y, Zhang A, Feng F, He X, Chua T-S (2023) Reinforced causal explainer for graph neural networks. IEEE Trans Pattern Anal Mach Intell 45(2):2297\u20132309. https:\/\/doi.org\/10.1109\/TPAMI.2022.3170302","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"1246_CR239","unstructured":"Wehner J, Oliehoek F, Siebert LC (2024) Explaining learned reward functions with counterfactual trajectories. arXiv preprint arXiv: 2402.04856"},{"key":"1246_CR240","doi-asserted-by":"publisher","DOI":"10.3389\/frai.2021.550030","volume":"4","author":"L Wells","year":"2021","unstructured":"Wells L, Bednarz T (2021) Explainable ai and reinforcement learning-a systematic review of current approaches and trends. Front Artif Intell 4:550030","journal-title":"Front Artif Intell"},{"key":"1246_CR241","doi-asserted-by":"crossref","unstructured":"Williams RJ (1992) Simple statistical gradient-following algorithms for connectionist reinforcement learning. Mach Learn 8(3):5\u201332","DOI":"10.1007\/978-1-4615-3618-5_2"},{"issue":"4","key":"1246_CR242","doi-asserted-by":"publisher","first-page":"4436","DOI":"10.1007\/s11227-022-04818-4","volume":"79","author":"B Wu","year":"2023","unstructured":"Wu B, He S (2023) Self-learning and explainable deep learning network toward the security of artificial intelligence of things. J Supercomput 79(4):4436\u20134467","journal-title":"J Supercomput"},{"key":"1246_CR243","doi-asserted-by":"publisher","first-page":"12386","DOI":"10.1609\/aaai.v34i07.6924","volume":"34","author":"J Wu","year":"2020","unstructured":"Wu J, Li G, Liu S, Lin L (2020) Tree-structured policy based progressive reinforcement learning for temporally language grounding in video. Proc AAAI Conf Artif Intell 34:12386\u201312393","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR244","doi-asserted-by":"publisher","first-page":"10311","DOI":"10.1609\/aaai.v35i12.17235","volume":"35","author":"H Wu","year":"2021","unstructured":"Wu H, Khetarpal K, Precup D (2021) Self-supervised attention-aware reinforcement learning. Proc AAAI Conf Artif Intell 35:10311\u201310319","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR245","doi-asserted-by":"publisher","first-page":"228","DOI":"10.1016\/j.neunet.2023.01.025","volume":"161","author":"J Xing","year":"2023","unstructured":"Xing J, Nagata T, Zou X, Neftci E, Krichmar JL (2023) Achieving efficient interpretability of reinforcement learning via policy distillation and selective input gradient regularization. Neural Netw 161:228\u2013241","journal-title":"Neural Netw"},{"key":"1246_CR246","doi-asserted-by":"crossref","unstructured":"Xu F, Uszkoreit H, Du Y, Fan W, Zhao D, Zhu J (2019) Explainable ai: A brief survey on history, research areas, approaches and challenges. In: Natural language processing and chinese computing: 8th CCF international conference, NLPCC 2019, Dunhuang, China, October 9\u201314, 2019, Proceedings, Part II 8. Springer, pp 563\u2013574","DOI":"10.1007\/978-3-030-32236-6_51"},{"key":"1246_CR247","unstructured":"Yang Z, Bai S, Zhang L, Torr PH (2018) Learn to interpret atari agents. arXiv preprint arXiv: 1812.11276"},{"key":"1246_CR248","doi-asserted-by":"crossref","unstructured":"Yapo A, Weiss J (2018) Ethical implications of bias in machine learning","DOI":"10.24251\/HICSS.2018.668"},{"key":"1246_CR249","first-page":"18375","volume":"33","author":"H Yau","year":"2020","unstructured":"Yau H, Russell C, Hadfield S (2020) What did you think would happen? explaining agent behaviour through intended outcomes. Adv Neural Inf Process Syst 33:18375\u201318386","journal-title":"Adv Neural Inf Process Syst"},{"issue":"10","key":"1246_CR250","doi-asserted-by":"publisher","first-page":"719","DOI":"10.1038\/s41551-018-0305-z","volume":"2","author":"K-H Yu","year":"2018","unstructured":"Yu K-H, Beam AL, Kohane IS (2018) Artificial intelligence in healthcare. Nature Biomed Eng 2(10):719\u2013731","journal-title":"Nature Biomed Eng"},{"key":"1246_CR251","unstructured":"Yu T, Quillen D, He Z, Julian R, Hausman K, Finn C, Levine S (2020) Meta-world: a benchmark and evaluation for multi-task and meta reinforcement learning. In: Conference on robot learning. PMLR, pp 1094\u20131100"},{"key":"1246_CR252","unstructured":"Zahavy T, Ben-Zrihem N, Mannor S (2016) Graying the black box: understanding DQNS. In: International conference on machine learning. PMLR, pp 1899\u20131908"},{"key":"1246_CR253","doi-asserted-by":"crossref","unstructured":"Zelvelder AE, Westberg M, Fr\u00e4mling K (2021) Assessing explainability in reinforcement learning. In: Explainable and transparent ai and multi-agent systems: third international workshop, EXTRAAMAS 2021, Virtual Event, May 3\u20137, 2021, Revised Selected Papers 3. Springer, pp 223\u2013240","DOI":"10.1007\/978-3-030-82017-6_14"},{"key":"1246_CR254","unstructured":"Zeng Y, Cai R, Sun F, Huang L, Hao Z (2023) A survey on causal reinforcement learning. arXiv preprint arXiv: 2302.05209"},{"key":"1246_CR255","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijepes.2023.109240","volume":"152","author":"K Zhang","year":"2023","unstructured":"Zhang K, Zhang J, Xu P, Gao T, Gao W (2023) A multi-hierarchical interpretable method for drl-based dispatching control in power systems. Int J Electr Power Energy Syst 152:109240","journal-title":"Int J Electr Power Energy Syst"},{"key":"1246_CR256","doi-asserted-by":"crossref","unstructured":"Zhu H, Xiong Z, Magill S, Jagannathan S (2019) An inductive synthesis framework for verifiable reinforcement learning. In: Proceedings of the 40th ACM SIGPLAN conference on programming language design and implementation. pp 686\u2013701","DOI":"10.1145\/3314221.3314638"},{"key":"1246_CR257","doi-asserted-by":"publisher","first-page":"6989","DOI":"10.1609\/aaai.v34i04.6183","volume":"34","author":"G Zhu","year":"2020","unstructured":"Zhu G, Wang J, Ren Z, Lin Z, Zhang C (2020) Object-oriented dynamics learning through multi-level abstraction. Proc AAAI Conf Artif Intell 34:6989\u20136998","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1246_CR258","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1631\/FITEE.1601883","volume":"18","author":"Y-T Zhuang","year":"2017","unstructured":"Zhuang Y-T, Wu F, Chen C, Pan Y-H (2017) Challenges and opportunities: from big data to knowledge in AI 2.0. Front Inf Technol Electron Eng 18:3\u201314","journal-title":"Front Inf Technol Electron Eng"},{"key":"1246_CR259","unstructured":"Ziebart BD, Maas AL, Bagnell JA, Dey AK (2008) Maximum entropy inverse reinforcement learning, vol 8. AAAI, pp 1433\u20131438"},{"issue":"1","key":"1246_CR260","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1038\/s41598-023-50879-7","volume":"14","author":"A Zingoni","year":"2024","unstructured":"Zingoni A, Taborri J, Calabr\u00f2 G (2024) A machine learning-based classification model to support university students with dyslexia with personalized tools and strategies. Sci Rep 14(1):273. https:\/\/doi.org\/10.1038\/s41598-023-50879-7","journal-title":"Sci Rep"}],"container-title":["Data Mining and Knowledge Discovery"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10618-026-01246-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10618-026-01246-3","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10618-026-01246-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,27]],"date-time":"2026-07-27T12:02:31Z","timestamp":1785153751000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10618-026-01246-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,27]]},"references-count":260,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,10]]}},"alternative-id":["1246"],"URL":"https:\/\/doi.org\/10.1007\/s10618-026-01246-3","relation":{},"ISSN":["1384-5810","1573-756X"],"issn-type":[{"value":"1384-5810","type":"print"},{"value":"1573-756X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,27]]},"assertion":[{"value":"22 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 July 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no conflict of interest.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"76"}}