{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,30]],"date-time":"2026-01-30T08:04:26Z","timestamp":1769760266771,"version":"3.49.0"},"reference-count":25,"publisher":"Association for Computing Machinery (ACM)","issue":"2","license":[{"start":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T00:00:00Z","timestamp":1769558400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["SLES 2331783"],"award-info":[{"award-number":["SLES 2331783"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000006","name":"Office of Naval Research","doi-asserted-by":"publisher","award":["N00014-20-1-2115"],"award-info":[{"award-number":["N00014-20-1-2115"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["Commun. ACM"],"published-print":{"date-parts":[[2026,2]]},"abstract":"<jats:p>A tutorial-style introduction to recent research on using logical specifications to encode RL tasks illustrates theoretical limitations and practical solutions.<\/jats:p>","DOI":"10.1145\/3744706","type":"journal-article","created":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T16:59:58Z","timestamp":1769619598000},"page":"80-87","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Specification-Guided Reinforcement Learning"],"prefix":"10.1145","volume":"69","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1733-7083","authenticated-orcid":false,"given":"Rajeev","family":"Alur","sequence":"first","affiliation":[{"name":"University of Pennsylvania, Philadelphia, Pennsylvania, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0405-073X","authenticated-orcid":false,"given":"Suguman","family":"Bansal","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, Georgia, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9990-7566","authenticated-orcid":false,"given":"Osbert","family":"Bastani","sequence":"additional","affiliation":[{"name":"University of Pennsylvania, Philadelphia, Pennsylvania, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1448-2947","authenticated-orcid":false,"given":"Kishor","family":"Jothimurugan","sequence":"additional","affiliation":[{"name":"Two Sigma Investments, LP, New York, New York, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,1,29]]},"reference":[{"key":"e_1_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2016.7799279"},{"key":"e_1_3_1_3_2","doi-asserted-by":"crossref","unstructured":"Alur R. Bansal S. Bastani O. and Jothimurugan K. A Framework for Transforming Specifications in Reinforcement Learning. Principles of Systems Design: Essays Dedicated to Thomas A. Henzinger on the Occasion of His 60th Birthday (2021).","DOI":"10.1007\/978-3-031-22337-2_29"},{"key":"e_1_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-37706-8_21"},{"key":"e_1_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196796"},{"key":"e_1_3_1_6_2","article-title":"LTLf\/LDLf non-markovian rewards","volume":"32","author":"Brafman R.","year":"2018","unstructured":"Brafman, R., Giacomo, G. D., and Patrizi, F. LTLf\/LDLf non-markovian rewards. In Proceedings of the AAAI Conf. on Artificial Intelligence, 32, 2018.","journal-title":"Proceedings of the AAAI Conf. on Artificial Intelligence"},{"key":"e_1_3_1_7_2","first-page":"128","article-title":"Foundations for restraining bolts: Reinforcement learning with LTLf\/LDLf restraining specifications","volume":"29","author":"Giacomo G. D.","year":"2019","unstructured":"Giacomo, G. D., Iocchi, L., Favorito, M., and Patrizi, F. Foundations for restraining bolts: Reinforcement learning with LTLf\/LDLf restraining specifications. In Proceedings of the Intern. Conf. on Automated Planning and Scheduling, 29, 2019, 128\u2013136.","journal-title":"Proceedings of the Intern. Conf. on Automated Planning and Scheduling"},{"key":"e_1_3_1_8_2","first-page":"233","volume-title":"Joint European Conf. on Machine Learning and Knowledge Discovery in Databases","author":"Eappen J.","year":"2022","unstructured":"Eappen, J. and Jagannathan, S. DistSPECTRL: Distributing specifications in multi-agent reinforcement learning systems. In Joint European Conf. on Machine Learning and Knowledge Discovery in Databases. Springer, 2022, 233\u2013250."},{"key":"e_1_3_1_9_2","doi-asserted-by":"crossref","unstructured":"Fu J. and Topcu U. Probably Approximately Correct MDP Learning and Control With Temporal Logic Constraints. In Robotics: Science and Systems. 2014.","DOI":"10.15607\/RSS.2014.X.039"},{"key":"e_1_3_1_10_2","doi-asserted-by":"crossref","unstructured":"Hahn E. M. et al. Omega-Regular Objectives in Model-Free Reinforcement Learning. In Tools and Algorithms for the Construction and Analysis of Systems. 2019 395\u2013412.","DOI":"10.1007\/978-3-030-17462-0_27"},{"key":"e_1_3_1_11_2","unstructured":"Hasanbeig M. Abate A. and Kroening D. Logically-constrained reinforcement learning. arXiv preprint arXiv:1801.08099 (2018)."},{"key":"e_1_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Hasanbeig M. et al. Reinforcement Learning for Temporal Logic Control Synthesis with Probabilistic Satisfaction Guarantees. In Conf. on Decision and Control (CDC). 2019 5338\u20135343.","DOI":"10.1109\/CDC40024.2019.9028919"},{"key":"e_1_3_1_13_2","first-page":"2107","volume-title":"Intern. Conf. on Machine Learning","author":"Icarte R. T.","year":"2018","unstructured":"Icarte, R. T., Klassen, T., Valenzano, R., and McIlraith, S. Using reward machines for high-level task specification and decomposition in reinforcement learning. In Intern. Conf. on Machine Learning. PMLR, 2018, 2107\u20132116."},{"key":"e_1_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Ivanov R. et al. Compositional Learning and Verification of Neural Network Controllers. ACM Transactions on Embedded Computing Systems (2021).","DOI":"10.1145\/3477023"},{"key":"e_1_3_1_15_2","unstructured":"Jiang Y. et al. Temporal-Logic-Based Reward Shaping for Continuing Learning Tasks. arXiv:2007.01498 [cs.AI] 2020."},{"key":"e_1_3_1_16_2","first-page":"13041","article-title":"A Composable Specification Language for Reinforcement Learning Tasks","volume":"32","author":"Jothimurugan K.","year":"2019","unstructured":"Jothimurugan, K., Alur, R., and Bastani, O. A Composable Specification Language for Reinforcement Learning Tasks. In Advances in Neural Information Processing Systems, 32. 2019, 13041\u201313051.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_1_17_2","unstructured":"Jothimurugan K. Bansal S. Bastani O. and Alur R. Compositional Reinforcement Learning from Logical Specifications. In Advances in Neural Information Processing Systems. 2021."},{"key":"e_1_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Jothimurugan K. Bansal S. Bastani O. and Alur R. Specification-Guided Learning of Nash Equilibria with High Social Welfare. (2022).","DOI":"10.1007\/978-3-031-13188-2_17"},{"key":"e_1_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2017.8206234"},{"key":"e_1_3_1_20_2","unstructured":"Littman M. L. et al. Environment-Independent Task Specifications via GLTL. arXiv:1704.04341 [cs.AI] 2017."},{"key":"e_1_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/SFCS.1977.32"},{"key":"e_1_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Strehl A. L. et al. PAC model-free reinforcement learning. In Proceedings of the 23rd Intern. Conf. on Machine learning. 2006 881\u2013888.","DOI":"10.1145\/1143844.1143955"},{"key":"e_1_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Xu Z. and Topcu U. Transfer of Temporal Logic Formulas in Reinforcement Learning. In Intern. Joint Conf. on Artificial Intelligence. 2019 4010\u20134018.","DOI":"10.24963\/ijcai.2019\/557"},{"key":"e_1_3_1_24_2","doi-asserted-by":"crossref","unstructured":"Yang C. Littman M. and Carbin M. Reinforcement Learning for General LTL Objectives Is Intractable. arXiv preprint arXiv:2111.12679 (2021).","DOI":"10.24963\/ijcai.2022\/507"},{"key":"e_1_3_1_25_2","unstructured":"Yuan L. Z. Hasanbeig M. Abate A. and Kroening D. Modular deep reinforcement learning with temporal logic specifications. arXiv preprint arXiv:1909.11591 (2019)."},{"key":"e_1_3_1_26_2","article-title":"Compositional policy learning in stochastic control systems with formal guarantees","volume":"36","author":"\u017dikeli\u0107 \u00d0.","year":"2024","unstructured":"\u017dikeli\u0107, \u00d0. et al. Compositional policy learning in stochastic control systems with formal guarantees. Advances in Neural Information Processing Systems 36, (2024).","journal-title":"Advances in Neural Information Processing Systems"}],"container-title":["Communications of the ACM"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/full\/10.1145\/3744706","content-type":"text\/html","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3744706","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T17:05:31Z","timestamp":1769706331000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3744706"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,29]]},"references-count":25,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["10.1145\/3744706"],"URL":"https:\/\/doi.org\/10.1145\/3744706","relation":{},"ISSN":["0001-0782","1557-7317"],"issn-type":[{"value":"0001-0782","type":"print"},{"value":"1557-7317","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,29]]},"assertion":[{"value":"2025-01-20","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2026-01-29","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}