{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:45:10Z","timestamp":1782762310429,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":84,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Natural Sciences and Engineering Research Council of Canada (NSERC)","award":["ALLRP 588144 - 23"],"award-info":[{"award-number":["ALLRP 588144 - 23"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,25]]},"DOI":"10.1145\/3805689.3812339","type":"proceedings-article","created":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T17:52:08Z","timestamp":1782755528000},"page":"4162-4198","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Formal Methods Meet LLMs: Auditing, Monitoring, and Intervention for Compliance of Advanced AI Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-5802-0548","authenticated-orcid":false,"given":"Parand A.","family":"Alamdari","sequence":"first","affiliation":[{"name":"Department of Computer Science, University of Toronto, Toronto, Ontario, Canada, and Vector Institute for Artificial Intelligence, Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4003-7759","authenticated-orcid":false,"given":"Toryn Q.","family":"Klassen","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Toronto, Toronto, Ontario, Canada, and Vector Institute for Artificial Intelligence, Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4953-0945","authenticated-orcid":false,"given":"Sheila A.","family":"McIlraith","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Toronto, Toronto, Ontario, Canada, and Vector Institute for Artificial Intelligence, Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,25]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"OpenAI's Approach to External Red Teaming for AI Models and Systems. arXiv preprint arXiv:2503.16431","author":"Ahmad Lama","year":"2025","unstructured":"Lama Ahmad, Sandhini Agarwal, Michael Lampe, and Pamela Mishkin. 2025. OpenAI's Approach to External Red Teaming for AI Models and Systems. arXiv preprint arXiv:2503.16431 (2025)."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning. PMLR, 906\u2013920","author":"Alamdari Parand A.","unstructured":"Parand A. Alamdari, Toryn Q. Klassen, Elliot Creager, and Sheila A. McIlraith. 2024. Remembering to Be Fair: Non-Markovian Fairness in Sequential Decision Making. In Proceedings of the 41st International Conference on Machine Learning. PMLR, 906\u2013920. https:\/\/proceedings.mlr.press\/v235\/alamdari24a.html"},{"key":"e_1_3_2_1_3_1","volume-title":"Rodrigo Toro Icarte, and Sheila A. McIlraith","author":"Alamdari Parand A.","year":"2024","unstructured":"Parand A. Alamdari, Toryn Q. Klassen, Rodrigo Toro Icarte, and Sheila A. McIlraith. 2024. Being considerate as a pathway towards pluralistic alignment for agentic AI. arXiv preprint arXiv:2411.10613 (2024)."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the International Conference on Autonomous Agents and Multiagent Systems (AAMAS). 18\u201326","author":"Alamdari Parand A.","unstructured":"Parand A. Alamdari, Toryn Q. Klassen, Rodrigo Toro Icarte, and Sheila A. McIlraith. 2022. Be Considerate: Avoiding Negative Side Effects in Reinforcement Learning. In Proceedings of the International Conference on Autonomous Agents and Multiagent Systems (AAMAS). 18\u201326."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3679222"},{"key":"e_1_3_2_1_6_1","volume-title":"Aman Chadha, Tanya Roosta, and Chirag Shah.","author":"Amirizaniani Maryam","year":"2024","unstructured":"Maryam Amirizaniani, Jihan Yao, Adrian Lavergne, Elizabeth Snell Okada, Aman Chadha, Tanya Roosta, and Chirag Shah. 2024. Developing a framework for auditing large language models using human-in-the-loop. arXiv preprint arXiv:2402.09346 (2024)."},{"key":"e_1_3_2_1_7_1","unstructured":"Anthropic. 2024. Introducing Claude 3. https:\/\/www.anthropic.com\/news\/claude-3-family."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2015.2398877"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(99)00071-5"},{"key":"e_1_3_2_1_10_1","volume-title":"Qwen technical report. arXiv preprint arXiv:2309.16609","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Yunfei Chu, Zeyu Cui, Kai Dang, Xiaodong Deng, Yang Fan, Wenbin Ge, Yu Han, Fei Huang, Binyuan Hui, Luo Ji, Mei Li, Junyang Lin, Runji Lin, Dayiheng Liu, Gao Liu, Chengqiang Lu, Keming Lu, Jianxin Ma, Rui Men, Xingzhang Ren, Xuancheng Ren, Chuanqi Tan, Sinan Tan, Jianhong Tu, Peng Wang, Shijie Wang, Wei Wang, Shengguang Wu, Benfeng Xu, Jin Xu, An Yang, Hao Yang, Jian Yang, Shusheng Yang, Yang Yao, Bowen Yu, Hongyi Yuan, Zheng Yuan, Jianwei Zhang, Xingxuan Zhang, Yichang Zhang, Zhenru Zhang, Chang Zhou, Jingren Zhou, Xiaohuan Zhou, and Tianhang Zhu. 2023. Qwen technical report. arXiv preprint arXiv:2309.16609 (2023)."},{"key":"e_1_3_2_1_11_1","unstructured":"Yuntao Bai Saurav Kadavath Sandipan Kundu Amanda Askell Jackson Kernion Andy Jones Anna Chen Anna Goldie Azalia Mirhoseini Cameron McKinnon Carol Chen Catherine Olsson Christopher Olah Danny Hernandez Dawn Drain Deep Ganguli Dustin Li Eli Tran-Johnson Ethan Perez Jamie Kerr Jared Mueller Jeffrey Ladish Joshua Landau Kamal Ndousse Kamile Lukosiute Liane Lovitt Michael Sellitto Nelson Elhage Nicholas Schiefer Noem\u00ed Mercado Nova DasSarma Robert Lasenby Robin Larson Sam Ringer Scott Johnston Shauna Kravec Sheer El Showk Stanislav Fort Tamera Lanham Timothy Telleen-Lawton Tom Conerly Tom Henighan Tristan Hume Samuel R. Bowman Zac Hatfield-Dodds Ben Mann Dario Amodei Nicholas Joseph Sam McCandlish Tom Brown and Jared Kaplan. 2022. Constitutional AI: Harmlessness from AI feedback. arXiv preprint arXiv:2212.08073 (2022)."},{"key":"e_1_3_2_1_12_1","volume-title":"Principles of Model Checking","author":"Baier Christel","unstructured":"Christel Baier, Joost-Pieter Katoen, and Kim Guldstrand Larsen. 2014. Principles of Model Checking. MIT Press."},{"key":"e_1_3_2_1_13_1","volume-title":"Monitoring reasoning models for misbehavior and the risks of promoting obfuscation. arXiv preprint arXiv:2503.11926","author":"Baker Bowen","year":"2025","unstructured":"Bowen Baker, Joost Huizinga, Leo Gao, Zehao Dou, Melody Y Guan, Aleksander Madry, Wojciech Zaremba, Jakub Pachocki, and David Farhi. 2025. Monitoring reasoning models for misbehavior and the risks of promoting obfuscation. arXiv preprint arXiv:2503.11926 (2025)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-24622-0_5"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/S10703-016-0253-8"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/11944836_25"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1093\/logcom\/exn075"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2000799.2000800"},{"key":"e_1_3_2_1_19_1","volume-title":"Technical Report DSIT 2026\/001","author":"Bengio Yoshua","year":"2026","unstructured":"Yoshua Bengio, Stephen Clare, Carina Prunkl, Malcolm Murray, Maksym Andriushchenko, Ben Bucknall, Rishi Bommasani, Stephen Casper, Tom Davidson, Raymond Douglas, David Duvenaud, Philip Fox, Usman Gohar, Rose Hadshar, Anson Ho, Tiancheng Hu, Cameron Jones, Sayash Kapoor, Atoosa Kasirzadeh, Sam Manning, Nestor Maslej, Vasilios Mavroudis, Conor McGlynn, Richard Moulange, Jessica Newman, Kwan Yee Ng, Patricia Paskov, Shalaleh Rismani, Girish Sastry, Elizabeth Seger, Scott Singer, Charlotte Stix, Lucia Velasco, Nicole Wheeler, Daron Acemoglu, Vincent Conitzer, Thomas G. Dietterich, Edward W. Felten, Fredrik Heintz, Geoffrey Hinton, Nick Jennings, Susan Leavy, Teresa Ludermir, Vidushi Marda, Helen Margetts, John McDermid, Jane Munga, Arvind Narayanan, Alondra Nelson, Clara Neppel, Sarvapali D. Ramchurn, Stuart Russell, Marietje Schaake, Bernhard Sch\u00f6lkopf, Alvaro Soto, Lee Tiedrich, Ga\u00eb l Varoquaux, Andrew Yao, Ya-Qin Zhang, Leandro Angelo Aguirre, Olubunmi Ajala, Fahad Albalawi, Noora AlMalek, Christian Busch, Jonathan Collas, Andr\u00e9 Carlos Ponce de Leon Ferreira de Carvalho, Amandeep Gill, Ahmet Halit Hatip, Juha Heikkil\u00e4, Chris Johnson, Gill Jolly, Ziv Katzir, Mary N. Kerema, Hiroaki Kitano, Antonio Kr\u00fcger, Kyoung Mu Lee, Jos\u00e9 Ram\u00f3n L\u00f3pez Portillo, Aoife McLysaght, Olexii Molchanovskyi, Andrea Monti, Mona Nemer, Nuria Oliver, Raquel Pezoa, Audrey Plonk, Balaraman Ravindran, Hammam Riza, Crystal Rugege, Haroon Sheikh, Denise Wong, Yi Zeng, Liming Zhu, Daniel Privitera, and S\u00f6ren Mindermann. 2026. International AI Safety Report 2026. Technical Report DSIT 2026\/001. Department for Science, Innovation and Technology. https:\/\/internationalaisafetyreport.org\/publication\/international-ai-safety-report-2026"},{"key":"e_1_3_2_1_20_1","volume-title":"International AI Safety Report. Technical Report DSIT 2025\/001","author":"Bengio Yoshua","year":"2025","unstructured":"Yoshua Bengio, S\u00f6ren Mindermann, Daniel Privitera, Tamay Besiroglu, Rishi Bommasani, Stephen Casper, Yejin Choi, Philip Fox, Ben Garfinkel, Danielle Goldfarb, Hoda Heidari, Anson Ho, Sayash Kapoor, Leila Khalatbari, Shayne Longpre, Sam Manning, Vasilios Mavroudis, Mantas Mazeika, Julian Michael, Jessica Newman, Kwan Yee Ng, Chinasa T. Okolo, Deborah Raji, Girish Sastry, Elizabeth Seger, Theodora Skeadas, Tobin South, Emma Strubell, Florian Tram\u00e8r, Lucia Velasco, Nicole Wheeler, Daron Acemoglu, Olubayo Adekanmbi, David Dalrymple, Thomas G. Dietterich, Edward W. Felten, Pascale Fung, Pierre-Olivier Gourinchas, Fredrik Heintz, Geoffrey Hinton, Nick Jennings, Andreas Krause, Susan Leavy, Percy Liang, Teresa Ludermir, Vidushi Marda, Helen Margetts, John McDermid, Jane Munga, Arvind Narayanan, Alondra Nelson, Clara Neppel, Alice Oh, Gopal Ramchurn, Stuart Russell, Marietje Schaake, Bernhard Sch\u00f6lkopf, Dawn Song, Alvaro Soto, Lee Tiedrich, Ga\u00ebl Varoquaux, Andrew Yao, Ya-Qin Zhang, Olubunmi Ajala, Fahad Albalawi, Marwan Alserkal, Guillaume Avrin, Christian Busch, Andr\u00e9 Carlos Ponce de Leon Ferreira de Carvalho, Bronwyn Fox, Amandeep Singh Gill, Ahmet Halit Hatip, Juha Heikkil\u00e4, Chris Johnson, Gill Jolly, Ziv Katzir, Saif M. Khan, Hiroaki Kitano, Antonio Kr\u00fcger, Kyoung Mu Lee, Dominic Vincent Ligot, Jos\u00e9 Ram\u00f3n L\u00f3pez Portillo, Oleksii Molchanovskyi, Andrea Monti, Nusu Mwamanzi, Mona Nemer, Nuria Oliver, Raquel Pezoa Rivera, Balaraman Ravindran, Hammam Riza, Crystal Rugege, Ciar\u00e1n Seoighe, Jerry Sheehan, Haroon Sheikh, Denise Wong, and Yi Zeng. 2025. International AI Safety Report. Technical Report DSIT 2025\/001. https:\/\/internationalaisafetyreport.org\/publication\/international-ai-safety-report-2025"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2010.11.021"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/SaTML59370.2024.00037"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.4230\/LIPIcs.TIME"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/840"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i22.34521"},{"key":"e_1_3_2_1_26_1","unstructured":"Stephen Casper Xander Davies Claudia Shi Thomas Krendl Gilbert J\u00e9r\u00e9my Scheurer Javier Rando Rachel Freedman Tomasz Korbak David Lindner Pedro Freire Tony Tong Wang Samuel Marks Charbel-Rapha\u00ebl Segerie Micah Carroll Andi Peng Phillip J. K. Christoffersen Mehul Damani Stewart Slocum Usman Anwar Anand Siththaranjan Max Nadeau Eric J. Michaud Jacob Pfau Dmitrii Krasheninnikov Xin Chen Lauro Langosco Peter Hase Erdem Biyik Anca D. Dragan David Krueger Dorsa Sadigh and Dylan Hadfield-Menell. 2023. Open problems and fundamental limitations of reinforcement learning from human feedback. arXiv preprint arXiv:2307.15217 (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3659037"},{"key":"e_1_3_2_1_28_1","volume-title":"establish, exploit: Red teaming language models from scratch. arXiv preprint arXiv:2306.09442","author":"Casper Stephen","year":"2023","unstructured":"Stephen Casper, Jason Lin, Joe Kwon, Gatlen Culp, and Dylan Hadfield-Menell. 2023. Explore, establish, exploit: Red teaming language models from scratch. arXiv preprint arXiv:2306.09442 (2023)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658948"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.985"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1956.1056813"},{"key":"e_1_3_2_1_32_1","unstructured":"2024 BCCRT 149 (canLII) Civil Resolution Tribunal of British Columbia. 2024. Moffatt v. Air Canada. https:\/\/canlii.ca\/t\/k2spq File Number SC-2023-005609."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-37703-7_18"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3531146.3533213"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-24337-1_3"},{"key":"e_1_3_2_1_36_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Dai Josef","year":"2024","unstructured":"Josef Dai, Xuehai Pan, Ruiyang Sun, Jiaming Ji, Xinbo Xu, Mickel Liu, Yizhou Wang, and Yaodong Yang. 2024. Safe RLHF: Safe Reinforcement Learning from Human Feedback. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIME.2005.26"},{"key":"e_1_3_2_1_38_1","volume-title":"Gemini: A Family of Highly Capable Multimodal Models. arXiv preprint arXiv:2312.11805","author":"DeepMind Google","year":"2023","unstructured":"Google DeepMind. 2023. Gemini: A Family of Highly Capable Multimodal Models. arXiv preprint arXiv:2312.11805 (2023). https:\/\/arxiv.org\/abs\/2312.11805"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/302405.302672"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.4230\/LIPIcs.CSL.2020.20"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i13.27068"},{"key":"e_1_3_2_1_42_1","volume-title":"Red teaming language models to reduce harms: Methods, scaling behaviors, and lessons learned. arXiv preprint arXiv:2209.07858","author":"Ganguli Deep","year":"2022","unstructured":"Deep Ganguli, Liane Lovitt, Jackson Kernion, Amanda Askell, Yuntao Bai, Saurav Kadavath, Ben Mann, Ethan Perez, Nicholas Schiefer, Kamal Ndousse, Andy Jones, Sam Bowman, Anna Chen, Tom Conerly, Nova DasSarma, Dawn Drain, Nelson Elhage, Sheer El-Showk, Stanislav Fort, Zac Hatfield-Dodds, Tom Henighan, Danny Hernandez, Tristan Hume, Josh Jacobson, Scott Johnston, Shauna Kravec, Catherine Olsson, Sam Ringer, Eli Tran-Johnson, Dario Amodei, Tom Brown, Nicholas Joseph, Sam McCandlish, Chris Olah, Jared Kaplan, and Jack Clark. 2022. Red teaming language models to reduce harms: Methods, scaling behaviors, and lessons learned. arXiv preprint arXiv:2209.07858 (2022)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2008.10.012"},{"key":"e_1_3_2_1_44_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The Llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"16336","author":"Greenblatt Ryan","year":"2024","unstructured":"Ryan Greenblatt, Buck Shlegeris, Kshitij Sachan, and Fabien Roger. 2024. AI Control: Improving Safety Despite Intentional Subversion. In Proceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 235). PMLR, 16295\u201316336."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658959"},{"key":"e_1_3_2_1_47_1","volume-title":"arXiv preprint arXiv:2512.18311","author":"Guan Melody Y.","year":"2025","unstructured":"Melody Y. Guan, Miles Wang, Micah Carroll, Zehao Dou, Annie Y. Wei, Marcus Williams, Benjamin Arnav, Joost Huizinga, Ian Kivlichan, Mia Glaese, Jakub Pachocki, and Bowen Baker. 2025. Monitoring Monitorability. arXiv preprint arXiv:2512.18311 (2025)."},{"key":"e_1_3_2_1_48_1","volume-title":"Logically-Correct Reinforcement Learning. arXiv preprint arXiv:1801.08099","author":"Hasanbeig Mohammadhosein","year":"2018","unstructured":"Mohammadhosein Hasanbeig, Alessandro Abate, and Daniel Kroening. 2018. Logically-Correct Reinforcement Learning. arXiv preprint arXiv:1801.08099 (2018). http:\/\/arxiv.org\/abs\/1801.08099"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.5555\/872023.872572"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.5555\/2944225.2944368"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-37703-7_17"},{"key":"e_1_3_2_1_52_1","volume-title":"The curious case of neural text degeneration. arXiv preprint arXiv:1904.09751","author":"Holtzman Ari","year":"2019","unstructured":"Ari Holtzman, Jan Buys, Li Du, Maxwell Forbes, and Yejin Choi. 2019. The curious case of neural text degeneration. arXiv preprint arXiv:1904.09751 (2019)."},{"key":"e_1_3_2_1_53_1","volume-title":"International Conference on Machine Learning. PMLR, 15307\u201315329","author":"Jones Erik","year":"2023","unstructured":"Erik Jones, Anca Dragan, Aditi Raghunathan, and Jacob Steinhardt. 2023. Automatically auditing large language models via discrete optimization. In International Conference on Machine Learning. PMLR, 15307\u201315329."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/13677.22723"},{"key":"e_1_3_2_1_55_1","volume-title":"Language models (mostly) know what they know. arXiv preprint arXiv:2207.05221","author":"Kadavath Saurav","year":"2022","unstructured":"Saurav Kadavath, Tom Conerly, Amanda Askell, Tom Henighan, Dawn Drain, Ethan Perez, Nicholas Schiefer, Zac Hatfield-Dodds, Nova DasSarma, Eli Tran-Johnson, Scott Johnston, Sheer El-Showk, Andy Jones, Nelson Elhage, Tristan Hume, Anna Chen, Yuntao Bai, Sam Bowman, Stanislav Fort, Deep Ganguli, Danny Hernandez, Josh Jacobson, Jackson Kernion, Shauna Kravec, Liane Lovitt, Kamal Ndousse, Catherine Olsson, Sam Ringer, Dario Amodei, Tom Brown, Jack Clark, Nicholas Joseph, Ben Mann, Sam McCandlish, Chris Olah, and Jared Kaplan. 2022. Language models (mostly) know what they know. arXiv preprint arXiv:2207.05221 (2022)."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-17196-3_10"},{"key":"e_1_3_2_1_57_1","volume-title":"Variational temporal abstraction. Advances in Neural Information Processing Systems 32","author":"Kim Taesup","year":"2019","unstructured":"Taesup Kim, Sungjin Ahn, and Yoshua Bengio. 2019. Variational temporal abstraction. Advances in Neural Information Processing Systems 32 (2019)."},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of International Conference on Autonomous Agents and Multiagent Systems (AAMAS). 1797\u20131801","author":"Klassen Toryn Q.","unstructured":"Toryn Q. Klassen, Parand A. Alamdari, and Sheila A. McIlraith. 2023. Epistemic Side Effects: An AI Safety Problem. In Proceedings of International Conference on Autonomous Agents and Multiagent Systems (AAMAS). 1797\u20131801."},{"key":"e_1_3_2_1_59_1","volume-title":"McIlraith","author":"Klassen Toryn Q.","year":"2024","unstructured":"Toryn Q. Klassen, Parand A. Alamdari, and Sheila A. McIlraith. 2024. Pluralistic Alignment Over Time. arXiv preprint arXiv:2411.10654 (2024)."},{"key":"e_1_3_2_1_60_1","unstructured":"Tomek Korbak Mikita Balesni Elizabeth Barnes Yoshua Bengio Joe Benton Joseph Bloom Mark Chen Alan Cooney Allan Dafoe Anca Dragan et al. 2025. Chain of Thought Monitorability: A New and Fragile Opportunity for AI Safety. arXiv preprint arXiv:2507.11473 (2025)."},{"key":"e_1_3_2_1_61_1","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Krakovna Victoria","year":"2020","unstructured":"Victoria Krakovna, Laurent Orseau, Richard Ngo, Miljan Martic, and Shane Legg. 2020. Avoiding Side Effects By Considering Future Tasks. In Advances in Neural Information Processing Systems, Vol. 33. Curran Associates, Inc."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658957"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00766-011-0129-9"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS58592.2024.10802696"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-23059-2_13"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1007\/s43681-023-00289-2"},{"key":"e_1_3_2_1_67_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. https:\/\/openai.com\/research\/gpt-4."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.225"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/SFCS.1977.32"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/75277.75293"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3514094.3534181"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600211.3604712"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.is.2007.07.001"},{"key":"e_1_3_2_1_74_1","unstructured":"Ajith Sankaran. 2025. How Small And Medium Businesses Can Take Advantage Of The Emerging Agentic AI Era. https:\/\/www.forbes.com\/councils\/forbesbusinesscouncil\/2025\/04\/01\/how-small-and-medium-businesses-can-take-advantage-of-the-emerging-agentic-ai-era\/ Forbes Business Council COUNCIL POST."},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/3287560.3287598"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-59536-8_30"},{"key":"e_1_3_2_1_77_1","volume-title":"International Conference on Machine Learning. PMLR, 10497\u201310508","author":"Vaezipoor Pashootan","year":"2021","unstructured":"Pashootan Vaezipoor, Andrew C Li, Rodrigo A Toro Icarte, and Sheila A Mcilraith. 2021. LTL2Action: Generalizing LTL instructions for multi-task RL. In International Conference on Machine Learning. PMLR, 10497\u201310508."},{"key":"e_1_3_2_1_78_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"35150","author":"Voloshin Cameron","year":"2023","unstructured":"Cameron Voloshin, Abhinav Verma, and Yisong Yue. 2023. Eventual Discounting Temporal Logic Counterfactual Experience Replay. In Proceedings of the 40th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 202). PMLR, 35137\u201335150. https:\/\/proceedings.mlr.press\/v202\/voloshin23a.html"},{"key":"e_1_3_2_1_79_1","volume-title":"Proceedings of the 2020 Conference on Robot Learning (Proceedings of Machine Learning Research","volume":"1718","author":"Wang Christopher","year":"2021","unstructured":"Christopher Wang, Candace Ross, Yen-Ling Kuo, Boris Katz, and Andrei Barbu. 2021. Learning a natural-language to LTL executable semantic parser for grounded robotics. In Proceedings of the 2020 Conference on Robot Learning (Proceedings of Machine Learning Research, Vol. 155). PMLR, 1706\u20131718."},{"key":"e_1_3_2_1_80_1","volume-title":"ScienceWorld: Is your agent smarter than a 5th grader? arXiv preprint arXiv:2203.07540","author":"Wang Ruoyao","year":"2022","unstructured":"Ruoyao Wang, Peter Jansen, Marc-Alexandre C\u00f4t\u00e9, and Prithviraj Ammanabrolu. 2022. ScienceWorld: Is your agent smarter than a 5th grader? arXiv preprint arXiv:2203.07540 (2022)."},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1200"},{"key":"e_1_3_2_1_82_1","volume-title":"Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency. 1701\u20131713","author":"Wright Lucas","unstructured":"Lucas Wright, Roxana Mika Muenster, Briana Vecchione, Tianyao Qu, Pika Cai, Alan Smith, COMM\/INFO 2450 Student Investigators, Jacob Metcalf, and J. Nathan Matias. 2024. Null Compliance: NYC Local Law 144 and the challenges of algorithm accountability. In Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency. 1701\u20131713."},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611447"},{"key":"e_1_3_2_1_84_1","volume-title":"Can Reasoning Models Obfuscate Reasoning? Stress-Testing Chain-of-Thought Monitorability. arXiv preprint arXiv:2510.19851","author":"Zolkowski Artur","year":"2025","unstructured":"Artur Zolkowski, Wen Xing, David Lindner, Florian Tram\u00e8r, and Erik Jenner. 2025. Can Reasoning Models Obfuscate Reasoning? Stress-Testing Chain-of-Thought Monitorability. arXiv preprint arXiv:2510.19851 (2025)."}],"event":{"name":"FAccT '26: The 2026 ACM Conference on Fairness, Accountability, and Transparency","location":"Montreal QC Canada","acronym":"FAccT '26","sponsor":["ACM\/SIG"]},"container-title":["Proceedings of the 2026 ACM Conference on Fairness, Accountability, and Transparency"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805689.3812339","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:03:20Z","timestamp":1782759800000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805689.3812339"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,25]]},"references-count":84,"alternative-id":["10.1145\/3805689.3812339","10.1145\/3805689"],"URL":"https:\/\/doi.org\/10.1145\/3805689.3812339","relation":{},"subject":[],"published":{"date-parts":[[2026,6,25]]},"assertion":[{"value":"2026-06-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}