{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:44:48Z","timestamp":1782762288050,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,25]]},"DOI":"10.1145\/3805689.3812308","type":"proceedings-article","created":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T17:52:08Z","timestamp":1782755528000},"page":"249-279","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Human-AI Complementarity: A Goal for Amplified Oversight"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8212-9427","authenticated-orcid":false,"given":"Rishub","family":"Jain","sequence":"first","affiliation":[{"name":"Google DeepMind, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4522-2697","authenticated-orcid":false,"given":"Sophie","family":"Bridgers","sequence":"additional","affiliation":[{"name":"Google DeepMind, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-1896-5517","authenticated-orcid":false,"given":"Lili","family":"Janzer","sequence":"additional","affiliation":[{"name":"Google DeepMind, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5795-9720","authenticated-orcid":false,"given":"Rory","family":"Greig","sequence":"additional","affiliation":[{"name":"Google DeepMind, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3838-8479","authenticated-orcid":false,"given":"Tian Huey","family":"Teh","sequence":"additional","affiliation":[{"name":"Google DeepMind, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0452-6785","authenticated-orcid":false,"given":"Vladimir","family":"Mikulik","sequence":"additional","affiliation":[{"name":"Anthropic, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,25]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1093\/cjres\/rsz022"},{"key":"e_1_3_2_1_2_1","volume-title":"Concrete problems in AI safety. arXiv preprint arXiv:1606.06565","author":"Amodei Dario","year":"2016","unstructured":"Dario Amodei, Chris Olah, Jacob Steinhardt, Paul Christiano, John Schulman, and Dan Man\u00e9. 2016. Concrete problems in AI safety. arXiv preprint arXiv:1606.06565 (2016)."},{"key":"e_1_3_2_1_3_1","unstructured":"Yuntao Bai Saurav Kadavath Sandipan Kundu Amanda Askell Jackson Kernion Andy Jones Anna Chen Anna Goldie Azalia Mirhoseini Cameron McKinnon Carol Chen Catherine Olsson Christopher Olah Danny Hernandez Dawn Drain Deep Ganguli Dustin Li Eli Tran-Johnson Ethan Perez Jamie Kerr Jared Mueller Jeffrey Ladish Joshua Landau Kamal Ndousse Kamile Lukosuite Liane Lovitt Michael Sellitto Nelson Elhage Nicholas Schiefer Noemi Mercado Nova DasSarma Robert Lasenby Robin Larson Sam Ringer Scott Johnston Shauna Kravec Sheer El Showk Stanislav Fort Tamera Lanham Timothy Telleen-Lawton Tom Conerly Tom Henighan Tristan Hume Samuel R. Bowman Zac Hatfield-Dodds Ben Mann Dario Amodei Nicholas Joseph Sam McCandlish Tom Brown and Jared Kaplan. 2022. Constitutional AI: Harmlessness from AI Feedback. arXiv:2212.08073 [cs.CL] https:\/\/arxiv.org\/abs\/2212.08073"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445717"},{"key":"e_1_3_2_1_5_1","volume-title":"\u201cO'Reilly Media","author":"Bird Steven","unstructured":"Steven Bird, Ewan Klein, and Edward Loper. 2009. Natural language processing with Python: analyzing text with the natural language toolkit. \u201cO'Reilly Media, Inc.\u201d."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i5.20465"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3449287"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3377325.3377498"},{"key":"e_1_3_2_1_9_1","volume-title":"Deep reinforcement learning from human preferences. (06","author":"Christiano Paul","year":"2017","unstructured":"Paul Christiano, Jan Leike, Tom Brown, Miljan Martic, Shane Legg, and Dario Amodei. 2017. Deep reinforcement learning from human preferences. (06 2017)."},{"key":"e_1_3_2_1_10_1","volume-title":"Supervising strong learners by amplifying weak experts. arXiv preprint arXiv:1810.08575","author":"Christiano Paul","year":"2018","unstructured":"Paul Christiano, Buck Shlegeris, and Dario Amodei. 2018. Supervising strong learners by amplifying weak experts. arXiv preprint arXiv:1810.08575 (2018)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41591-023-02437-x"},{"key":"e_1_3_2_1_12_1","volume-title":"Official Journal of the European Union","author":"European Parliament and Council of the European Union. 2024. Regulation (EU) 2024\/1689 laying down harmonised rules on artificial intelligence (Artificial Intelligence Act).","year":"2024","unstructured":"European Parliament and Council of the European Union. 2024. Regulation (EU) 2024\/1689 laying down harmonised rules on artificial intelligence (Artificial Intelligence Act). Official Journal of the European Union (2024). http:\/\/data.europa.eu\/eli\/reg\/2024\/1689\/oj"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1287\/isre.2021.1079"},{"key":"e_1_3_2_1_14_1","unstructured":"Yonatan Geifman and Ran El-Yaniv. 2017. Selective Prediction with Deep Neural Networks. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_15_1","volume-title":"Fine-Grained Appropriate Reliance: Human-AI Collaboration with a Multi-Step Transparent Decision Workflow for Complex Task Decomposition. arXiv preprint arXiv.2501.10909","author":"He Gaole","year":"2025","unstructured":"Gaole He, Patrick Hemmer, Michael V\u00f6ssing, Max Schemmer, and Ujwal Gadiraju. 2025. Fine-Grained Appropriate Reliance: Human-AI Collaboration with a Multi-Step Transparent Decision Workflow for Complex Task Decomposition. arXiv preprint arXiv.2501.10909 (2025)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581641.3584052"},{"key":"e_1_3_2_1_17_1","volume-title":"Retrieved","author":"Hubinger Evan","year":"2020","unstructured":"Evan Hubinger. 2020. AI safety via market making. Retrieved September 12, 2024 from https:\/\/www.alignmentforum.org\/posts\/YWwzccGbcHMJMpT45\/ai-safety-via-market-making"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.14569\/IJACSA.2025.01612122"},{"key":"e_1_3_2_1_19_1","volume-title":"AI safety via debate. arXiv preprint arXiv.1805.00899","author":"Irving Geoffrey","year":"2018","unstructured":"Geoffrey Irving, Paul Christiano, and Dario Amodei. 2018. AI safety via debate. arXiv preprint arXiv.1805.00899 (2018)."},{"key":"e_1_3_2_1_20_1","volume-title":"Second Conference of the International Association for Safe and Ethical Artificial Intelligence (IASEAI'26)","author":"Jha Tushita","year":"2026","unstructured":"Tushita Jha, Tom Everitt, and Alex Grzankowski. 2026. Human Amplification, Intelligent Agents, and the Aims of AI Research. In Second Conference of the International Association for Safe and Ethical Artificial Intelligence (IASEAI'26). https:\/\/philpapers.org\/archive\/JHAHAI.pdf"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376219"},{"key":"e_1_3_2_1_22_1","volume-title":"On scalable oversight with weak LLMs judging strong LLMs. arXiv preprint arXiv:2407.04622","author":"Kenton Zachary","year":"2024","unstructured":"Zachary Kenton, Noah Y. Siegel, J\u00e1nos Kram\u00e1r, Jonah Brown-Cohen, Samuel Albanie, Jannis Bulian, Rishabh Agarwal, David Lindner, Yunhao Tang, Noah D. Goodman1, and Rohin Shah. 2024. On scalable oversight with weak LLMs judging strong LLMs. arXiv preprint arXiv:2407.04622 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Debating with More Persuasive LLMs Leads to More Truthful Answers. arXiv preprint arXiv:2402.06782","author":"Khan Akbir","year":"2024","unstructured":"Akbir Khan, John Hughes, Dan Valentine, Laura Ruis, Kshitij Sachan, Ansh Radhakrishnan, Edward Grefenstette, Samuel R Bowman, Tim Rockt\u00e4schel, and Ethan Perez. 2024. Debating with More Persuasive LLMs Leads to More Truthful Answers. arXiv preprint arXiv:2402.06782 (2024)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658941"},{"key":"e_1_3_2_1_25_1","volume-title":"Retrieved","author":"Krakovna Victoria","year":"2020","unstructured":"Victoria Krakovna, Jonathan Uesato, Matthew Rahtz Vladimir Mikulik, Tom Everitt, Ramana Kumar, Zac Kenton, Jan Leike, and Shane Legg. 2020. Specification gaming : the flip side of AI ingenuity. Retrieved September 12, 2024 from https:\/\/deepmind.google\/discover\/blog\/specification-gaming-the-flip-side-of-ai-ingenuity\/"},{"key":"e_1_3_2_1_26_1","volume-title":"Gradual Disempowerment: Systemic Existential Risks from Incremental AI Development. arXiv:2501.16946 [cs.CY] https:\/\/arxiv.org\/abs\/2501.16946","author":"Kulveit Jan","year":"2025","unstructured":"Jan Kulveit, Raymond Douglas, Nora Ammann, Deger Turan, David Krueger, and David Duvenaud. 2025. Gradual Disempowerment: Systemic Existential Risks from Incremental AI Development. arXiv:2501.16946 [cs.CY] https:\/\/arxiv.org\/abs\/2501.16946"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3287560.3287590"},{"key":"e_1_3_2_1_28_1","volume-title":"Trust in automation: Designing for appropriate reliance. Human factors 46, 1","author":"Lee John D","year":"2004","unstructured":"John D Lee and Katrina A See. 2004. Trust in automation: Designing for appropriate reliance. Human factors 46, 1 (2004), 50\u201380."},{"key":"e_1_3_2_1_29_1","volume-title":"Scalable agent alignment via reward modeling: a research direction. arXiv preprint arXiv:1811.07871","author":"Leike Jan","year":"2018","unstructured":"Jan Leike, David Krueger, Tom Everitt, Miljan Martic, Vishal Maini, and Shane Legg. 2018. Scalable agent alignment via reward modeling: a research direction. arXiv preprint arXiv:1811.07871 (2018)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447803"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581058"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642671"},{"key":"e_1_3_2_1_33_1","volume-title":"Predict Responsibly: Improving Fairness and Accuracy by Learning to Defer. In Advances in Neural Information Processing Systems (NeurIPS).","author":"Madras David","year":"2018","unstructured":"David Madras, Toni Pitassi, and Richard Zemel. 2018. Predict Responsibly: Improving Fairness and Accuracy by Learning to Defer. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0159"},{"key":"e_1_3_2_1_35_1","volume-title":"Juan Felipe Ceron Uribe, Evgenia Nitishinskaya, Maja Trebacz, and Jan Leike.","author":"McAleese Nat","year":"2024","unstructured":"Nat McAleese, Rai Michael Pokorny, Juan Felipe Ceron Uribe, Evgenia Nitishinskaya, Maja Trebacz, and Jan Leike. 2024. LLM Critics Help Catch LLM Bugs. arXiv:2407.00215 [cs.SE] https:\/\/arxiv.org\/abs\/2407.00215"},{"key":"e_1_3_2_1_36_1","volume-title":"Frontier models are capable of in-context scheming. arXiv preprint arXiv:2412.04984","author":"Meinke Alexander","year":"2024","unstructured":"Alexander Meinke, Bronson Schoen, J\u00e9r\u00e9my Scheurer, Mikita Balesni, Rusheb Shah, and Marius Hobbhahn. 2024. Frontier models are capable of in-context scheming. arXiv preprint arXiv:2412.04984 (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"Debate helps supervise unreliable experts. arXiv preprint arXiv:2311.08702","author":"Michael Julian","year":"2023","unstructured":"Julian Michael, Salsabila Mahdi, David Rein, Jackson Petty, Julien Dirani, Vishakh Padmakumar, and Samuel R Bowman. 2023. Debate helps supervise unreliable experts. arXiv preprint arXiv:2311.08702 (2023)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3593013.3594001"},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Mozannar Hussein","year":"2020","unstructured":"Hussein Mozannar and David Sontag. 2020. Consistent estimators for learning to defer to an expert. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_40_1","volume-title":"The alignment problem from a deep learning perspective. arXiv preprint arXiv:2209.00626","author":"Ngo Richard","year":"2022","unstructured":"Richard Ngo, Lawrence Chan, and S\u00f6ren Mindermann. 2022. The alignment problem from a deep learning perspective. arXiv preprint arXiv:2209.00626 (2022)."},{"key":"e_1_3_2_1_41_1","volume-title":"Humans and automation: Use, misuse, disuse, abuse. Human factors 39, 2","author":"Parasuraman Raja","year":"1997","unstructured":"Raja Parasuraman and Victor Riley. 1997. Humans and automation: Use, misuse, disuse, abuse. Human factors 39, 2 (1997), 230\u2013253."},{"key":"e_1_3_2_1_42_1","volume-title":"Amanpreet Singh Saimbhi, and Samuel R Bowman","author":"Parrish Alicia","year":"2022","unstructured":"Alicia Parrish, Harsh Trivedi, Nikita Nangia, Vishakh Padmakumar, Jason Phang, Amanpreet Singh Saimbhi, and Samuel R Bowman. 2022. Two-Turn Debate Doesn't Help Humans Answer Hard Reading Comprehension Questions. arXiv preprint arXiv:2210.10860 (2022)."},{"key":"e_1_3_2_1_43_1","volume-title":"Single-turn debate does not help humans answer hard reading-comprehension questions. arXiv preprint arXiv:2204.05212","author":"Parrish Alicia","year":"2022","unstructured":"Alicia Parrish, Harsh Trivedi, Ethan Perez, Angelica Chen, Nikita Nangia, Jason Phang, and Samuel R Bowman. 2022. Single-turn debate does not help humans answer hard reading-comprehension questions. arXiv preprint arXiv:2204.05212 (2022)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3640543.3645144"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3580794"},{"key":"e_1_3_2_1_46_1","volume-title":"Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems 36","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2024. Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_47_1","volume-title":"The Algorithmic Automation Problem: Prediction, Triage, and Human Effort. arXiv preprint arXiv:1903.12220","author":"Raghu Maithra","year":"2019","unstructured":"Maithra Raghu, Katy Blumer, Greg Corrado, Jon Kleinberg, Ziad Obermeyer, and Sendhil Mullainathan. 2019. The Algorithmic Automation Problem: Prediction, Triage, and Human Effort. arXiv preprint arXiv:1903.12220 (2019)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581641.3584066"},{"key":"e_1_3_2_1_49_1","volume-title":"Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency (FAccT).","author":"Vera","unstructured":"Vera Schmitt et al. 2024. The Role of Explainability in Collaborative Human-AI Disinformation Detection. In Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency (FAccT)."},{"key":"e_1_3_2_1_50_1","volume-title":"AGI Safety and Alignment at Google DeepMind: A Summary of Recent Work. AI Alignment Forum (aug","author":"Shah Rohin","year":"2024","unstructured":"Rohin Shah, Seb Farquhar, and Anca Dragan. 2024. AGI Safety and Alignment at Google DeepMind: A Summary of Recent Work. AI Alignment Forum (aug 2024). https:\/\/www.alignmentforum.org\/posts\/79BPxvSsjzBkiSyTq\/agi-safety-and-alignment-at-google-deepmind-a-summary-of"},{"key":"e_1_3_2_1_51_1","volume-title":"Anna Wang, Arthur Conmy, David Lindner, Jonah Brown-Cohen, Lewis Ho, Neel Nanda, Raluca Ada Popa, et al.","author":"Shah Rohin","year":"2025","unstructured":"Rohin Shah, Alex Irpan, Alexander Matt Turner, Anna Wang, Arthur Conmy, David Lindner, Jonah Brown-Cohen, Lewis Ho, Neel Nanda, Raluca Ada Popa, et al. 2025. An approach to technical agi safety and security. arXiv preprint arXiv:2504.01849 (2025)."},{"key":"e_1_3_2_1_52_1","unstructured":"Mrinank Sharma Meg Tong Tomasz Korbak David Duvenaud Amanda Askell Samuel R Bowman Newton Cheng Esin Durmus Zac Hatfield-Dodds Scott R Johnston et al. 2023. Towards understanding sycophancy in language models. arXiv preprint arXiv:2310.13548 (2023)."},{"key":"e_1_3_2_1_53_1","volume-title":"Chen Zhao, Shi Feng, Hal Daum\u00e9 III, and Jordan Boyd-Graber.","author":"Si Chenglei","year":"2023","unstructured":"Chenglei Si, Navita Goyal, Sherry Tongshuang Wu, Chen Zhao, Shi Feng, Hal Daum\u00e9 III, and Jordan Boyd-Graber. 2023. Large Language Models Help Humans Verify Truthfulness-Except When They Are Convincingly Wrong. arXiv preprint arXiv:2310.12558 (2023)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2024.103672"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2111547119"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","unstructured":"Joshua Strong Emma Sun Harry Rogers Helen Higham and Alison Noble. 2025. Learning to Defer: A Survey. doi:10.5281\/zenodo.17843044 Preprint. Comprehensive survey of the learning-to-defer literature.","DOI":"10.5281\/zenodo.17843044"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","unstructured":"Alaa Tharwat and Wolfram Schenck. 2023. A Survey on Active Learning: State-of-the-Art Practical Challenges and Research Directions. Mathematics 11 4(2023). doi:10.3390\/math11040820","DOI":"10.3390\/math11040820"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41562-024-02024-1"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579605"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3641960"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"crossref","unstructured":"Jerry Wei Chengrun Yang Xinying Song Yifeng Lu Nathan Hu Dustin Tran Daiyi Peng Ruibo Liu Da Huang Cosmo Du et al. 2024. Long-form factuality in large language models. arXiv preprint arXiv:2403.18802 (2024).","DOI":"10.52202\/079017-2567"}],"event":{"name":"FAccT '26: The 2026 ACM Conference on Fairness, Accountability, and Transparency","location":"Montreal QC Canada","acronym":"FAccT '26","sponsor":["ACM\/SIG"]},"container-title":["Proceedings of the 2026 ACM Conference on Fairness, Accountability, and Transparency"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805689.3812308","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T18:54:32Z","timestamp":1782759272000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805689.3812308"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,25]]},"references-count":61,"alternative-id":["10.1145\/3805689.3812308","10.1145\/3805689"],"URL":"https:\/\/doi.org\/10.1145\/3805689.3812308","relation":{},"subject":[],"published":{"date-parts":[[2026,6,25]]},"assertion":[{"value":"2026-06-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}