{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T07:44:15Z","timestamp":1784619855023,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":102,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,9,28]]},"DOI":"10.1145\/3746059.3747680","type":"proceedings-article","created":{"date-parts":[[2025,9,27]],"date-time":"2025-09-27T07:44:49Z","timestamp":1758959089000},"page":"1-24","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Policy Maps: Tools for Guiding the Unbounded Space of LLM Behaviors"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3448-5961","authenticated-orcid":false,"given":"Michelle S.","family":"Lam","sequence":"first","affiliation":[{"name":"Stanford University, Stanford, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4164-844X","authenticated-orcid":false,"given":"Fred","family":"Hohman","sequence":"additional","affiliation":[{"name":"Apple, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3110-1053","authenticated-orcid":false,"given":"Dominik","family":"Moritz","sequence":"additional","affiliation":[{"name":"Apple, Pittsburgh, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2072-0625","authenticated-orcid":false,"given":"Jeffrey P","family":"Bigham","sequence":"additional","affiliation":[{"name":"Apple, Pittsburgh, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6730-922X","authenticated-orcid":false,"given":"Kenneth","family":"Holstein","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1771-0565","authenticated-orcid":false,"given":"Mary Beth","family":"Kery","sequence":"additional","affiliation":[{"name":"Apple Inc., Pittsburgh, PA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,9,27]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Anthropic. 2023. Claude\u2019s Constitution. https:\/\/www.anthropic.com\/news\/claudes-constitution"},{"key":"e_1_3_3_2_3_2","unstructured":"Anthropic. 2024. The Claude 3 Model Family: Opus Sonnet Haiku. https:\/\/www-cdn.anthropic.com\/f2986af8d052f26236f6251da62d16172cfabd6e\/claude-3-model-card.pdf"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642016"},{"key":"e_1_3_3_2_5_2","unstructured":"Joshua Ashkinaze Ruijia Guan Laura Kurek Eytan Adar Ceren Budak and Eric Gilbert. 2024. Seeing Like an AI: How LLMs Apply (and Misapply) Wikipedia Neutrality Norms. arxiv:https:\/\/arXiv.org\/abs\/2407.04183\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2407.04183"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","unstructured":"Xuechunzi Bai Angelina Wang Ilia Sucholutsky and Thomas\u00a0L. Griffiths. 2025. Explicitly Unbiased Large Language Models Still Form Biased Associations. Proceedings of the National Academy of Sciences 122 8 (2025) e2416228122. 10.1073\/pnas.2416228122 arXiv:https:\/\/www.pnas.org\/doi\/pdf\/10.1073\/pnas.2416228122","DOI":"10.1073\/pnas.2416228122"},{"key":"e_1_3_3_2_7_2","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova DasSarma Dawn Drain Stanislav Fort Deep Ganguli Tom Henighan et\u00a0al. 2022. Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.05862 (2022)."},{"key":"e_1_3_3_2_8_2","unstructured":"Yuntao Bai Saurav Kadavath Sandipan Kundu Amanda Askell John Kernion Andy Jones Anna Chen Anna Goldie Azalia Mirhoseini Cameron McKinnon Carol Chen Catherine Olsson Christopher Olah Danny Hernandez Dawn Drain Deep Ganguli Dustin Li Eli Tran-Johnson E Perez Jamie Kerr Jared Mueller Jeff Ladish J Landau Kamal Ndousse Kamile Lukosuite Liane Lovitt Michael Sellitto Nelson Elhage Nicholas Schiefer Noem\u2019i Mercado Nova Dassarma Robert Lasenby Robin Larson Sam Ringer Scott Johnston Shauna Kravec Sheer\u00a0El Showk Stanislav Fort Tamera Lanham Timothy Telleen-Lawton Tom Conerly Tom Henighan Tristan Hume Sam Bowman Zac Hatfield-Dodds Benjamin Mann Dario Amodei Nicholas Joseph Sam McCandlish Tom\u00a0B. Brown and Jared Kaplan. 2022. Constitutional AI: Harmlessness from AI Feedback. ArXiv abs\/2212.08073 (2022). https:\/\/arxiv.org\/abs\/2212.08073"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Theodore\u00a0X Barber and Maurice\u00a0J Silver. 1968. Fact Fiction and the Experimenter Bias Effect. Psychological Bulletin 70 6p2 (1968) 1.","DOI":"10.1037\/h0026724"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3461702.3462610"},{"key":"e_1_3_3_2_11_2","unstructured":"Jorge\u00a0Luis Borges. 1998. On the exactitude of science. Collected Fictions. Translated by Andrew Hurley. New York: Penguin 325 (1998)."},{"key":"e_1_3_3_2_12_2","unstructured":"Zana Bu\u00e7inca Chau\u00a0Minh Pham Maurice Jakesch Marco\u00a0Tulio Ribeiro Alexandra Olteanu and Saleema Amershi. 2023. AHA!: Facilitating AI Impact Assessment by Generating Examples of Harms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.03280 (2023)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581268"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Eshwar Chandrasekharan Mattia Samory Shagun Jhaver Hunter Charvat Amy Bruckman Cliff Lampe Jacob Eisenstein and Eric Gilbert. 2018. The Internet\u2019s Hidden Rules: An Empirical Study of Reddit Norm Violations at Micro Meso and Macro Scales. Proc. ACM Hum.-Comput. Interact. 2 CSCW Article 32 (Nov. 2018) 25\u00a0pages. 10.1145\/3274301","DOI":"10.1145\/3274301"},{"key":"e_1_3_3_2_15_2","volume-title":"Constructing Grounded Theory: A Practical Guide through Qualitative Analysis","author":"Charmaz Kathy","year":"2006","unstructured":"Kathy Charmaz. 2006. Constructing Grounded Theory: A Practical Guide through Qualitative Analysis. Sage."},{"key":"e_1_3_3_2_16_2","unstructured":"Quan\u00a0Ze Chen and Amy\u00a0X. Zhang. 2023. Case Law Grounding: Aligning Judgments of Humans and AI on Socially-Constructed Concepts. arxiv:https:\/\/arXiv.org\/abs\/2310.07019\u00a0[cs.HC] https:\/\/arxiv.org\/abs\/2310.07019"},{"key":"e_1_3_3_2_17_2","unstructured":"Robert Chew John Bollenbacher Michael Wenger Jessica Speer and Annice Kim. 2023. LLM-Assisted Content Analysis: Using Large Language Models to Support Deductive Coding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.14924 (2023)."},{"key":"e_1_3_3_2_18_2","unstructured":"Wei-Lin Chiang Lianmin Zheng Ying Sheng Anastasios\u00a0Nikolas Angelopoulos Tianle Li Dacheng Li Hao Zhang Banghua Zhu Michael Jordan Joseph\u00a0E. Gonzalez and Ion Stoica. 2024. Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference. arxiv:https:\/\/arXiv.org\/abs\/2403.04132\u00a0[cs.AI]"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581026"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3593013.3594037"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3531146.3533240"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517441"},{"key":"e_1_3_3_2_23_2","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan Anirudh Goyal Anthony Hartshorn Aobo Yang Archi Mitra Archie Sravankumar Artem Korenev Arthur Hinsvark Arun Rao Aston Zhang Aurelien Rodriguez Austen Gregerson Ava Spataru Baptiste Roziere Bethany Biron Binh Tang Bobbie Chern Charlotte Caucheteux Chaya Nayak Chloe Bi Chris Marra and Chris\u00a0McConnell et al.2024. The Llama 3 Herd of Models. arxiv:https:\/\/arXiv.org\/abs\/2407.21783\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_3_2_24_2","volume-title":"International Conference on Learning Representations","author":"Eyuboglu Sabri","year":"2022","unstructured":"Sabri Eyuboglu, Maya Varma, Khaled\u00a0Kamal Saab, Jean-Benoit Delbrouck, Christopher Lee-Messer, Jared Dunnmon, James Zou, and Christopher Re. 2022. Domino: Discovering Systematic Errors with Cross-Modal Embeddings. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=FPCMqjI0jXN"},{"key":"e_1_3_3_2_25_2","unstructured":"K.\u00a0J.\u00a0Kevin Feng Quan\u00a0Ze Chen Inyoung Cheong King Xia and Amy\u00a0X. Zhang. 2023. Case Repositories: Towards Case-Based Reasoning for AI Alignment. arxiv:https:\/\/arXiv.org\/abs\/2311.10934\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2311.10934"},{"key":"e_1_3_3_2_26_2","unstructured":"K.\u00a0J.\u00a0Kevin Feng Inyoung Cheong Quan\u00a0Ze Chen and Amy\u00a0X Zhang. 2024. Policy Prototyping for LLMs: Pluralistic Alignment via Interactive and Collaborative Policymaking. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.08622 (2024)."},{"key":"e_1_3_3_2_27_2","unstructured":"Shangbin Feng Taylor Sorensen Yuhan Liu Jillian Fisher Chan\u00a0Young Park Yejin Choi and Yulia Tsvetkov. 2024. Modular Pluralism: Pluralistic Alignment via Multi-LLM Collaboration. arxiv:https:\/\/arXiv.org\/abs\/2406.15951\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2406.15951"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.1609\/icwsm.v12i1.15033"},{"key":"e_1_3_3_2_29_2","unstructured":"Arduin Findeis Timo Kaufmann Eyke H\u00fcllermeier Samuel Albanie and Robert Mullins. 2024. Inverse Constitutional AI: Compressing Preferences into Principles. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.06560 (2024)."},{"key":"e_1_3_3_2_30_2","unstructured":"Deep Ganguli Liane Lovitt Jackson Kernion Amanda Askell Yuntao Bai Saurav Kadavath Ben Mann Ethan Perez Nicholas Schiefer Kamal Ndousse Andy Jones Sam Bowman Anna Chen Tom Conerly Nova DasSarma Dawn Drain Nelson Elhage Sheer El-Showk Stanislav Fort Zac Hatfield-Dodds Tom Henighan Danny Hernandez Tristan Hume Josh Jacobson Scott Johnston Shauna Kravec Catherine Olsson Sam Ringer Eli Tran-Johnson Dario Amodei Tom Brown Nicholas Joseph Sam McCandlish Chris Olah Jared Kaplan and Jack Clark. 2022. Red Teaming Language Models to Reduce Harms: Methods Scaling Behaviors and Lessons Learned. arxiv:https:\/\/arXiv.org\/abs\/2209.07858\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2209.07858"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"Zorik Gekhman Gal Yona Roee Aharoni Matan Eyal Amir Feder Roi Reichart and Jonathan Herzig. 2024. Does Fine-Tuning LLMs on New Knowledge Encourage Hallucinations? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.05904 (2024).","DOI":"10.18653\/v1\/2024.emnlp-main.444"},{"key":"e_1_3_3_2_32_2","unstructured":"Amelia Glaese Nat McAleese Maja Tr\u0119bacz John Aslanides Vlad Firoiu Timo Ewalds Maribeth Rauh Laura Weidinger Martin Chadwick Phoebe Thacker et\u00a0al. 2022. Improving Alignment of Dialogue Agents via Targeted Human Judgements. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.14375 (2022)."},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658933"},{"key":"e_1_3_3_2_34_2","unstructured":"Tom Gunter Zirui Wang Chong Wang Ruoming Pang Andy Narayanan and et\u00a0al. Aonan\u00a0Zhang. 2024. Apple Intelligence Foundation Language Models. arxiv:https:\/\/arXiv.org\/abs\/2407.21075\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2407.21075"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","unstructured":"Jeffrey Heer and Dominik Moritz. 2024. Mosaic: An Architecture for Scalable & Interoperable Data Views. IEEE Transactions on Visualization and Computer Graphics 30 1 (2024) 436\u2013446. 10.1109\/TVCG.2023.3327189","DOI":"10.1109\/TVCG.2023.3327189"},{"key":"e_1_3_3_2_36_2","unstructured":"Dan Hendrycks Collin Burns Steven Basart Andy Zou Mantas Mazeika Dawn Song and Jacob Steinhardt. 2021. Measuring Massive Multitask Language Understanding. Proceedings of the International Conference on Learning Representations (ICLR) (2021)."},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300830"},{"key":"e_1_3_3_2_38_2","volume-title":"International Conference on Learning Representations","author":"Hu Edward\u00a0J","year":"2022","unstructured":"Edward\u00a0J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658979"},{"key":"e_1_3_3_2_40_2","unstructured":"Hakan Inan Kartikeya Upasani Jianfeng Chi Rashi Rungta Krithika Iyer Yuning Mao Michael Tontchev Qing Hu Brian Fuller Davide Testuggine and Madian Khabsa. 2023. Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations. arxiv:https:\/\/arXiv.org\/abs\/2312.06674\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2312.06674"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","DOI":"10.1145\/3442188.3445901"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","unstructured":"Nari Johnson \u00c1ngel\u00a0Alexander Cabrera Gregory Plumb and Ameet Talwalkar. 2023. Where Does My Model Underperform? A Human Evaluation of Slice Discovery Algorithms. Proceedings of the AAAI Conference on Human Computation and Crowdsourcing 11 1 (Nov. 2023) 65\u201376. 10.1609\/hcomp.v11i1.27548","DOI":"10.1609\/hcomp.v11i1.27548"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.324"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642216"},{"key":"e_1_3_3_2_45_2","unstructured":"Hannah\u00a0Rose Kirk Bertie Vidgen Paul R\u00f6ttger and Scott\u00a0A. Hale. 2023. Personalisation within bounds: A risk taxonomy and policy framework for the alignment of large language models with personalised feedback. arxiv:https:\/\/arXiv.org\/abs\/2303.05453\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2303.05453"},{"key":"e_1_3_3_2_46_2","unstructured":"Hannah\u00a0Rose Kirk Alexander Whitefield Paul R\u00f6ttger Andrew Bean Katerina Margatina Juan Ciro Rafael Mosquera Max Bartolo Adina Williams He He Bertie Vidgen and Scott\u00a0A. Hale. 2024. The PRISM Alignment Project: What Participatory Representative and Individualised Human Feedback Reveals About the Subjective and Multicultural Alignment of Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2404.16019\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2404.16019"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3650828"},{"key":"e_1_3_3_2_48_2","unstructured":"Deepak Kumar Yousef AbuHashem and Zakir Durumeric. 2024. Watch Your Language: Investigating Content Moderation with Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2309.14517\u00a0[cs.HC] https:\/\/arxiv.org\/abs\/2309.14517"},{"key":"e_1_3_3_2_49_2","unstructured":"Sandipan Kundu Yuntao Bai Saurav Kadavath Amanda Askell Andrew Callahan Anna Chen Anna Goldie Avital Balwit Azalia Mirhoseini Brayden McLean et\u00a0al. 2023. Specific versus General Principles for Constitutional AI. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.13798 (2023)."},{"key":"e_1_3_3_2_50_2","unstructured":"Tzu-Sheng Kuo Quan\u00a0Ze Chen Amy\u00a0X. Zhang Jane Hsieh Haiyi Zhu and Kenneth Holstein. 2024. PolicyCraft: Supporting Collaborative and Participatory Policy Design through Case-Grounded Deliberation. arxiv:https:\/\/arXiv.org\/abs\/2409.15644\u00a0[cs.HC] https:\/\/arxiv.org\/abs\/2409.15644"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642278"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","unstructured":"Michelle\u00a0S. Lam Mitchell\u00a0L. Gordon Dana\u00eb Metaxa Jeffrey\u00a0T. Hancock James\u00a0A. Landay and Michael\u00a0S. Bernstein. 2022. End-User Audits: A System Empowering Communities to Lead Large-Scale Investigations of Harmful Algorithmic Behavior. Proc. ACM Hum.-Comput. Interact. 6 CSCW2 Article 512 (Nov 2022) 34\u00a0pages. 10.1145\/3555625","DOI":"10.1145\/3555625"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581290"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642830"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539147"},{"key":"e_1_3_3_2_56_2","unstructured":"Kenneth Li Oam Patel Fernanda Vi\u00e9gas Hanspeter Pfister and Martin Wattenberg. 2024. Inference-Time Intervention: Eliciting Truthful Answers from a Language Model. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_2_57_2","unstructured":"Xinyu Li Zachary\u00a0C. Lipton and Liu Leqi. 2024. Personalized Language Modeling from Personalized Human Feedback. arxiv:https:\/\/arXiv.org\/abs\/2402.05133\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2402.05133"},{"key":"e_1_3_3_2_58_2","unstructured":"Yuxi Li. 2017. Deep Reinforcement Learning: An Overview. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1701.07274 (2017)."},{"key":"e_1_3_3_2_59_2","unstructured":"Percy Liang Rishi Bommasani Tony Lee Dimitris Tsipras Dilara Soylu Michihiro Yasunaga Yian Zhang Deepak Narayanan Yuhuai Wu Ananya Kumar Benjamin Newman Binhang Yuan Bobby Yan Ce Zhang Christian Cosgrove Christopher\u00a0D. Manning Christopher R\u00e9 Diana Acosta-Navas Drew\u00a0A. Hudson Eric Zelikman Esin Durmus Faisal Ladhak Frieda Rong Hongyu Ren Huaxiu Yao Jue Wang Keshav Santhanam Laurel Orr Lucia Zheng Mert Yuksekgonul Mirac Suzgun Nathan Kim Neel Guha Niladri Chatterji Omar Khattab Peter Henderson Qian Huang Ryan Chi Sang\u00a0Michael Xie Shibani Santurkar Surya Ganguli Tatsunori Hashimoto Thomas Icard Tianyi Zhang Vishrav Chaudhary William Wang Xuechen Li Yifan Mai Yuhui Zhang and Yuta Koreeda. 2023. Holistic Evaluation of Language Models. arxiv:https:\/\/arXiv.org\/abs\/2211.09110\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2211.09110"},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","unstructured":"Michael Madaio Lisa Egede Hariharan Subramonyam Jennifer Wortman\u00a0Vaughan and Hanna Wallach. 2022. Assessing the Fairness of AI Systems: AI Practitioners\u2019 Processes Challenges and Needs for Support. Proc. ACM Hum.-Comput. Interact. 6 CSCW1 Article 52 (April 2022) 26\u00a0pages. 10.1145\/3512899","DOI":"10.1145\/3512899"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"crossref","unstructured":"Vivien Marx. 2024. Seeing Data as t-SNE and UMAP Do. Nature Methods 21 6 (2024) 930\u2013933.","DOI":"10.1038\/s41592-024-02301-x"},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"publisher","unstructured":"Nora McDonald Sarita Schoenebeck and Andrea Forte. 2019. Reliability and Inter-Rater Reliability in Qualitative Research: Norms and Guidelines for CSCW and HCI Practice. Proc. ACM Hum.-Comput. Interact. 3 CSCW Article 72 (nov 2019) 23\u00a0pages. 10.1145\/3359174","DOI":"10.1145\/3359174"},{"key":"e_1_3_3_2_63_2","unstructured":"L. McInnes J. Healy and J. Melville. 2018. UMAP: Uniform Manifold Approximation and Projection for Dimension Reduction. ArXiv e-prints (Feb. 2018). arxiv:https:\/\/arXiv.org\/abs\/1802.03426\u00a0[stat.ML]"},{"key":"e_1_3_3_2_64_2","unstructured":"Microsoft. 2025. Content Filtering Overview. https:\/\/learn.microsoft.com\/en-us\/azure\/ai-foundry\/openai\/concepts\/content-filter"},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581242"},{"key":"e_1_3_3_2_66_2","unstructured":"Tong Mu Alec Helyar Johannes Heidecke Joshua Achiam Andrea Vallone Ian Kivlichan Molly Lin Alex Beutel John Schulman and Lilian Weng. 2024. Rule Based Rewards for Language Model Safety. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.01111 (2024)."},{"key":"e_1_3_3_2_67_2","unstructured":"Nadia Nahar Christian K\u00e4stner Jenna Butler Chris Parnin Thomas Zimmermann and Christian Bird. 2024. Beyond the Comfort Zone: Emerging Solutions to Overcome Challenges in Integrating LLMs into Software Products. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.12071 (2024)."},{"key":"e_1_3_3_2_68_2","unstructured":"OpenAI. 2025. Sharing the latest Model Spec. https:\/\/openai.com\/index\/sharing-the-latest-model-spec\/"},{"key":"e_1_3_3_2_69_2","series-title":"(NIPS \u201922)","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeff Wu, Xu Jiang, Diogo Almeida, Carroll\u00a0L. Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kelton, Luke Miller, Maddie Simens, Amanda Askell, Peter Welinder, Paul Christiano, Jan Leike, and Ryan Lowe. 2022. Training language models to follow instructions with human feedback. In Proceedings of the 36th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201922). Curran Associates Inc., Red Hook, NY, USA, Article 2011, 15\u00a0pages."},{"key":"e_1_3_3_2_70_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642054"},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640543.3645144"},{"key":"e_1_3_3_2_72_2","doi-asserted-by":"publisher","DOI":"10.1145\/3299869.3320212"},{"key":"e_1_3_3_2_73_2","volume-title":"Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks","volume":"1","author":"Raji Deborah","year":"2021","unstructured":"Deborah Raji, Emily Denton, Emily\u00a0M. Bender, Alex Hanna, and Amandalynne Paullada. 2021. AI and the Everything in the Whole Wide World Benchmark. In Proceedings of the Neural Information Processing Systems Track on Datasets and Benchmarks , J.\u00a0Vanschoren and S.\u00a0Yeung (Eds.), Vol.\u00a01. https:\/\/datasets-benchmarks-proceedings.neurips.cc\/paper_files\/paper\/2021\/file\/084b6fbb10729ed4da8c3d3f5a3ae7c9-Paper-round2.pdf"},{"key":"e_1_3_3_2_74_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_3_2_75_2","unstructured":"Donghao Ren Fred Hohman Halden Lin and Dominik Moritz. 2025. Embedding Atlas: Low-Friction Interactive Embedding Visualization. arxiv:https:\/\/arXiv.org\/abs\/2505.06386\u00a0[cs.HC] https:\/\/arxiv.org\/abs\/2505.06386"},{"key":"e_1_3_3_2_76_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.399"},{"key":"e_1_3_3_2_77_2","doi-asserted-by":"publisher","DOI":"10.1145\/3375627.3375804"},{"key":"e_1_3_3_2_78_2","volume-title":"The Twelfth International Conference on Learning Representations","author":"Sclar Melanie","year":"2024","unstructured":"Melanie Sclar, Yejin Choi, Yulia Tsvetkov, and Alane Suhr. 2024. Quantifying Language Models\u2019 Sensitivity to Spurious Features in Prompt Design or: How I learned to start worrying about prompt formatting. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=RIu5lyNXjT"},{"key":"e_1_3_3_2_79_2","doi-asserted-by":"crossref","unstructured":"Shreya Shankar J.\u00a0D. Zamfirescu-Pereira Bj\u00f6rn Hartmann Aditya\u00a0G. Parameswaran and Ian Arawjo. 2024. Who Validates the Validators? Aligning LLM-Assisted Evaluation of LLM Outputs with Human Preferences. arxiv:https:\/\/arXiv.org\/abs\/2404.12272\u00a0[cs.HC] https:\/\/arxiv.org\/abs\/2404.12272","DOI":"10.1145\/3654777.3676450"},{"key":"e_1_3_3_2_80_2","unstructured":"Mrinank Sharma Meg Tong Tomasz Korbak David\u00a0Kristjanson Duvenaud Amanda Askell Samuel\u00a0R. Bowman Newton Cheng Esin Durmus Zac Hatfield-Dodds Scott Johnston Shauna Kravec Tim Maxwell Sam McCandlish Kamal Ndousse Oliver Rausch Nicholas Schiefer Da Yan Miranda Zhang and Ethan Perez. 2023. Towards Understanding Sycophancy in Language Models. ArXiv abs\/2310.13548 (2023). https:\/\/arxiv.org\/abs\/2310.13548"},{"key":"e_1_3_3_2_81_2","doi-asserted-by":"publisher","unstructured":"Hong Shen Alicia DeVos Motahhare Eslami and Kenneth Holstein. 2021. Everyday Algorithm Auditing: Understanding the Power of Everyday Users in Surfacing Harmful Algorithmic Behaviors. Proc. ACM Hum.-Comput. Interact. 5 CSCW2 Article 433 (oct 2021) 29\u00a0pages. 10.1145\/3479577","DOI":"10.1145\/3479577"},{"key":"e_1_3_3_2_82_2","unstructured":"Taylor Sorensen Jared Moore Jillian Fisher Mitchell Gordon Niloofar Mireshghallah Christopher\u00a0Michael Rytting Andre Ye Liwei Jiang Ximing Lu Nouha Dziri Tim Althoff and Yejin Choi. 2024. A Roadmap to Pluralistic Alignment. arxiv:https:\/\/arXiv.org\/abs\/2402.05070\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2402.05070"},{"key":"e_1_3_3_2_83_2","unstructured":"Asa\u00a0Cooper Stickland Alexander Lyzhov Jacob Pfau Salsabila Mahdi and Samuel\u00a0R Bowman. 2024. Steering Without Side Effects: Improving Post-Deployment Control of Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.15518 (2024)."},{"key":"e_1_3_3_2_84_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581482"},{"key":"e_1_3_3_2_85_2","doi-asserted-by":"publisher","DOI":"10.1145\/3630106.3658992"},{"key":"e_1_3_3_2_86_2","unstructured":"Richard\u00a0S Sutton. 2018. Reinforcement Learning: An Introduction. A Bradford Book (2018)."},{"key":"e_1_3_3_2_87_2","unstructured":"Alex Tamkin Amanda Askell Liane Lovitt Esin Durmus Nicholas Joseph Shauna Kravec Karina Nguyen Jared Kaplan and Deep Ganguli. 2023. Evaluating and Mitigating Discrimination in Language Model Decisions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.03689 (2023)."},{"key":"e_1_3_3_2_88_2","unstructured":"Alex Tamkin Miles McCain Kunal Handa Esin Durmus Liane Lovitt Ankur Rathi Saffron Huang Alfred Mountfield Jerry Hong Stuart Ritchie et\u00a0al. 2024. Clio: Privacy-Preserving Insights into Real-World AI Use. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.13678 (2024)."},{"key":"e_1_3_3_2_89_2","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk and et al.2024. Gemini: A Family of Highly Capable Multimodal Models. arxiv:https:\/\/arXiv.org\/abs\/2312.11805\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2312.11805"},{"key":"e_1_3_3_2_90_2","doi-asserted-by":"crossref","unstructured":"Inga Ulnicane William Knight Tonii Leach Bernd\u00a0Carsten Stahl and Winter-Gladys Wanjiku. 2021. Framing governance for a contested emerging technology: insights from AI policy. Policy and Society 40 2 (2021) 158\u2013177.","DOI":"10.1080\/14494035.2020.1855800"},{"key":"e_1_3_3_2_91_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581278"},{"key":"e_1_3_3_2_92_2","unstructured":"Zijie\u00a0J. Wang Fred Hohman and Duen\u00a0Horng Chau. 2023. WizMap: Scalable Interactive Visualization for Exploring Large Machine Learning Embeddings. arXiv 2306.09328 (2023). http:\/\/arxiv.org\/abs\/2306.09328"},{"key":"e_1_3_3_2_93_2","volume-title":"CHI Conference on Human Factors in Computing Systems","author":"Wang Zijie\u00a0J.","year":"2024","unstructured":"Zijie\u00a0J. Wang, Chinmay Kulkarni, Lauren Wilcox, Michael Terry, and Michael Madaio. 2024. Farsight: Fostering Responsible AI Awareness During AI Application Prototyping. In CHI Conference on Human Factors in Computing Systems."},{"key":"e_1_3_3_2_94_2","doi-asserted-by":"crossref","unstructured":"Martin Wattenberg Fernanda Vi\u00e9gas and Ian Johnson. 2016. How to Use t-SNE Effectively. Distill 1 10 (2016).","DOI":"10.23915\/distill.00002"},{"key":"e_1_3_3_2_95_2","doi-asserted-by":"crossref","unstructured":"Laura Weidinger John Mellor Bernat\u00a0Guillen Pegueroles Nahema Marchal Ravin Kumar Kristian Lum Canfer Akbulut Mark Diaz Stevie Bergman Mikel Rodriguez et\u00a0al. 2024. STAR: SocioTechnical Approach to Red Teaming Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.11757 (2024).","DOI":"10.18653\/v1\/2024.emnlp-main.1200"},{"key":"e_1_3_3_2_96_2","unstructured":"Lilian Weng Vik Goel and Andrea Vallone. 2023. Using GPT-4 for Content Moderation. OpenAI (2023). https:\/\/openai.com\/index\/using-gpt-4-for-content-moderation\/"},{"key":"e_1_3_3_2_97_2","unstructured":"Zhengxuan Wu Aryaman Arora Zheng Wang Atticus Geiger Dan Jurafsky Christopher\u00a0D. Manning and Christopher Potts. 2024. ReFT: Representation Finetuning for Language Models. arxiv.org\/abs\/2404.03592"},{"key":"e_1_3_3_2_98_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581754.3584136"},{"key":"e_1_3_3_2_99_2","unstructured":"Leon Yin Davey Alba and Leonardo Nicolleti. 2024. OpenAI\u2019s GPT is a Recruiter\u2019s Dream Tool. Tests Show There\u2019s Racial Bias. Bloomberg (2024). https:\/\/www.bloomberg.com\/graphics\/2024-openai-gpt-hiring-racial-discrimination"},{"key":"e_1_3_3_2_100_2","doi-asserted-by":"publisher","unstructured":"Angie Zhang Olympia Walker Kaci Nguyen Jiajun Dai Anqing Chen and Min\u00a0Kyung Lee. 2023. Deliberating with AI: Improving Decision-Making for the Future through Participatory AI Design and Stakeholder Deliberation. Proc. ACM Hum.-Comput. Interact. 7 CSCW1 Article 125 (apr 2023) 32\u00a0pages. 10.1145\/3579601","DOI":"10.1145\/3579601"},{"key":"e_1_3_3_2_101_2","series-title":"(NIPS \u201923)","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Zheng Lianmin","year":"2024","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric\u00a0P. Xing, Hao Zhang, Joseph\u00a0E. Gonzalez, and Ion Stoica. 2024. Judging LLM-as-a-judge with MT-bench and Chatbot Arena. In Proceedings of the 37th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201923). Curran Associates Inc., Red Hook, NY, USA, Article 2020, 29\u00a0pages."},{"key":"e_1_3_3_2_102_2","doi-asserted-by":"publisher","unstructured":"Caleb Ziems William Held Omar Shaikh Jiaao Chen Zhehao Zhang and Diyi Yang. 2024. Can Large Language Models Transform Computational Social Science? Computational Linguistics (02 2024) 1\u201355. 10.1162\/colia00502 arXiv:https:\/\/direct.mit.edu\/coli\/article-pdf\/doi\/10.1162\/coli_a_00502\/2332904\/coli_a_00502.pdf","DOI":"10.1162\/colia00502"},{"key":"e_1_3_3_2_103_2","unstructured":"Andy Zou Long Phan Sarah Chen James Campbell Phillip Guo Richard Ren Alexander Pan Xuwang Yin Mantas Mazeika Ann-Kathrin Dombrowski Shashwat Goel Nathaniel Li Michael\u00a0J. Byun Zifan Wang Alex Mallen Steven Basart Sanmi Koyejo Dawn Song Matt Fredrikson Zico Kolter and Dan Hendrycks. 2023. Representation Engineering: A Top-Down Approach to AI Transparency. arxiv:https:\/\/arXiv.org\/abs\/2310.01405\u00a0[cs.CL]"}],"event":{"name":"UIST '25: The 38th Annual ACM Symposium on User Interface Software and Technology","location":"Busan Republic of Korea","acronym":"UIST '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction","SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the 38th Annual ACM Symposium on User Interface Software and Technology"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746059.3747680","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,27]],"date-time":"2025-09-27T22:10:17Z","timestamp":1759011017000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746059.3747680"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,27]]},"references-count":102,"alternative-id":["10.1145\/3746059.3747680","10.1145\/3746059"],"URL":"https:\/\/doi.org\/10.1145\/3746059.3747680","relation":{},"subject":[],"published":{"date-parts":[[2025,9,27]]},"assertion":[{"value":"2025-09-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}