{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T18:45:07Z","timestamp":1782758707453,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":107,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,25]]},"DOI":"10.1145\/3805689.3806459","type":"proceedings-article","created":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T17:52:08Z","timestamp":1782755528000},"page":"6510-6553","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Automated bias blind spots: Examining stereotype associations in VLM-as-a-judge paradigms"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-4275-653X","authenticated-orcid":false,"given":"Rachel","family":"Hong","sequence":"first","affiliation":[{"name":"University of Washington, Seattle, Washington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0531-7520","authenticated-orcid":false,"given":"Anaelia","family":"Ovalle","sequence":"additional","affiliation":[{"name":"FAIR at Meta, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3063-3207","authenticated-orcid":false,"given":"Megan","family":"Ung","sequence":"additional","affiliation":[{"name":"FAIR at Meta, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7269-6813","authenticated-orcid":false,"given":"Evangelia","family":"Spiliopoulou","sequence":"additional","affiliation":[{"name":"FAIR at Meta, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5403-4124","authenticated-orcid":false,"given":"Levent","family":"Sagun","sequence":"additional","affiliation":[{"name":"FAIR at Meta, Paris, France"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5281-3343","authenticated-orcid":false,"given":"Adina","family":"Williams","sequence":"additional","affiliation":[{"name":"FAIR at Meta, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6715-8086","authenticated-orcid":false,"given":"Candace","family":"Ross","sequence":"additional","affiliation":[{"name":"FAIR at Meta, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,25]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Speceval: Evaluating model adherence to behavior specifications.","author":"Ahmed Ahmed","year":"2025","unstructured":"Ahmed Ahmed, Kevin Klyman, Yi Zeng, Sanmi Koyejo, and Percy Liang. 2025. Speceval: Evaluating model adherence to behavior specifications."},{"key":"e_1_3_2_1_2_1","unstructured":"Meta AI. 2025. Llama 4: Leading multimodal intelligence. https:\/\/ai.meta.com\/blog\/llama-4-multimodal-intelligence\/"},{"key":"e_1_3_2_1_3_1","volume-title":"Sejin Paik, and Derry Wijaya.","author":"Aky\u00fcrek Afra Feyza","year":"2022","unstructured":"Afra Feyza Aky\u00fcrek, Muhammed Yusuf Kocyigit, Sejin Paik, and Derry Wijaya. 2022. Challenges in measuring bias via open-ended language generation."},{"key":"e_1_3_2_1_4_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2.5-VL technical report."},{"key":"e_1_3_2_1_5_1","volume-title":"Constitutional AI: Harmlessness from AI feedback.","author":"Bai Yuntao","year":"2022","unstructured":"Yuntao Bai, Saurav Kadavath, Sandipan Kundu, Amanda Askell, Jackson Kernion, Andy Jones, Anna Chen, Anna Goldie, Azalia Mirhoseini, Cameron McKinnon, et al. 2022. Constitutional AI: Harmlessness from AI feedback."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.169"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-025-14875-3"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.81"},{"key":"e_1_3_2_1_9_1","volume-title":"Eval Factsheets: A Structured Framework for Documenting AI Evaluations. arXiv:2512.04062 [cs.LG] https:\/\/arxiv.org\/abs\/2512.04062","author":"Bordes Florian","year":"2025","unstructured":"Florian Bordes, Candace Ross, Justine T Kao, Evangelia Spiliopoulou, and Adina Williams. 2025. Eval Factsheets: A Structured Framework for Documenting AI Evaluations. arXiv:2512.04062 [cs.LG] https:\/\/arxiv.org\/abs\/2512.04062"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-short.62"},{"key":"e_1_3_2_1_11_1","volume-title":"Jackie Chi Kit Cheung, and Golnoosh Farnadi","author":"Chehbouni Khaoula","year":"2025","unstructured":"Khaoula Chehbouni, Mohammed Haddou, Jackie Chi Kit Cheung, and Golnoosh Farnadi. 2025. Neither valid nor reliable? Investigating the use of LLMs as judges."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning. JMLR.org","author":"Chen Dongping","year":"2024","unstructured":"Dongping Chen, Ruoxi Chen, Shilin Zhang, Yaochen Wang, Yinuo Liu, Huichi Zhou, Qihui Zhang, Yao Wan, Pan Zhou, and Lichao Sun. 2024. MLLM-as-a-judge: Assessing multimodal LLM-as-a-judge with vision-language benchmark. In Proceedings of the 41st International Conference on Machine Learning. JMLR.org, Vienna, Austria, Article 254, 34 pages."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.474"},{"key":"e_1_3_2_1_14_1","unstructured":"Zhaorun Chen Yichao Du Zichen Wen Yiyang Zhou Chenhang Cui Zhenzhen Weng Haoqin Tu Chaoqi Wang Zhengwei Tong Qinglan Huang et al. 2024. MJ-Bench: Is your multimodal reward model really a good judge for text-to-image generation?"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.84"},{"key":"e_1_3_2_1_16_1","volume-title":"Kartikeya Upasani, and Mahesh Pasupuleti.","author":"Chi Jianfeng","year":"2024","unstructured":"Jianfeng Chi, Ujjwal Karn, Hongyuan Zhan, Eric Smith, Javier Rando, Yiming Zhang, Kate Plawiak, Zacharie Delpierre Coudert, Kartikeya Upasani, and Mahesh Pasupuleti. 2024. Llama Guard 3 Vision: Safeguarding human-AI image understanding conversations."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.870"},{"key":"e_1_3_2_1_18_1","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.122"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.656"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlpmain.23"},{"key":"e_1_3_2_1_22_1","unstructured":"Florian E Dorner Vivian Y Nastl and Moritz Hardt. 2024. Limits to scalable evaluation at the frontier: LLM as Judge won't beat twice the data."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.739"},{"key":"e_1_3_2_1_24_1","unstructured":"Tyna Eloundou Alex Beutel David G Robinson Keren Gu-Lemberg Anna-Luisa Brakman Pamela Mishkin Meghan Shah Johannes Heidecke Lilian Weng and Adam Tauman Kalai. 2024. First-person fairness in chatbots."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.230"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.eacl-long.41"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.955"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715275.3732181"},{"key":"e_1_3_2_1_29_1","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et al. 2024. A survey on LLM-as-a-judge."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01863"},{"key":"e_1_3_2_1_31_1","unstructured":"Xanh Ho Jiahao Huang Florian Boudin and Akiko Aizawa. 2025. LLM-as-a-judge: Reassessing the performance of LLMs in extractive QA."},{"key":"e_1_3_2_1_32_1","unstructured":"Tom Hosking Phil Blunsom and Max Bartolo. 2023. Human feedback is not gold standard."},{"key":"e_1_3_2_1_33_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3713220"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715275.3732005"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.29"},{"key":"e_1_3_2_1_37_1","unstructured":"Michael Krumdick Charles Lovering Varshini Reddy Seth Ebner and Chris Tanner. 2025. No free labels: Limitations of LLM-as-a-judge without human grounding."},{"key":"e_1_3_2_1_38_1","unstructured":"Shachi H Kumar Saurav Sahay Sahisnu Mazumder Eda Okur Ramesh Manuvinakurike Nicole Beckage Hsuan Su Hung-yi Lee and Lama Nachman. 2024. Decoding biases: Automated methods and LLM judges for gender bias detection in language models."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-naacl.96"},{"key":"e_1_3_2_1_41_1","unstructured":"Bo Li Yuanhan Zhang Dong Guo Renrui Zhang Feng Li Hao Zhang Kaichen Zhang Peiyuan Zhang Yanwei Li Ziwei Liu et al. 2024. LLaVA-OneVision: Easy visual task transfer."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.138"},{"key":"e_1_3_2_1_43_1","unstructured":"Haitao Li Qian Dong Junjie Chen Huixue Su Yujia Zhou Qingyao Ai Ziyi Ye and Yiqun Liu. 2024. LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods. arXiv:2412.05579 [cs.CL] https:\/\/arxiv.org\/abs\/2412.05579"},{"key":"e_1_3_2_1_44_1","unstructured":"Xiang Lisa Li Vaishnavi Shrivastava Siyan Li Tatsunori Hashimoto and Percy Liang. 2023. Benchmarking and improving generator-validator consistency of language models."},{"key":"e_1_3_2_1_45_1","volume-title":"Chongren Sun, Di Wu, and Benoit Boulet.","author":"Li Yuran","year":"2025","unstructured":"Yuran Li, Jama Hussein Mohamud, Chongren Sun, Di Wu, and Benoit Boulet. 2025. Leveraging LLMs as meta-judges: A multi-agent framework for evaluating LLM judgments."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1580"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.eacl-long.74"},{"key":"e_1_3_2_1_48_1","unstructured":"Yixin Liu Pengfei Liu and Arman Cohan. 2025. On Evaluating LLM Alignment by Evaluating LLMs as Judges."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i13.29333"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.7"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1098\/rsos.240255"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2019"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Simon Malberg Roman Poletukhin Carolin M Schuster and Georg Groh. 2024. A comprehensive evaluation of cognitive biases in LLMs.","DOI":"10.18653\/v1\/2025.nlp4dh-1.50"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3492853"},{"key":"e_1_3_2_1_55_1","unstructured":"Vishal Narnaware Ashmal Vayani Rohit Gupta Sirnam Swetha and Mubarak Shah. 2025. SB-Bench: Stereotype bias benchmark for large multimodal models."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.eacl-srw.19"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715275.3732196"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2197"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.165"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.130"},{"key":"e_1_3_2_1_61_1","volume-title":"Objectivity in the eye of the beholder: Divergent perceptions of bias in self versus others. Psychological review 111, 3","author":"Pronin Emily","year":"2004","unstructured":"Emily Pronin, Thomas Gilovich, and Lee Ross. 2004. Objectivity in the eye of the beholder: Divergent perceptions of bias in self versus others. Psychological review 111, 3 (2004), 781."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1177\/0146167202286008"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.646"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.243"},{"key":"e_1_3_2_1_65_1","volume-title":"Pushkar Mishra, Roma Patel, Ding Wang, Mark D\u00edaz, Alicia Parrish, Aida Mostafazadeh Davani, Zoe Ashwood, Michela Paganini, et al.","author":"Rastogi Charvi","year":"2025","unstructured":"Charvi Rastogi, Tian Huey Teh, Pushkar Mishra, Roma Patel, Ding Wang, Mark D\u00edaz, Alicia Parrish, Aida Mostafazadeh Davani, Zoe Ashwood, Michela Paganini, et al. 2025. Whose view of safety? A deep DIVE dataset for pluralistic alignment of text-to-image models."},{"key":"e_1_3_2_1_66_1","volume-title":"Beads: Bias evaluation across domains.","author":"Raza Shaina","year":"2024","unstructured":"Shaina Raza, Mizanur Rahman, and Michael R Zhang. 2024. Beads: Bias evaluation across domains."},{"key":"e_1_3_2_1_67_1","unstructured":"Shaina Raza Caesar Saleh Emrul Hasan Franklin Ogidi Maximus Powers Veronica Chatrath Marcelo Lotif Roya Javadi Anam Zahid and Vahid Reza Khazaie. 2024. ViLBias: A study of bias detection through linguistic and visual cues presenting annotation strategies evaluation and key challenges."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.78"},{"key":"e_1_3_2_1_69_1","volume-title":"Values and knowledge","author":"Ross Lee","unstructured":"Lee Ross and Andrew Ward. 1996. Naive realism in everyday life: Implications for social conflict and misunderstanding. In Values and knowledge. Lawrence Erlbaum Associates, Inc, Mahwah, NJ, USA, 103\u2013135."},{"key":"e_1_3_2_1_70_1","volume-title":"Rojin Ziaei, Jason Eshraghian, Peter Abadir, and Rama Chellappa.","author":"Schmidgall Samuel","year":"2024","unstructured":"Samuel Schmidgall, Carl Harris, Ime Essien, Daniel Olshvang, Tawsifur Rahman, Ji Woong Kim, Rojin Ziaei, Jason Eshraghian, Peter Abadir, and Rama Chellappa. 2024. Evaluation and mitigation of cognitive biases in medical language models. npj Digital Medicine 7, 1 (2024), 295."},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00659"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.244"},{"key":"e_1_3_2_1_73_1","unstructured":"Mrinank Sharma Meg Tong Tomasz Korbak David Duvenaud Amanda Askell Samuel R Bowman Newton Cheng Esin Durmus Zac Hatfield-Dodds Scott R Johnston et al. 2023. Towards understanding sycophancy in language models."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600211.3604673"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.154"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"crossref","unstructured":"Lin Shi Chiyu Ma Wenhua Liang Xingjian Diao Weicheng Ma and Soroush Vosoughi. 2024. Judging the judges: A systematic study of position bias in LLM-as-a-judge.","DOI":"10.18653\/v1\/2025.ijcnlp-long.18"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.625"},{"key":"e_1_3_2_1_78_1","unstructured":"Evangelia Spiliopoulou Riccardo Fogliato Hanna Burnsky Tamer Soliman Jie Ma Graham Horwood and Miguel Ballesteros. 2025. Play favorites: A statistical method to measure self-bias in LLM-as-a-judge."},{"key":"e_1_3_2_1_79_1","unstructured":"Andreas Stephan Dawei Zhu Matthias A\u00dfenmacher Xiaoyu Shen and Benjamin Roth. 2025. From Calculation to Adjudication: Examining LLM Judges on Mathematical Reasoning Tasks. In Proceedings of the Fourth Workshop on Generation Evaluation and Metrics (GEM2) Ofir Arviv Miruna Clinciu Kaustubh Dhole Rotem Dror Sebastian Gehrmann Eliya Habba Itay Itzhak Simon Mille Yotam Perlitz Enrico Santus Jo\u00e4o Sedoc Michal Shmueli Scheuer Gabriel Stanovsky and Oyvind Tafjord (Eds.). Association for Computational Linguistics Vienna Austria and virtual meeting 759\u2013773. https:\/\/aclanthology.org\/2025.gem-1.65\/"},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-long.50"},{"key":"e_1_3_2_1_81_1","unstructured":"Zhengwei Tao Ting-En Lin Xiancai Chen Hangyu Li Yuchuan Wu Yongbin Li Zhi Jin Fei Huang Dacheng Tao and Jingren Zhou. 2024. A survey on self-evolution of large language models."},{"key":"e_1_3_2_1_82_1","unstructured":"Gemma Team Aishwarya Kamath Johan Ferret Shreya Pathak Nino Vieillard Ramona Merhej Sarah Perrin Tatiana Matejovicova Alexandre Ram\u00e9 Morgane Rivi\u00e8re et al. 2025. Gemma 3 technical report."},{"key":"e_1_3_2_1_83_1","volume-title":"Proceedings of the Fourth Workshop on Generation, Evaluation and Metrics (GEM2). Association for Computational Linguistics","author":"Thakur Aman Singh","year":"2025","unstructured":"Aman Singh Thakur, Kartik Choudhary, Venkat Srinik Ramayapally, Sankaran Vaidyanathan, and Dieuwke Hupkes. 2025. Judging the judges: Evaluating alignment and vulnerabilities in LLMs-as-judges. In Proceedings of the Fourth Workshop on Generation, Evaluation and Metrics (GEM2). Association for Computational Linguistics, Vienna, Austria and virtual meeting, 404\u2013430. https: \/\/aclanthology.org\/2025.gem-1.33\/"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1247"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.330"},{"key":"e_1_3_2_1_86_1","unstructured":"Tuhina Tripathi Manya Wadhwa Greg Durrett and Scott Niekum. 2025. Pairwise or pointwise? Evaluating feedback protocols for bias in LLM-based evaluation."},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3275"},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715275.3732046"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-025-00986-z"},{"key":"e_1_3_2_1_90_1","unstructured":"Qian Wang Zhanzhi Lou Zhenheng Tang Nuo Chen Xuandong Zhao Wenxuan Zhang Dawn Song and Bingsheng He. 2025. Assessing judging bias in large reasoning models: An empirical study."},{"key":"e_1_3_2_1_91_1","unstructured":"Sibo Wang Xiangkui Cao Jie Zhang Zheng Yuan Shiguang Shan Xilin Chen and Wen Gao. 2024. VLBiasBench: A comprehensive benchmark for evaluating bias in large vision-language model."},{"key":"e_1_3_2_1_92_1","unstructured":"Koki Wataoka Tsubasa Takahashi and Ryokan Ri. 2024. Self-preference bias in LLM-as-a-judge."},{"key":"e_1_3_2_1_93_1","unstructured":"Peter West Ximing Lu Nouha Dziri Faeze Brahman Linjie Li Jena D Hwang Liwei Jiang Jillian Fisher Abhilasha Ravichander Khyathi Chandu et al. 2023. The Generative AI Paradox: \u201cWhat it can create it may not understand\u201d."},{"key":"e_1_3_2_1_94_1","unstructured":"Addison J Wu Ryan Liu Xuechunzi Bai and Thomas L Griffiths. 2025. Large language models develop novel social biases through adaptive exploration."},{"key":"e_1_3_2_1_95_1","unstructured":"Tianhao Wu Weizhe Yuan Olga Golovneva Jing Xu Yuandong Tian Jiantao Jiao Jason Weston and Sainbayar Sukhbaatar. 2024. Meta-rewarding language models: Self-improving alignment with LLM-as-a-meta-judge. arXiv:2407.19594 [cs.CL] https:\/\/arxiv.org\/abs\/2407.19594"},{"key":"e_1_3_2_1_96_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.1006"},{"key":"e_1_3_2_1_97_1","unstructured":"Zekun Wu Sahan Bulathwela Maria Perez-Ortiz and Adriano Soares Koshiyama. 2024. Stereotype detection in LLMs: A multiclass explainable and benchmark-Driven Approach."},{"key":"e_1_3_2_1_98_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. IEEE","author":"Xiong Tianyi","year":"2025","unstructured":"Tianyi Xiong, Xiyao Wang, Dong Guo, Qinghao Ye, Haoqi Fan, Quanquan Gu, Heng Huang, and Chunyuan Li. 2025. LLaVA-Critic: Learning to evaluate multimodal models. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. IEEE, Nashville, Tennessee, 13618\u201313628."},{"key":"e_1_3_2_1_99_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00202"},{"key":"e_1_3_2_1_100_1","unstructured":"Michihiro Yasunaga Luke Zettlemoyer and Marjan Ghazvininejad. 2025. Multimodal RewardBench: Holistic evaluation of reward models for vision language models."},{"key":"e_1_3_2_1_101_1","unstructured":"Jiayi Ye Yanbo Wang Yue Huang Dongping Chen Qihui Zhang Nuno Moniz Tian Gao Werner Geyer Chao Huang Pin-Yu Chen et al. 2024. Justice or prejudice? Quantifying biases in LLM-as-a-judge."},{"key":"e_1_3_2_1_102_1","volume-title":"Taken for Granted","author":"Zerubavel Eviatar","unstructured":"Eviatar Zerubavel. 2018. Taken for granted: The remarkable power of the unremarkable. In Taken for Granted. Princeton University Press, Princeton, NJ, USA."},{"key":"e_1_3_2_1_103_1","volume-title":"Mm-llms: Recent advances in multimodal large language models.","author":"Zhang Duzhen","year":"2024","unstructured":"Duzhen Zhang, Yahan Yu, Jiahua Dong, Chenxing Li, Dan Su, Chenhui Chu, and Dong Yu. 2024. Mm-llms: Recent advances in multimodal large language models."},{"key":"e_1_3_2_1_104_1","unstructured":"Zhuosheng Zhang Aston Zhang Mu Li and Alex Smola. 2022. Automatic chain of thought prompting in large language models."},{"key":"e_1_3_2_1_105_1","doi-asserted-by":"crossref","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric Xing et al. 2023. Judging LLM-as-a-judge with MT-Bench and Chatbot Arena. Advances in neural information processing systems 36 (2023) 46595\u201346623.","DOI":"10.52202\/075280-2020"},{"key":"e_1_3_2_1_106_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1034"},{"key":"e_1_3_2_1_107_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715275.3732067"}],"event":{"name":"FAccT '26: The 2026 ACM Conference on Fairness, Accountability, and Transparency","location":"Montreal QC Canada","acronym":"FAccT '26","sponsor":["ACM\/SIG"]},"container-title":["Proceedings of the 2026 ACM Conference on Fairness, Accountability, and Transparency"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805689.3806459","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T17:57:01Z","timestamp":1782755821000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805689.3806459"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,25]]},"references-count":107,"alternative-id":["10.1145\/3805689.3806459","10.1145\/3805689"],"URL":"https:\/\/doi.org\/10.1145\/3805689.3806459","relation":{},"subject":[],"published":{"date-parts":[[2026,6,25]]},"assertion":[{"value":"2026-06-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}