{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T06:34:47Z","timestamp":1782801287232,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","funder":[{"name":"Notre Dame\u2013IBM Technology Ethics Lab"},{"name":"IBM PhD Fellowship Award"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,23]]},"DOI":"10.1145\/3729176.3729199","type":"proceedings-article","created":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T11:30:16Z","timestamp":1750505416000},"page":"1-18","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":13,"title":["MetricMate: An Interactive Tool for Generating Evaluation Criteria for LLM-as-a-Judge Workflow"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1772-6065","authenticated-orcid":false,"given":"Simret Araya","family":"Gebreegziabher","sequence":"first","affiliation":[{"name":"University of Notre Dame, Notre Dame, Indiana, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6079-4355","authenticated-orcid":false,"given":"Charles","family":"Chiang","sequence":"additional","affiliation":[{"name":"University of Notre Dame, Notre Dame, Indiana, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5366-8167","authenticated-orcid":false,"given":"Zichu","family":"Wang","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0686-7911","authenticated-orcid":false,"given":"Zahra","family":"Ashktorab","sequence":"additional","affiliation":[{"name":"IBM Research, Yorktown Heights, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8152-441X","authenticated-orcid":false,"given":"Michelle","family":"Brachman","sequence":"additional","affiliation":[{"name":"IBM Research, Cambridge, Massachusetts, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4699-5026","authenticated-orcid":false,"given":"Werner","family":"Geyer","sequence":"additional","affiliation":[{"name":"IBM Research, Cambridge, Massachusetts, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7902-7625","authenticated-orcid":false,"given":"Toby Jia-Jun","family":"Li","sequence":"additional","affiliation":[{"name":"University of Notre Dame, Notre Dame, Indiana, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4609-6293","authenticated-orcid":false,"given":"Diego","family":"G\u00f3mez-Zar\u00e1","sequence":"additional","affiliation":[{"name":"University of Notre Dame, Notre Dame, Indiana, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,22]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","first-page":"1742","DOI":"10.1109\/ASE56229.2023.00065","volume-title":"2023 38th IEEE\/ACM International Conference on Automated Software Engineering (ASE)","author":"Ahmed Toufique","year":"2023","unstructured":"Toufique Ahmed and Premkumar Devanbu. 2023. Better patching using llm prompting, via self-consistency. In 2023 38th IEEE\/ACM International Conference on Automated Software Engineering (ASE). IEEE, 1742\u20131746."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642016"},{"key":"e_1_3_3_2_4_2","first-page":"1","volume-title":"2024 International Joint Conference on Neural Networks (IJCNN)","author":"Bahrami Mehdi","year":"2024","unstructured":"Mehdi Bahrami, Ryosuke Sonoda, and Ramya Srinivasan. 2024. LLM Diagnostic Toolkit: Evaluating LLMs for Ethical Issues. In 2024 International Joint Conference on Neural Networks (IJCNN). IEEE, 1\u20138."},{"key":"e_1_3_3_2_5_2","unstructured":"Ga\u0161per Begu\u0161 Maksymilian D\u0105bkowski and Ryan Rhodes. 2023. Large linguistic models: Analyzing theoretical linguistic abilities of LLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.00948 (2023)."},{"key":"e_1_3_3_2_6_2","volume-title":"The rapid adoption of generative ai","author":"Bick Alexander","year":"2024","unstructured":"Alexander Bick, Adam Blandin, and David\u00a0J Deming. 2024. The rapid adoption of generative ai. Technical Report. National Bureau of Economic Research."},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3650841"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"crossref","unstructured":"Virginia Braun and Victoria Clarke. 2006. Using thematic analysis in psychology. Qualitative research in psychology 3 2 (2006) 77\u2013101.","DOI":"10.1191\/1478088706qp063oa"},{"key":"e_1_3_3_2_9_2","unstructured":"Nuo Chen Quanyu Dai Xiaoyu Dong Xiao-Ming Wu and Zhenhua Dong. 2025. Evaluating Conversational Recommender Systems with Large Language Models: A User-Centric Evaluation Framework. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.09493 (2025)."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Jacob Cohen. 1960. A coefficient of agreement for nominal scales. Educational and psychological measurement 20 1 (1960) 37\u201346.","DOI":"10.1177\/001316446002000104"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"crossref","unstructured":"Lacey Colligan Henry\u00a0WW Potts Chelsea\u00a0T Finn and Robert\u00a0A Sinkin. 2015. Cognitive workload changes for nurses transitioning from a legacy system with paper documentation to a commercial electronic health record. International journal of medical informatics 84 7 (2015) 469\u2013476.","DOI":"10.1016\/j.ijmedinf.2015.03.003"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.5555\/534996"},{"key":"e_1_3_3_2_13_2","unstructured":"Teresa Datta and John\u00a0P Dickerson. 2023. Who\u2019s Thinking? A Push for Human-Centered Evaluation of LLMs using the XAI Playbook. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.06223 (2023)."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","first-page":"461","DOI":"10.1109\/AUTEST.2006.283707","volume-title":"2006 IEEE Autotestcon","author":"Delgado Santiago","year":"2006","unstructured":"Santiago Delgado. 2006. Designing Modular Software Architectures for Next-Generation Heterogeneous Networked Test Systems. In 2006 IEEE Autotestcon. IEEE, 461\u2013466."},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"crossref","unstructured":"Brenda Dervin. 1998. Sense-making theory and practice: An overview of user interests in knowledge seeking and use. Journal of knowledge management 2 2 (1998) 36\u201346.","DOI":"10.1108\/13673279810249369"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3640544.3645216"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3586182.3615810"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"John\u00a0J Dudley and Per\u00a0Ola Kristensson. 2018. A review of user interface design for interactive machine learning. ACM Transactions on Interactive Intelligent Systems (TiiS) 8 2 (2018) 1\u201337.","DOI":"10.1145\/3185517"},{"key":"e_1_3_3_2_19_2","unstructured":"Upol Ehsan and Mark\u00a0O Riedl. 2019. On design and evaluation of human-centered explainable AI systems. Glasgow\u201919 (2019)."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.18653\/v1\/2024.acl-long.63","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","author":"Elangovan Aparna","year":"2024","unstructured":"Aparna Elangovan, Ling Liu, Lei Xu, Sravan\u00a0Babu Bodapati, and Dan Roth. 2024. ConSiDERS-The-Human Evaluation Framework: Rethinking Human Evaluation for Generative Large Language Models. In Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Lun-Wei Ku, Andre Martins, and Vivek Srikumar (Eds.). Association for Computational Linguistics, Bangkok, Thailand, 1137\u20131160. https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.63"},{"key":"e_1_3_3_2_21_2","first-page":"14778","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","author":"Fan Zhiting","year":"2024","unstructured":"Zhiting Fan, Ruizhe Chen, Ruiling Xu, and Zuozhu Liu. 2024. BiasAlert: A Plug-and-play Tool for Social Bias Detection in LLMs. In Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing. 14778\u201314790."},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Faiha Fareez Tishya Parikh Christopher Wavell Saba Shahab Meghan Chevalier Scott Good Isabella De\u00a0Blasi Rafik Rhouma Christopher McMahon Jean-Paul Lam et\u00a0al. 2022. A dataset of simulated patient-physician medical interviews with a focus on respiratory cases. Scientific Data 9 1 (2022) 313.","DOI":"10.1038\/s41597-022-01423-1"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3616855.3635739"},{"key":"e_1_3_3_2_24_2","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et\u00a0al. 2024. A Survey on LLM-as-a-Judge. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.15594 (2024)."},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3663384.3663398"},{"key":"e_1_3_3_2_26_2","volume-title":"Contextual design: defining customer-centered systems","author":"Holtzblatt Karen","year":"1997","unstructured":"Karen Holtzblatt and Hugh Beyer. 1997. Contextual design: defining customer-centered systems. Elsevier."},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"crossref","unstructured":"Fred Jelinek Robert\u00a0L Mercer Lalit\u00a0R Bahl and James\u00a0K Baker. 1977. Perplexity\u2014a measure of the difficulty of speech recognition tasks. The Journal of the Acoustical Society of America 62 S1 (1977) S63\u2013S63.","DOI":"10.1121\/1.2016299"},{"key":"e_1_3_3_2_28_2","first-page":"7367","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics","author":"Jung Sangkeun","year":"2025","unstructured":"Sangkeun Jung and Jeesu Jung. 2025. Courtroom-LLM: A Legal-Inspired Multi-LLM Framework for Resolving Ambiguous Text Classifications. In Proceedings of the 31st International Conference on Computational Linguistics. 7367\u20137385."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642216"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3641913"},{"key":"e_1_3_3_2_31_2","unstructured":"Dawei Li Bohan Jiang Liangjie Huang Alimohammad Beigi Chengshuai Zhao Zhen Tan Amrita Bhattacharjee Yuxuan Jiang Canyu Chen Tianhao Wu et\u00a0al. 2024. From generation to judgment: Opportunities and challenges of llm-as-a-judge. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.16594 (2024)."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3604237.3626869"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"crossref","unstructured":"Q\u00a0Vera Liao and Jennifer\u00a0Wortman Vaughan. 2024. AI Transparency in the Age of LLMs: A Human-Centered Research Roadmap. Harvard Data Science ReviewSpecial Issue 5 (2024).","DOI":"10.1162\/99608f92.8036d03b"},{"key":"e_1_3_3_2_34_2","first-page":"74","volume-title":"Text summarization branches out","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74\u201381."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3501825"},{"key":"e_1_3_3_2_36_2","first-page":"79","volume-title":"Advances in computers","author":"Masri Wes","year":"2016","unstructured":"Wes Masri and Fadi\u00a0A Zaraket. 2016. Coverage-based software testing: Beyond basic test requirements. In Advances in computers. Vol.\u00a0103. Elsevier, 79\u2013142."},{"key":"e_1_3_3_2_37_2","volume-title":"Software testing and quality assurance: theory and practice","author":"Naik Kshirasagar","year":"2011","unstructured":"Kshirasagar Naik and Priyadarshi Tripathy. 2011. Software testing and quality assurance: theory and practice. John Wiley & Sons."},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"crossref","unstructured":"Shakked Noy and Whitney Zhang. 2023. Experimental evidence on the productivity effects of generative artificial intelligence. Science 381 6654 (2023) 187\u2013192.","DOI":"10.1126\/science.adh2586"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"Lawrence\u00a0A Palinkas Sarah\u00a0M Horwitz Carla\u00a0A Green Jennifer\u00a0P Wisdom Naihua Duan and Kimberly Hoagwood. 2015. Purposeful sampling for qualitative data collection and analysis in mixed method implementation research. Administration and policy in mental health and mental health services research 42 (2015) 533\u2013544.","DOI":"10.1007\/s10488-013-0528-y"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.hucllm-1.2"},{"key":"e_1_3_3_2_41_2","first-page":"311","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318."},{"key":"e_1_3_3_2_42_2","unstructured":"Ji-Lun Peng Sijia Cheng Egil Diau Yung-Yu Shih Po-Heng Chen Yen-Ting Lin and Yun-Nung Chen. 2024. A Survey of Useful LLM Evaluation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.00936 (2024)."},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"crossref","unstructured":"Jianing Qiu Kyle Lam Guohao Li Amish Acharya Tien\u00a0Yin Wong Ara Darzi Wu Yuan and Eric\u00a0J Topol. 2024. LLM-based agentic systems in medicine and healthcare. Nature Machine Intelligence 6 12 (2024) 1418\u20131420.","DOI":"10.1038\/s42256-024-00944-1"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939778"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676450"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"crossref","unstructured":"Samuel\u00a0Sanford Shapiro and Martin\u00a0B Wilk. 1965. An analysis of variance test for normality (complete samples). Biometrika 52 3-4 (1965) 591\u2013611.","DOI":"10.1093\/biomet\/52.3-4.591"},{"key":"e_1_3_3_2_47_2","unstructured":"Hua Shen Tiffany Knearem Reshmi Ghosh Kenan Alkiek Kundan Krishna Yachuan Liu Ziqiao Ma Savvas Petridis Yi-Hao Peng Li Qiwei et\u00a0al. 2024. Towards Bidirectional Human-AI Alignment: A Systematic Review for Clarifications Framework and Future Directions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.09264 (2024)."},{"key":"e_1_3_3_2_48_2","unstructured":"Aditya Singh Dibyendu Mishra Sanchit Bansal Vinayak Agarwal Anjali Goyal and Ashish Sureka. 2018. Email Dataset for Automatic Response Suggestion within a University. (2018)."},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"crossref","unstructured":"Student. 1908. The probable error of a mean. Biometrika (1908) 1\u201325.","DOI":"10.2307\/2331554"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642400"},{"key":"e_1_3_3_2_51_2","unstructured":"Annalisa Szymanski Simret\u00a0Araya Gebreegziabher Oghenemaro Anuyah Ronald\u00a0A Metoyer and Toby Jia-Jun Li. 2024. Comparing Criteria Development Across Domain Experts Lay Users and Models in Large Language Model Evaluation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.02054 (2024)."},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3708359.3712091"},{"key":"e_1_3_3_2_53_2","unstructured":"Yufei Wang Wanjun Zhong Liangyou Li Fei Mi Xingshan Zeng Wenyong Huang Lifeng Shang Xin Jiang and Qun Liu. 2023. Aligning large language models with human: A survey. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.12966 (2023)."},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/3531146.3533088"},{"key":"e_1_3_3_2_55_2","unstructured":"Michael Williams and Tami Moser. 2019. The art of coding and thematic exploration in qualitative research. International management review 15 1 (2019) 45\u201355."},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517582"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3636302"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1145\/3531146.3533118"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581388"},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1145\/2786805.2786858"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.830"},{"key":"e_1_3_3_2_62_2","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric Xing et\u00a0al. 2024. Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_2_63_2","doi-asserted-by":"publisher","DOI":"10.1145\/1240624.1240704"}],"event":{"name":"CHIWORK '25: Proceedings of the 4th Annual Symposium on Human-Computer Interaction for Work","location":"Amsterdam Netherlands","acronym":"CHIWORK '25"},"container-title":["Proceedings of the 4th Annual Symposium on Human-Computer Interaction for Work"],"original-title":[],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T11:33:08Z","timestamp":1750505588000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3729176.3729199"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,22]]},"references-count":62,"alternative-id":["10.1145\/3729176.3729199","10.1145\/3729176"],"URL":"https:\/\/doi.org\/10.1145\/3729176.3729199","relation":{},"subject":[],"published":{"date-parts":[[2025,6,22]]},"assertion":[{"value":"2025-06-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}