{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T10:50:58Z","timestamp":1784371858620,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":71,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,27]],"date-time":"2024-10-27T00:00:00Z","timestamp":1729987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000104","name":"National Aeronautics and Space Administration","doi-asserted-by":"publisher","award":["80NSSC23M0058"],"award-info":[{"award-number":["80NSSC23M0058"]}],"id":[{"id":"10.13039\/100000104","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,27]]},"DOI":"10.1145\/3691620.3695296","type":"proceedings-article","created":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T15:39:19Z","timestamp":1729265959000},"page":"2262-2267","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["CoDefeater: Using LLMs To Find Defeaters in Assurance Cases"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8859-4928","authenticated-orcid":false,"given":"Usman","family":"Gohar","sequence":"first","affiliation":[{"name":"Iowa State University, Ames, Iowa, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3775-7922","authenticated-orcid":false,"given":"Michael C.","family":"Hunter","sequence":"additional","affiliation":[{"name":"Iowa State University, Ames, Iowa, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5390-7982","authenticated-orcid":false,"given":"Robyn R.","family":"Lutz","sequence":"additional","affiliation":[{"name":"Iowa State University, Ames, Iowa, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2443-2425","authenticated-orcid":false,"given":"Myra B.","family":"Cohen","sequence":"additional","affiliation":[{"name":"Iowa State University, Ames, Iowa, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Adelard. 2024. https:\/\/www.adelard.com\/asce\/cae\/"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3551349.3559555"},{"key":"e_1_3_2_1_3_1","unstructured":"Golden Gate Bird Alliance. 2024. https:\/\/goldengatebirdalliance.org\/conservation\/make-the-city-safe-for-wildlife\/drone-dangers-and-birds\/"},{"key":"e_1_3_2_1_4_1","volume-title":"Generative AI for Effective Software Development","author":"Arora Chetan","unstructured":"Chetan Arora, John Grundy, and Mohamed Abdelrazek. 2024. Advancing requirements engineering through generative ai: Assessing the role of llms. In Generative AI for Effective Software Development. Springer, 129--148."},{"key":"e_1_3_2_1_6_1","volume-title":"Safety case templates for autonomous systems. arXiv preprint arXiv:2102.02625","author":"Bloomfield Robin","year":"2021","unstructured":"Robin Bloomfield, Gareth Fletcher, Heidy Khlaaf, Luke Hinde, and Philippa Ryan. 2021. Safety case templates for autonomous systems. arXiv preprint arXiv:2102.02625 (2021)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Robin Bloomfield and John Rushby. 2024. Confidence in Assurance 2.0 Cases. In The Practice of Formal Methods: Essays in Honour of Cliff Jones Part I. Springer 1--23.","DOI":"10.1007\/978-3-031-66676-6_1"},{"key":"e_1_3_2_1_8_1","volume-title":"Using thematic analysis in psychology. Qualitative research in psychology 3, 2","author":"Braun Virginia","year":"2006","unstructured":"Virginia Braun and Victoria Clarke. 2006. Using thematic analysis in psychology. Qualitative research in psychology 3, 2 (2006), 77--101."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/RE.2006.68"},{"key":"e_1_3_2_1_10_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/REW57809.2023.00052"},{"key":"e_1_3_2_1_13_1","volume-title":"Unleashing the potential of prompt engineering in large language models: a comprehensive review. arXiv preprint arXiv:2310.14735","author":"Chen Banghao","year":"2023","unstructured":"Banghao Chen, Zhaofeng Zhang, Nicolas Langren\u00e9, and Shengxin Zhu. 2023. Unleashing the potential of prompt engineering in large language models: a comprehensive review. arXiv preprint arXiv:2310.14735 (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"How is ChatGPT's behavior changing over time? arXiv preprint arXiv:2307.09009","author":"Chen Lingjiao","year":"2023","unstructured":"Lingjiao Chen, Matei Zaharia, and James Zou. 2023. How is ChatGPT's behavior changing over time? arXiv preprint arXiv:2307.09009 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)."},{"key":"e_1_3_2_1_16_1","volume-title":"Can large language models be an alternative to human evaluations? arXiv preprint arXiv:2305.01937","author":"Chiang Cheng-Han","year":"2023","unstructured":"Cheng-Han Chiang and Hung-yi Lee. 2023. Can large language models be an alternative to human evaluations? arXiv preprint arXiv:2305.01937 (2023)."},{"key":"e_1_3_2_1_17_1","volume-title":"A coefficient of agreement for nominal scales. Educational and psychological measurement 20, 1","author":"Cohen Jacob","year":"1960","unstructured":"Jacob Cohen. 1960. A coefficient of agreement for nominal scales. Educational and psychological measurement 20, 1 (1960), 37--46."},{"key":"e_1_3_2_1_18_1","volume-title":"LLM-in-the-loop: Leveraging large language model for thematic analysis. arXiv preprint arXiv:2310.15100","author":"Dai Shih-Chieh","year":"2023","unstructured":"Shih-Chieh Dai, Aiping Xiong, and Lun-Wei Ku. 2023. LLM-in-the-loop: Leveraging large language model for thematic analysis. arXiv preprint arXiv:2310.15100 (2023)."},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings 31","author":"Denney Ewen","year":"2012","unstructured":"Ewen Denney, Ganesh Pai, and Josef Pohl. 2012. AdvoCATE: An assurance case automation toolset. In Computer Safety, Reliability, and Security: SAFECOMP 2012 Workshops: Sassur, ASCoMS, DESEC4LCCI, ERCIM\/EWICS, IWDE, Proceedings 31. Springer, 8--21."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10916-018-0921-x"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.56094\/jss.v58i1.215"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-40953-0_35"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-63194-3_5"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-FoSE59343.2023.00008"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Meric Altug Gemalmaz and Ming Yin. 2021. Accounting for Confirmation Bias in Crowdsourced Label Aggregation.. In IJCAI. 1729--1735.","DOI":"10.24963\/ijcai.2021\/238"},{"key":"e_1_3_2_1_26_1","volume-title":"Cohen","author":"Gohar Usman","year":"2024","unstructured":"Usman Gohar, Michael C. Hunter, Robyn R. Lutz, and Myra B. Cohen. 2024. Supplemental Data. https:\/\/github.com\/UsmanGohar\/CoDefeater"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3639475.3640103"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.21236\/ADA609836"},{"key":"e_1_3_2_1_29_1","volume-title":"Eliminative argumentation: A basis for arguing confidence in system properties","author":"Goodenough John B","year":"2015","unstructured":"John B Goodenough, Charles B Weinstock, and Ari Z Klein. 2015. Eliminative argumentation: A basis for arguing confidence in system properties. Software Engineering Institute, Carnegie Mellon University (2015)."},{"key":"e_1_3_2_1_30_1","volume-title":"An investigation of proposed techniques for quantifying confidence in assurance arguments. Safety science 92","author":"Graydon Patrick J","year":"2017","unstructured":"Patrick J Graydon and C Michael Holloway. 2017. An investigation of proposed techniques for quantifying confidence in assurance arguments. Safety science 92 (2017), 53--65."},{"key":"e_1_3_2_1_31_1","volume-title":"24th International System Safety Conference.","author":"Greenwell William S","year":"2006","unstructured":"William S Greenwell, John C Knight, C Michael Holloway, and Jacob J Pease. 2006. A taxonomy of fallacies in system safety arguments. In 24th International System Safety Conference."},{"key":"e_1_3_2_1_32_1","volume-title":"Safety and Systems Development: IFIP 18th World Computer Congress TC13\/WC13. 5 7th Working Conference on Human Error, Safety and Systems Development 22--27","author":"Greenwell William S","year":"2004","unstructured":"William S Greenwell, Elisabeth A Strunk, and John C Knight. 2004. Failure analysis and the safety-case lifecycle. In Human Error, Safety and Systems Development: IFIP 18th World Computer Congress TC13\/WC13. 5 7th Working Conference on Human Error, Safety and Systems Development 22--27. Springer, 163--176."},{"key":"e_1_3_2_1_33_1","first-page":"4242","article-title":"A survey on concepts, applications, and challenges in cyber-physical systems","volume":"8","author":"Gunes Volkan","year":"2014","unstructured":"Volkan Gunes, Steffen Peter, Tony Givargis, and Frank Vahid. 2014. A survey on concepts, applications, and challenges in cyber-physical systems. KSII Transactions on Internet and Information Systems (TIIS) 8, 12 (2014), 4242--4268.","journal-title":"KSII Transactions on Internet and Information Systems (TIIS)"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-85729-133-2_1"},{"key":"e_1_3_2_1_35_1","volume-title":"Large Language Models for Software Engineering: A Systematic Literature Review. ArXiv abs\/2308.10620","author":"Hou Xinying","year":"2023","unstructured":"Xinying Hou, Yanjie Zhao, Yue Liu, Zhou Yang, Kailong Wang, Li Li, Xiapu Luo, David Lo, John C. Grundy, and Haoyu Wang. 2023. Large Language Models for Software Engineering: A Systematic Literature Review. ArXiv abs\/2308.10620 (2023)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.2514\/6.2024-4626"},{"key":"e_1_3_2_1_37_1","first-page":"11","article-title":"DO-178B: Software considerations in airborne systems and equipment certification","volume":"199","author":"Johnson Leslie A","year":"1998","unstructured":"Leslie A Johnson et al. 1998. DO-178B: Software considerations in airborne systems and equipment certification. Crosstalk 199 (1998), 11--20.","journal-title":"Crosstalk"},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the dependable systems and networks 2004 workshop on assurance cases","volume":"6","author":"Kelly Tim","year":"2004","unstructured":"Tim Kelly and Rob Weaver. 2004. The goal structuring notation-a safety argument notation. In Proceedings of the dependable systems and networks 2004 workshop on assurance cases, Vol. 6. Citeseer."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3650105.3652291"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/581339.581406"},{"key":"e_1_3_2_1_41_1","volume-title":"Machel Reid, Yutaka Matsuo, and Yusuke Iwasawa.","author":"Kojima Takeshi","year":"2022","unstructured":"Takeshi Kojima, Shixiang Shane Gu, Machel Reid, Yutaka Matsuo, and Yusuke Iwasawa. 2022. Large language models are zero-shot reasoners. Advances in neural information processing systems 35 (2022), 22199--22213."},{"key":"e_1_3_2_1_42_1","volume-title":"Better zero-shot reasoning with role-play prompting. arXiv preprint arXiv:2308.07702","author":"Kong Aobo","year":"2023","unstructured":"Aobo Kong, Shiwan Zhao, Hao Chen, Qicheng Li, Yong Qin, Ruiqi Sun, and Xin Zhou. 2023. Better zero-shot reasoning with role-play prompting. arXiv preprint arXiv:2308.07702 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"Stanford Encyclopedia of Philosophy.","author":"Koons Robert C.","unstructured":"Robert C. Koons. 2008. Defeasible Reasoning. In Stanford Encyclopedia of Philosophy."},{"key":"e_1_3_2_1_44_1","volume-title":"How Safe Is Safe Enough: Measuring and Predicting Autonomous Vehicle Safety","author":"Koopman Philip","unstructured":"Philip Koopman. 2022. How Safe Is Safe Enough: Measuring and Predicting Autonomous Vehicle Safety. Carnegie Mellon University."},{"key":"e_1_3_2_1_45_1","unstructured":"Udayangani Kulatunga Dilanthi Amaratunga and Richard Haigh. 2007. Structuring the unstructured data: the use of content analysis. (2007)."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings 37","author":"Maksimov Mike","year":"2018","unstructured":"Mike Maksimov, Nick LS Fung, Sahar Kokaly, and Marsha Chechik. 2018. Two decades of assurance case tools: a survey. In Computer Safety, Reliability, and Security: SAFECOMP 2018 Workshops, ASSURE, DECSoS, SASSUR, STRIVE, and WAISE, Proceedings 37. Springer, 49--59."},{"key":"e_1_3_2_1_47_1","volume-title":"Article 101","author":"Maksimov Mike","year":"2019","unstructured":"Mike Maksimov, Sahar Kokaly, and Marsha Chechik. 2019. A Survey of Tool-supported Assurance Case Assessment Techniques. ACM Comput. Surv. 52, 5, Article 101 (2019), 34 pages."},{"key":"e_1_3_2_1_48_1","volume-title":"Interrater reliability: the kappa statistic. Biochemia medica 22, 3","author":"McHugh Mary L","year":"2012","unstructured":"Mary L McHugh. 2012. Interrater reliability: the kappa statistic. Biochemia medica 22, 3 (2012), 276--282."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-NIER58687.2023.00030"},{"key":"e_1_3_2_1_50_1","volume-title":"Assurance Case Arguments in the Large: The CERN LHC Machine Protection System. In International Conference on Computer Safety, Reliability, and Security. Springer, 3--10","author":"Millet Laure","year":"2023","unstructured":"Laure Millet, Simon Diemert, Chris Rees, Torin Viger, Marsha Chechik, Claudio Menghi, and Jeffrey Joyce. 2023. Assurance Case Arguments in the Large: The CERN LHC Machine Protection System. In International Conference on Computer Safety, Reliability, and Security. Springer, 3--10."},{"key":"e_1_3_2_1_51_1","volume-title":"Proceedings of the International Conference on Logic Programming 2023 Workshops co-located with the 39th International Conference on Logic Programming (ICLP","author":"Murugesan Anitha","year":"2023","unstructured":"Anitha Murugesan, Isaac Hong Wong, Robert J. Stroud, Joaqu\u00edn Arias, Elmer Salazar, Gopal Gupta, Robin Bloomfield, Srivatsan Varadarajan, and John Rushby. [n. d.]. Semantic Analysis of Assurance Cases using s(CASP). In Proceedings of the International Conference on Logic Programming 2023 Workshops co-located with the 39th International Conference on Logic Programming (ICLP 2023)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.infsof.2014.03.001"},{"key":"e_1_3_2_1_53_1","unstructured":"OpenAI. 2024. Retrieved 2024 from https:\/\/platform.openai.com\/docs\/guides\/prompt-engineering"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Rob Palin David Ward Ibrahim Habli and Roger Rivett. 2011. ISO 26262 safety cases: Compliance and assurance. (2011).","DOI":"10.1049\/cp.2011.0251"},{"key":"e_1_3_2_1_55_1","volume-title":"General data protection regulation. official Journal of the European Union 59","author":"European Parliament and E Council","year":"2016","unstructured":"European Parliament and E Council. 2016. General data protection regulation. official Journal of the European Union 59 (2016), 294."},{"key":"e_1_3_2_1_56_1","unstructured":"Baolin Peng Michel Galley Pengcheng He Hao Cheng Yujia Xie Yu Hu Qiuyuan Huang Lars Liden Zhou Yu Weizhu Chen et al. 2023. Check your facts and try again: Improving large language models with external knowledge and automated feedback. arXiv preprint arXiv:2302.12813 (2023)."},{"key":"e_1_3_2_1_57_1","volume-title":"Defeasible reasoning. Cognitive science 11, 4","author":"Pollock John L","year":"1987","unstructured":"John L Pollock. 1987. Defeasible reasoning. Cognitive science 11, 4 (1987), 481--518."},{"key":"e_1_3_2_1_58_1","volume-title":"Reliability of safety-critical systems: theory and applications","author":"Rausand Marvin","unstructured":"Marvin Rausand. 2014. Reliability of safety-critical systems: theory and applications. John Wiley & Sons."},{"key":"e_1_3_2_1_59_1","unstructured":"Chris Rees Rolf Lippelt Marsha Chechik Lukas Felsberger Mateo Delgado Markus Zerlauth Jeff Joyce Claudio Menghi Jan Uythoven Simon Diemert and et al. 2023. Assessing the usefulness of assurance cases: An experience with the CERN Large Hadron Collider. http:\/\/cds.cern.ch\/record\/2854725"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1016\/0951-8320(94)90065-5"},{"key":"e_1_3_2_1_62_1","volume-title":"A PRISMA-driven systematic mapping study on system assurance weakeners. Information and Software Technology","author":"Shahandashti Kimya Khakzad","year":"2024","unstructured":"Kimya Khakzad Shahandashti, Alvine B Belle, Timothy C Lethbridge, Oluwafemi Odu, and Mithila Sivakumar. 2024. A PRISMA-driven systematic mapping study on system assurance weakeners. Information and Software Technology (2024), 107526."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/REW61692.2024.00011"},{"key":"e_1_3_2_1_64_1","volume-title":"Jinjun Shan, and Kimya Khakzad Shahandashti.","author":"Sivakumar Mithila","year":"2023","unstructured":"Mithila Sivakumar, Alvine Boaye Belle, Jinjun Shan, and Kimya Khakzad Shahandashti. 2023. GPT-4 and Safety Case Generation: An Exploratory Analysis. arXiv preprint arXiv:2312.05696 (2023)."},{"key":"e_1_3_2_1_65_1","volume-title":"Preventing Omission of Key Evidence Fallacy in Process-Based Argumentations. In 2018 11th International Conference on the Quality of Information and Communications Technology (QUATIC). 65--73","author":"Muram Faiz UL","year":"2018","unstructured":"Faiz UL Muram, Barbara Gallina, and Laura G\u00f3mez Rodr\u00edguez. 2018. Preventing Omission of Key Evidence Fallacy in Process-Based Argumentations. In 2018 11th International Conference on the Quality of Information and Communications Technology (QUATIC). 65--73."},{"key":"e_1_3_2_1_66_1","volume-title":"Tools & Automation for Assurance Cases. In 2023 IEEE\/AIAA 42nd Digital Avionics Systems Conference (DASC). 1--10","author":"Varadarajan Srivatsan","year":"2023","unstructured":"Srivatsan Varadarajan, Robin Bloomfield, John Rushby, Gopal Gupta, Anitha Murugesan, Robert Stroud, Kateryna Netkachova, and Isaac Hong Wong. 2023. CLARISSA: Foundations, Tools & Automation for Assurance Cases. In 2023 IEEE\/AIAA 42nd Digital Avionics Systems Conference (DASC). 1--10."},{"key":"e_1_3_2_1_67_1","volume-title":"Supporting Assurance Case Development Using Generative AI. In SAFECOMP","author":"Viger Torin","year":"2023","unstructured":"Torin Viger, Logan Murphy, Simon Diemert, Claudio Menghi, Alessio Di, and Marsha Chechik. 2023. Supporting Assurance Case Development Using Generative AI. In SAFECOMP 2023, Position Paper."},{"key":"e_1_3_2_1_68_1","volume-title":"2024 International Symposium on Software Reliability Engineering ISSRE. IEEE.","author":"Viger Torin","year":"2024","unstructured":"Torin Viger, Logan Murphy, Simon Diemert, Claudio Menghi, Jeffrey Joyce, Alessio Di Sandro, and Marsha Chechik. 2024. AI-Supported Eliminative Argumentation: Practical Experience Generating Defeaters to Increase Confidence in Assurance Cases. In 2024 International Symposium on Software Reliability Engineering ISSRE. IEEE."},{"key":"e_1_3_2_1_69_1","volume-title":"Exploring the Reasoning Abilities of Multimodal Large Language Models (MLLMs): A Comprehensive Survey on Emerging Trends in Multimodal Reasoning. ArXiv abs\/2401.06805","author":"Wang Yiqi","year":"2024","unstructured":"Yiqi Wang, Wentao Chen, Xiaotian Han, Xudong Lin, Haiteng Zhao, Yongfei Liu, Bohan Zhai, Jianbo Yuan, Quanzeng You, and Hongxia Yang. 2024. Exploring the Reasoning Abilities of Multimodal Large Language Models (MLLMs): A Comprehensive Survey on Emerging Trends in Multimodal Reasoning. ArXiv abs\/2401.06805 (2024)."},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSREW.2019.00091"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3639478.3643065"},{"key":"e_1_3_2_1_72_1","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric Xing et al. 2024. Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_73_1","volume-title":"Judging LLM-as-a-judge with MT-Bench and Chatbot Arena. ArXiv abs\/2306.05685","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric P. Xing, Haotong Zhang, Joseph Gonzalez, and Ion Stoica. 2023. Judging LLM-as-a-judge with MT-Bench and Chatbot Arena. ArXiv abs\/2306.05685 (2023)."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1145\/3639476.3639762"}],"event":{"name":"ASE '24: 39th IEEE\/ACM International Conference on Automated Software Engineering","location":"Sacramento CA USA","acronym":"ASE '24","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGSOFT ACM Special Interest Group on Software Engineering","IEEE CS"]},"container-title":["Proceedings of the 39th IEEE\/ACM International Conference on Automated Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3691620.3695296","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/abs\/10.1145\/3691620.3695296","content-type":"text\/html","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3691620.3695296","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:04:07Z","timestamp":1750291447000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3691620.3695296"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,27]]},"references-count":71,"alternative-id":["10.1145\/3691620.3695296","10.1145\/3691620"],"URL":"https:\/\/doi.org\/10.1145\/3691620.3695296","relation":{},"subject":[],"published":{"date-parts":[[2024,10,27]]},"assertion":[{"value":"2024-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}