{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T09:15:45Z","timestamp":1783242945518,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3805760.3814892","type":"proceedings-article","created":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:46:13Z","timestamp":1783241173000},"page":"51-60","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Using Mutation-Analysis to Examine an LLM's Ability to Summarize Code"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4770-8900","authenticated-orcid":false,"given":"Lara","family":"Khatib","sequence":"first","affiliation":[{"name":"University of Waterloo, Waterloo, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3374-8957","authenticated-orcid":false,"given":"Michael","family":"Pu","sequence":"additional","affiliation":[{"name":"University of Waterloo, Waterloo, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4418-5783","authenticated-orcid":false,"given":"Bogdan","family":"Vasilescu","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4533-4728","authenticated-orcid":false,"given":"Meiyappan","family":"Nagappan","sequence":"additional","affiliation":[{"name":"University of Waterloo, Waterloo, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3639183"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2006.83"},{"key":"e_1_3_2_1_3_1","unstructured":"Anthropic. 2025. Tracing the thoughts of a large language model. https:\/\/www. anthropic.com\/research\/tracing-thoughts-language-model. Accessed: 2025-05- 22."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72."},{"key":"e_1_3_2_1_5_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","first-page":"81","DOI":"10.1007\/s10664-023-10311-0","article-title":"Seeing confusion through a new lens: on the impact of atoms of confusion on novices' code comprehension","volume":"28","author":"Silva da Costa Jos\u00e9 Aldo","year":"2023","unstructured":"Jos\u00e9 Aldo Silva da Costa, Rohit Gheyi, Fernando Castor, Pablo Roberto Fernandes de Oliveira, M\u00e1rcio Ribeiro, and Baldoino Fonseca. 2023. Seeing confusion through a new lens: on the impact of atoms of confusion on novices' code comprehension. Empirical Software Engineering 28, 4 (2023), 81.","journal-title":"Empirical Software Engineering"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the XXXIV Brazilian Symposium on Software Engineering. 243-252","author":"Oliveira Benedito De","year":"2020","unstructured":"Benedito De Oliveira, M\u00e1rcio Ribeiro, Jos\u00e9 Aldo Silva Da Costa, Rohit Gheyi, Guilherme Amaral, Rafael de Mello, Anderson Oliveira, Alessandro Garcia, Rodrigo Bonif\u00e1cio, and Baldoino Fonseca. 2020. Atoms of confusion: The eyes do not lie. In Proceedings of the XXXIV Brazilian Symposium on Software Engineering. 243-252."},{"key":"e_1_3_2_1_8_1","unstructured":"GitHub. 2021. GitHub Copilot. https:\/\/github.com\/features\/copilot. Accessed: 2026-05-07."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3106237.3106264"},{"key":"e_1_3_2_1_10_1","volume-title":"Ahmad Humayun, Waris Gill, Abdul Haddi Amjad, Ali R Butt, Mohammad Taha Khan, and Muhammad Ali Gulzar.","author":"Haroon Sabaat","year":"2025","unstructured":"Sabaat Haroon, Ahmad Faraz Khan, Ahmad Humayun, Waris Gill, Abdul Haddi Amjad, Ali R Butt, Mohammad Taha Khan, and Muhammad Ali Gulzar. 2025. How Accurately Do Large Language Models Understand Code? arXiv preprint arXiv:2504.04372 (2025)."},{"key":"e_1_3_2_1_11_1","volume-title":"Artificial Intelligence -The State of Developer Ecosystem","year":"2023","unstructured":"JetBrains. 2023. Artificial Intelligence -The State of Developer Ecosystem in 2023. https:\/\/www.jetbrains.com\/lp\/devecosystem-2023\/ai\/."},{"key":"e_1_3_2_1_12_1","unstructured":"Grzegorz Kocur. 2015. MutPy: A Mutation Testing Tool for Python. https: \/\/github.com\/mutpy\/mutpy. Accessed: 2026-04-07."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3587102.3588785"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the IEEE\/ACM 3rd International Conference on AI Engineering-Software Engineering for AI. 150- 159","author":"Li Ziyu","year":"2024","unstructured":"Ziyu Li and Donghwan Shin. 2024. Mutation-based consistency testing for evaluating the code understanding capability of llms. In Proceedings of the IEEE\/ACM 3rd International Conference on AI Engineering-Software Engineering for AI. 150- 159."},{"key":"e_1_3_2_1_15_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81."},{"key":"e_1_3_2_1_16_1","volume-title":"Rigorous Evaluation of Large Language Models for Code Generation. In Thirty-seventh Conference on Neural Information Processing Systems. https:\/\/openreview.net\/forum?id=1qvx610Cu7","author":"Liu Jiawei","year":"2023","unstructured":"Jiawei Liu, Chunqiu Steven Xia, Yuyao Wang, and Lingming Zhang. 2023. Is Your Code Generated by ChatGPT Really Correct? Rigorous Evaluation of Large Language Models for Code Generation. In Thirty-seventh Conference on Neural Information Processing Systems. https:\/\/openreview.net\/forum?id=1qvx610Cu7"},{"key":"e_1_3_2_1_17_1","volume-title":"Yuyao Wang, and Lingming Zhang.","author":"Liu Jiawei","year":"2024","unstructured":"Jiawei Liu, Chunqiu Steven Xia, Yuyao Wang, and Lingming Zhang. 2024. Is your code generated by chatgpt really correct? rigorous evaluation of large language models for code generation. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00638"},{"key":"e_1_3_2_1_19_1","volume-title":"Codexglue: A machine learning benchmark dataset for code understanding and generation. arXiv preprint arXiv:2102.04664","author":"Lu Shuai","year":"2021","unstructured":"Shuai Lu, Daya Guo, Shuo Ren, Junjie Huang, Alexey Svyatkovskiy, Ambrosio Blanco, Colin Clement, Dawn Drain, Daxin Jiang, Duyu Tang, et al. 2021. Codexglue: A machine learning benchmark dataset for code understanding and generation. arXiv preprint arXiv:2102.04664 (2021)."},{"key":"e_1_3_2_1_20_1","volume-title":"Lms: Understanding code syntax and semantics for code analysis. arXiv preprint arXiv:2305.12138","author":"Ma Wei","year":"2023","unstructured":"Wei Ma, Shangqing Liu, Zhihao Lin, Wenhan Wang, Qiang Hu, Ye Liu, Cen Zhang, Liming Nie, Li Li, and Yang Liu. 2023. Lms: Understanding code syntax and semantics for code analysis. arXiv preprint arXiv:2305.12138 (2023)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3639174"},{"key":"e_1_3_2_1_22_1","first-page":"13215","article-title":"On leakage of code generation evaluation datasets","volume":"2024","author":"Matton Alexandre","year":"2024","unstructured":"Alexandre Matton, Tom Sherborne, Dennis Aumiller, Elena Tommasone, Milad Alizadeh, Jingyi He, Raymond Ma, Maxime Voisin, Ellen Gilsenan-McMahon, and Matthias Gall\u00e9. 2024. On leakage of code generation evaluation datasets. In Findings of the Association for Computational Linguistics: EMNLP 2024. 13215- 13223.","journal-title":"Findings of the Association for Computational Linguistics: EMNLP"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPC.2015.12"},{"key":"e_1_3_2_1_24_1","volume-title":"OpenAi's GPT4 as coding assistant. arXiv preprint arXiv:2309.12732","author":"Moussiades Lefteris","year":"2023","unstructured":"Lefteris Moussiades and George Zografos. 2023. OpenAi's GPT4 as coding assistant. arXiv preprint arXiv:2309.12732 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"Swayam Singh, Xiangru Tang, Leandro Von Werra, and Shayne Longpre.","author":"Muennighoff Niklas","year":"2023","unstructured":"Niklas Muennighoff, Qian Liu, Armel Zebaze, Qinkai Zheng, Binyuan Hui, Terry Yue Zhuo, Swayam Singh, Xiangru Tang, Leandro Von Werra, and Shayne Longpre. 2023. Octopack: Instruction tuning code large language models. arXiv preprint arXiv:2308.07124 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"In-ide generation-based information support with a large language model. arXiv preprint arXiv:2307.08177","author":"Nam Daye","year":"2023","unstructured":"Daye Nam, Andrew Macvean, Vincent Hellendoorn, Bogdan Vasilescu, and Brad Myers. 2023. In-ide generation-based information support with a large language model. arXiv preprint arXiv:2307.08177 (2023)."},{"key":"e_1_3_2_1_27_1","unstructured":"OpenAI. 2022. Introducing ChatGPT. https:\/\/openai.com\/index\/chatgpt. Accessed: 2026-05-07."},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318."},{"key":"e_1_3_2_1_29_1","volume-title":"Muhammad Waseem, Kai-Kristian Kemell, Xiaofeng Wang, Anh Nguyen, Kari Syst\u00e4, and Pekka Abrahamsson.","author":"Rasheed Zeeshan","year":"2024","unstructured":"Zeeshan Rasheed, Malik Abdul Sami, Muhammad Waseem, Kai-Kristian Kemell, Xiaofeng Wang, Anh Nguyen, Kari Syst\u00e4, and Pekka Abrahamsson. 2024. AI- powered Code Review with LLMs: Early Results. arXiv preprint arXiv:2404.18496 (2024)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00322"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3468264.3468588"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3501385.3543957"},{"key":"e_1_3_2_1_33_1","unstructured":"David A Sousa. 2016. Primacy\/recency effect."},{"key":"e_1_3_2_1_34_1","volume-title":"Source code summarization in the era of large language models. arXiv preprint arXiv:2407.07959","author":"Sun Weisong","year":"2024","unstructured":"Weisong Sun, Yun Miao, Yuekang Li, Hongyu Zhang, Chunrong Fang, Yi Liu, Gelei Deng, Yang Liu, and Zhenyu Chen. 2024. Source code summarization in the era of large language models. arXiv preprint arXiv:2407.07959 (2024)."},{"key":"e_1_3_2_1_35_1","unstructured":"Weisong Sun Yiran Zhang Jie Zhu Zhihui Wang Chunrong Fang Yonglong Zhang Yebo Feng Jiangping Huang Xingya Wang Zhi Jin et al. 2025. Commenting Higher-level Code Unit: Full Code Reduced Code or Hierarchical Code Summarization. arXiv preprint arXiv:2503.10737 (2025)."},{"key":"e_1_3_2_1_36_1","volume-title":"Large Language Models for Code Summarization. arXiv preprint arXiv:2405.19032","author":"Szalontai Bal\u00e1zs","year":"2024","unstructured":"Bal\u00e1zs Szalontai, Gerg\u0151 Szalay, Tam\u00e1s M\u00e1rton, Anna Sike, Bal\u00e1zs Pint\u00e9r, and Tibor Gregorics. 2024. Large Language Models for Code Summarization. arXiv preprint arXiv:2405.19032 (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_38_1","volume-title":"An exploratory study on using large language models for mutation testing. arXiv preprint arXiv:2406.09843","author":"Wang Bo","year":"2024","unstructured":"Bo Wang, Mingda Chen, Youfang Lin, Mike Papadakis, and Jie M Zhang. 2024. An exploratory study on using large language models for mutation testing. arXiv preprint arXiv:2406.09843 (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"Can Large Language Models Serve as Evaluators for Code Summarization? arXiv preprint arXiv:2412.01333","author":"Wu Yang","year":"2024","unstructured":"Yang Wu, Yao Wan, Zhaoyang Chu, Wenting Zhao, Ye Liu, Hongyu Zhang, Xuanhua Shi, and Philip S Yu. 2024. Can Large Language Models Serve as Evaluators for Code Summarization? arXiv preprint arXiv:2412.01333 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"Bertscore: Evaluating text generation with bert. arXiv preprint arXiv:1904.09675","author":"Zhang Tianyi","year":"2019","unstructured":"Tianyi Zhang, Varsha Kishore, Felix Wu, Kilian Q Weinberger, and Yoav Artzi. 2019. Bertscore: Evaluating text generation with bert. arXiv preprint arXiv:1904.09675 (2019)."}],"event":{"name":"AIware '26: 3rd ACM International Conference on AI-Powered Software","location":"Montreal QC Canada","acronym":"AIware '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 3rd ACM International Conference on AI-Powered Software"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805760.3814892","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:47:21Z","timestamp":1783241241000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805760.3814892"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":40,"alternative-id":["10.1145\/3805760.3814892","10.1145\/3805760"],"URL":"https:\/\/doi.org\/10.1145\/3805760.3814892","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}