{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T11:01:46Z","timestamp":1785495706596,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3793302.3793579","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T10:42:54Z","timestamp":1785494574000},"page":"807-811","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Where Do AI Coding Agents Fail? An Empirical Study of Failed Agentic Pull Requests in GitHub"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1517-7135","authenticated-orcid":false,"given":"Ramtin","family":"Ehsani","sequence":"first","affiliation":[{"name":"Drexel University, Philadelphia, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0565-7831","authenticated-orcid":false,"given":"Sakshi","family":"Pathak","sequence":"additional","affiliation":[{"name":"Drexel University, Philadelphia, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0856-9151","authenticated-orcid":false,"given":"Shriya","family":"Rawal","sequence":"additional","affiliation":[{"name":"Drexel University, Philadelphia, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4395-1612","authenticated-orcid":false,"given":"Abdullah","family":"Al Mujahid","sequence":"additional","affiliation":[{"name":"Missouri University of Science and Technology, Rolla, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2477-9971","authenticated-orcid":false,"given":"Mia Mohammad","family":"Imran","sequence":"additional","affiliation":[{"name":"Missouri University of Science and Technology, Rolla, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3057-7807","authenticated-orcid":false,"given":"Preetha","family":"Chatterjee","sequence":"additional","affiliation":[{"name":"Drexel University, Philadelphia, Pennsylvania, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","unstructured":"Theophilus Azungah. 2018. Qualitative research: deductive and inductive approaches to data analysis. Qualitative Research Journal 18 4 (2018) 383\u2013400. 10.1108\/QRJ-D-18-00035","DOI":"10.1108\/QRJ-D-18-00035"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"crossref","unstructured":"Shraddha Barke Michael\u00a0B James and Nadia Polikarpova. 2023. Grounded copilot: How programmers interact with code-generating models. Proceedings of the ACM on Programming Languages 7 OOPSLA1 (2023) 85\u2013111.","DOI":"10.1145\/3586030"},{"key":"e_1_3_3_1_4_2","unstructured":"Ira Ceka Saurabh Pujar Shyam Ramji Luca Buratti Gail Kaiser and Baishakhi Ray. 2025. Understanding Software Engineering Agents Through the Lens of Traceability: An Empirical Study. arxiv:https:\/\/arXiv.org\/abs\/2506.08311\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2506.08311"},{"key":"e_1_3_3_1_5_2","unstructured":"Mark Chen. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.03374 (2021)."},{"key":"e_1_3_3_1_6_2","unstructured":"Mark Chen and Jerry\u00a0Tworek et al.2021. Evaluating Large Language Models Trained on Code. (2021). arxiv:https:\/\/arXiv.org\/abs\/2107.03374\u00a0[cs.LG]"},{"key":"e_1_3_3_1_7_2","volume-title":"Sampling Techniques (3rd ed.)","author":"Cochran William\u00a0G.","year":"1977","unstructured":"William\u00a0G. Cochran. 1977. Sampling Techniques (3rd ed.). John Wiley & Sons, New York, NY."},{"key":"e_1_3_3_1_8_2","unstructured":"ConventionalCommits. 2025. https:\/\/www.conventionalcommits.org\/"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASE63991.2025.00122"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","unstructured":"Ramtin Ehsani Sakshi Pathak Esteban Parra Sonia Haiduc and Preetha Chatterjee. 2025. What characteristics make ChatGPT effective for software issue resolution? An empirical study of task project and conversational signals in GitHub issues. Empirical Software Engineering 31 1 (Nov. 2025). 10.1007\/s10664-025-10745-8","DOI":"10.1007\/s10664-025-10745-8"},{"key":"e_1_3_3_1_11_2","unstructured":"GitHub. 2025. https:\/\/github.com\/graphistry\/pygraphistry\/pull\/706"},{"key":"e_1_3_3_1_12_2","unstructured":"GitHub. 2025. https:\/\/github.com\/ruvnet\/ruv-FANN\/pull\/59"},{"key":"e_1_3_3_1_13_2","unstructured":"GitHub. 2025. https:\/\/github.com\/bmad-code-org\/BMAD-METHOD\/pull\/196"},{"key":"e_1_3_3_1_14_2","unstructured":"GitHub. 2025. https:\/\/github.com\/coder\/coder\/pull\/16917"},{"key":"e_1_3_3_1_15_2","unstructured":"GitHub. 2025. https:\/\/github.com\/microsoft\/vscode-cpptools\/pull\/13763"},{"key":"e_1_3_3_1_16_2","unstructured":"GitHub. 2025. https:\/\/github.com\/getsentry\/sentry-javascript\/pull\/16526"},{"key":"e_1_3_3_1_17_2","unstructured":"GitHub. 2025. https:\/\/github.com\/hyperlight-dev\/hyperlight\/pull\/641"},{"key":"e_1_3_3_1_18_2","unstructured":"GitHub. 2025. https:\/\/github.com\/firecrawl\/firecrawl\/pull\/1645"},{"key":"e_1_3_3_1_19_2","unstructured":"GitHub. 2025. https:\/\/github.com\/reflex-dev\/reflex-web\/pull\/1479"},{"key":"e_1_3_3_1_20_2","unstructured":"GitHub. 2025. https:\/\/github.com\/netdata\/netdata\/pull\/20631"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/WSC.2015.7408210"},{"key":"e_1_3_3_1_22_2","unstructured":"Kosei Horikawa Hao Li Yutaro Kashiwa Bram Adams Hajimu Iida and Ahmed\u00a0E. Hassan. 2025. Agentic Refactoring: An Empirical Study of AI Coding Agents. arxiv:https:\/\/arXiv.org\/abs\/2511.04824\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2511.04824"},{"key":"e_1_3_3_1_23_2","unstructured":"Carlos\u00a0E. Jimenez John Yang Alexander Wettig Shunyu Yao Kexin Pei Ofir Press and Karthik Narasimhan. 2024. SWE-bench: Can Language Models Resolve Real-World GitHub Issues? arxiv:https:\/\/arXiv.org\/abs\/2310.06770\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2310.06770"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","unstructured":"Sungmin Kang Gabin An and Shin Yoo. 2024. A Quantitative and Qualitative Evaluation of LLM-Based Explainable Fault Localization. Proc. ACM Softw. Eng. 1 FSE Article 64 (July 2024) 23\u00a0pages. 10.1145\/3660771","DOI":"10.1145\/3660771"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","unstructured":"Barbara Kitchenham Lech Madeyski David Budgen Jacky Keung Pearl Brereton Stuart Charters Shirley Gibbs and Amnart Pohthong. 2017. Robust Statistical Methods for Empirical Software Engineering. Empirical Software Engineering 22 2 (April 2017) 579\u2013630. 10.1007\/s10664-016-9437-5","DOI":"10.1007\/s10664-016-9437-5"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","unstructured":"Valentina Lenarduzzi Vili Nikkola Nyyti Saarim\u00e4ki and Davide Taibi. 2021. Does code quality affect pull request acceptance? An empirical study. Journal of Systems and Software 171 (2021) 110806. 10.1016\/j.jss.2020.110806","DOI":"10.1016\/j.jss.2020.110806"},{"key":"e_1_3_3_1_27_2","unstructured":"Hao Li Haoxiang Zhang and Ahmed\u00a0E. Hassan. 2025. The Rise of AI Teammates in Software Engineering (SE) 3.0: How Autonomous Coding Agents Are Reshaping Software Engineering. arxiv:https:\/\/arXiv.org\/abs\/2507.15003\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2507.15003"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASE56229.2023.00030"},{"key":"e_1_3_3_1_29_2","unstructured":"Oorja Majgaonkar Zhiwei Fei Xiang Li Federica Sarro and He Ye. 2025. Understanding Code Agent Behaviour: An Empirical Study of Success and Failure Trajectories. arxiv:https:\/\/arXiv.org\/abs\/2511.00197\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2511.00197"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Mary McHugh. 2012. Interrater reliability: The kappa statistic. Biochemia medica : \u010dasopis Hrvatskoga dru\u0161tva medicinskih biokemi\u010dara \/ HDMB 22 (10 2012) 276\u201382.","DOI":"10.11613\/BM.2012.031"},{"key":"e_1_3_3_1_31_2","unstructured":"Noor Nashid Daniel Ding Keheliya Gallaba Ahmed\u00a0E. Hassan and Ali Mesbah. 2025. Characterizing Multi-Hunk Patches: Divergence Proximity and LLM Repair Challenges. arxiv:https:\/\/arXiv.org\/abs\/2506.04418\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2506.04418"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","unstructured":"John Pangas Suhaib Mujahid Ahmad Abdellatif and Marco Castelluccio. 2025. Using LLMs to Bridge the Gaps in QA Test Plans at Firefox. IEEE Software (2025) 1\u20137. 10.1109\/MS.2025.3621128","DOI":"10.1109\/MS.2025.3621128"},{"key":"e_1_3_3_1_33_2","unstructured":"ReplicationPackage. 2025. https:\/\/github.com\/SOAR-Lab\/MSR2026_AIDev"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"crossref","unstructured":"Amirali Sajadi Binh Le Anh Nguyen Kostadin Damevski and Preetha Chatterjee. 2025. Do LLMs consider security? an empirical study on responses to programming questions. Empirical Software Engineering 30 3 (2025) 101.","DOI":"10.1007\/s10664-025-10658-6"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"crossref","unstructured":"Noah Shinn Federico Cassano Ashwin Gopinath Karthik Narasimhan and Shunyu Yao. 2023. Reflexion: Language agents with verbal reinforcement learning. Advances in Neural Information Processing Systems 36 (2023) 8634\u20138652.","DOI":"10.52202\/075280-0377"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICMLA.2015.41"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","unstructured":"Gail\u00a0M. Sullivan and Richard Feinn. 2012. Using Effect Size\u2014or Why the P Value Is Not Enough. Journal of Graduate Medical Education 4 3 (Sept. 2012) 279\u2013282. 10.4300\/JGME-D-12-00156.1","DOI":"10.4300\/JGME-D-12-00156.1"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"publisher","unstructured":"Michael\u00a0C. Thrun Tino Gehlert and Alfred Ultsch. 2020. Analyzing the fine structure of distributions. PLOS ONE 15 10 (Oct. 2020) e0238835. 10.1371\/journal.pone.0238835","DOI":"10.1371\/journal.pone.0238835"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/MSR.2015.40"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3623342"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"crossref","unstructured":"John Yang Carlos\u00a0E Jimenez Alexander Wettig Kilian Lieret Shunyu Yao Karthik Narasimhan and Ofir Press. 2024. Swe-agent: Agent-computer interfaces enable automated software engineering. Advances in Neural Information Processing Systems 37 (2024) 50528\u201350652.","DOI":"10.52202\/079017-1601"},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/SANER.2019.8667996"},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"publisher","unstructured":"Xunhui Zhang Yue Yu Georgios Gousios and Ayushi Rastogi. 2023. Pull Request Decisions Explained: An Empirical Overview. IEEE Transactions on Software Engineering 49 2 (2023) 849\u2013871. 10.1109\/TSE.2022.3165056","DOI":"10.1109\/TSE.2022.3165056"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASE.2017.8115619"}],"event":{"name":"MSR '26: 23rd International Conference on Mining Software Repositories","location":"Rio de Janeiro Brazil","acronym":"MSR '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 23rd International Conference on Mining Software Repositories"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3793302.3793579","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T10:48:40Z","timestamp":1785494920000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3793302.3793579"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":43,"alternative-id":["10.1145\/3793302.3793579","10.1145\/3793302"],"URL":"https:\/\/doi.org\/10.1145\/3793302.3793579","relation":{},"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}