{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T17:06:55Z","timestamp":1784653615907,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,12]],"date-time":"2026-04-12T00:00:00Z","timestamp":1775952000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,12]]},"DOI":"10.1145\/3793655.3793709","type":"proceedings-article","created":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T16:04:46Z","timestamp":1784649886000},"page":"249-253","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Tricky\u00b2: Towards a Benchmark for Evaluating Human and LLM Error Interactions"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6787-8101","authenticated-orcid":false,"given":"Cole","family":"Granger","sequence":"first","affiliation":[{"name":"William &amp; Mary, Williamsburg, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4489-7733","authenticated-orcid":false,"given":"Dipin","family":"Khati","sequence":"additional","affiliation":[{"name":"William &amp; Mary, Williamsburg, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3238-1229","authenticated-orcid":false,"given":"Daniel","family":"Rodriguez-Cardenas","sequence":"additional","affiliation":[{"name":"William &amp; Mary, Williamsburg, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5626-7586","authenticated-orcid":false,"given":"Denys","family":"Poshyvanyk","sequence":"additional","affiliation":[{"name":"William &amp; Mary, Williamsburg, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,21]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","unstructured":"Mamdouh Alenezi and Mohammed Akour. 2025. AI-Driven Innovations in Software Engineering: A Review of Current Practices and Future Directions. Applied Sciences 15 (01 2025). 10.3390\/app15031344","DOI":"10.3390\/app15031344"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","unstructured":"Nabeel Alzahrani and Frank Vahid. 2021. Common Logic Errors for Programming Learners: A Three-decade Literature Survey. 10.18260\/1-2\u201336814","DOI":"10.18260\/1-2\u201336814"},{"key":"e_1_3_3_2_4_2","unstructured":"Enna Basic and Alberto Giaretta. 2025. From Vulnerabilities to Remediation: A Systematic Literature Review of LLMs in Code Security. arxiv:https:\/\/arXiv.org\/abs\/2412.15004\u00a0[cs.CR] https:\/\/arxiv.org\/abs\/2412.15004"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","unstructured":"Dipin\u00a0Khati Cole\u00a0Granger. 2025. Tricky\u00b2 Benchmark Dataset. 10.5281\/zenodo.17679850","DOI":"10.5281\/zenodo.17679850"},{"key":"e_1_3_3_2_6_2","unstructured":"Dipin\u00a0Khati Cole\u00a0Granger. 2025. Tricky\u00b2 Scripts and Evaluation Code. https:\/\/github.com\/WM-SEMERU\/prj-syntax-errors"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISSRE.2009.14"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"crossref","unstructured":"Tuan Dinh Jinman Zhao Samson Tan Renato Negrinho Leonard Lausen Sheng Zha and George Karypis. 2023. Large Language Models of Code Fail at Completing Code with Potential Bugs. arxiv:https:\/\/arXiv.org\/abs\/2306.03438\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2306.03438","DOI":"10.52202\/075280-1794"},{"key":"e_1_3_3_2_9_2","unstructured":"Sean Endicott. 2025. GitHub copilot just crossed 15 million users - is Microsoft\u2019s AI coding push working?https:\/\/www.windowscentral.com\/software-apps\/over-15-million-developers-now-use-this-ai-coding-tool-from-microsoft"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","unstructured":"Md.\u00a0Asraful Haque. 2025. LLMs: A game-changer for software engineers? BenchCouncil Transactions on Benchmarks Standards and Evaluations 5 1 (March 2025) 100204. 10.1016\/j.tbench.2025.100204","DOI":"10.1016\/j.tbench.2025.100204"},{"key":"e_1_3_3_2_11_2","unstructured":"Carlos\u00a0E. Jimenez John Yang Alexander Wettig Shunyu Yao Kexin Pei Ofir Press and Karthik Narasimhan. 2024. SWE-bench: Can Language Models Resolve Real-World GitHub Issues? arxiv:https:\/\/arXiv.org\/abs\/2310.06770\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2310.06770"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/2610384.2628055"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642596"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Claire Le\u00a0Goues Neal Holtschulte Edward\u00a0K. Smith Yuriy Brun Premkumar Devanbu Stephanie Forrest and Westley Weimer. 2015. The ManyBugs and IntroClass Benchmarks for Automated Repair of C Programs. IEEE Trans. Softw. Eng. 41 12 (Dec. 2015) 1236\u20131256. 10.1109\/TSE.2015.2454513","DOI":"10.1109\/TSE.2015.2454513"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3643991.3644870"},{"key":"e_1_3_3_2_16_2","unstructured":"Nat McAleese Rai\u00a0Michael Pokorny Juan Felipe\u00a0Ceron Uribe Evgenia Nitishinskaya Maja Trebacz and Jan Leike. 2024. LLM Critics Help Catch LLM Bugs. arxiv:https:\/\/arXiv.org\/abs\/2407.00215\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2407.00215"},{"key":"e_1_3_3_2_17_2","unstructured":"Hammond Pearce Baleegh Ahmad Benjamin Tan Brendan Dolan-Gavitt and Ramesh Karri. 2021. Asleep at the Keyboard? Assessing the Security of GitHub Copilot\u2019s Code Contributions. arxiv:https:\/\/arXiv.org\/abs\/2108.09293\u00a0[cs.CR] https:\/\/arxiv.org\/abs\/2108.09293"},{"key":"e_1_3_3_2_18_2","unstructured":"Joe Procopio. [n. d.]. https:\/\/www.inc.com\/joe-procopio\/anthropics-ceo-said-all-code-will-be-ai-generated-in-a-year\/91163367"},{"key":"e_1_3_3_2_19_2","unstructured":"Florian Tambon Arghavan\u00a0Moradi Dakhel Amin Nikanjam Foutse Khomh Michel\u00a0C. Desmarais and Giuliano Antoniol. 2024. Bugs in Large Language Models Generated Code: An Empirical Study. arxiv:https:\/\/arXiv.org\/abs\/2403.08937\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2403.08937"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3519665"},{"key":"e_1_3_3_2_21_2","unstructured":"Linas Vidziunas David Binkley and Leon Moonen. 2024. The Impact of Program Reduction on Automated Program Repair. arxiv:https:\/\/arXiv.org\/abs\/2408.01134\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2408.01134"},{"key":"e_1_3_3_2_22_2","unstructured":"Zhijie Wang Zijie Zhou Da Song Yuheng Huang Shengmai Chen Lei Ma and Tianyi Zhang. 2025. Towards Understanding the Characteristics of Code Generation Errors Made by Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2406.08731\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2406.08731"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3368089.3417943"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","unstructured":"Qi Xin Haojun Wu Jinran Tang Xinyu Liu Steven\u00a0P. Reiss and Jifeng Xuan. 2024. Detecting Creating Repairing and Understanding Indivisible Multi-Hunk Bugs. Proc. ACM Softw. Eng. 1 FSE Article 121 (July 2024) 24\u00a0pages. 10.1145\/3660828","DOI":"10.1145\/3660828"}],"event":{"name":"FORGE '26: IEEE\/ACM Third International Conference on AI Foundation Models and Software Engineering","location":"Rio de Janeiro , Brazil","acronym":"FORGE '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 2026 IEEE\/ACM Third International Conference on AI Foundation Models and Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3793655.3793709","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T16:47:54Z","timestamp":1784652474000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3793655.3793709"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":23,"alternative-id":["10.1145\/3793655.3793709","10.1145\/3793655"],"URL":"https:\/\/doi.org\/10.1145\/3793655.3793709","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-07-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}