{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T04:06:48Z","timestamp":1779422808633,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,5,26]],"date-time":"2026-05-26T00:00:00Z","timestamp":1779753600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,5,26]]},"DOI":"10.1145\/3786335.3813180","type":"proceedings-article","created":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T03:16:22Z","timestamp":1779419782000},"page":"514-536","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Vibe Code Bench: Evaluating AI Models on End-to-End Web Application Development"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-6044-2067","authenticated-orcid":false,"given":"Hung","family":"Tran","sequence":"first","affiliation":[{"name":"Vals AI, San Francisco, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6554-5198","authenticated-orcid":false,"given":"Langston","family":"Nashold","sequence":"additional","affiliation":[{"name":"Vals AI, San Francisco, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3481-020X","authenticated-orcid":false,"given":"Rayan","family":"Krishnan","sequence":"additional","affiliation":[{"name":"Vals AI, San Francisco, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4808-8028","authenticated-orcid":false,"given":"Antoine","family":"Bigeard","sequence":"additional","affiliation":[{"name":"Vals AI, San Francisco, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4814-0796","authenticated-orcid":false,"given":"Alex","family":"Gu","sequence":"additional","affiliation":[{"name":"MIT, Cambridge, Massachusetts, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,5,26]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Anthropic. 2024. Claude Code. https:\/\/claude.com\/product\/claude-code. Accessed: 2026-03-03."},{"key":"e_1_3_3_1_3_2","unstructured":"Jacob Austin Augustus Odena Maxwell Nye Maarten Bosma Henryk Michalewski David Dohan Ellen Jiang Carrie Cai Michael Terry Quoc Le et\u00a0al. 2021. Program Synthesis with Large Language Models. arXiv preprint (2021). arxiv:https:\/\/arXiv.org\/abs\/2108.07732https:\/\/arxiv.org\/abs\/2108.07732"},{"key":"e_1_3_3_1_4_2","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique Ponde de\u00a0Oliveira Pinto Jared Kaplan Harri Edwards Yuri Burda Nicholas Joseph Greg Brockman et\u00a0al. 2021. Evaluating Large Language Models Trained on Code. arXiv preprint (2021). arxiv:https:\/\/arXiv.org\/abs\/2107.03374https:\/\/arxiv.org\/abs\/2107.03374"},{"key":"e_1_3_3_1_5_2","unstructured":"Cursor. 2024. Cursor: The AI-First Code Editor. https:\/\/cursor.sh."},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","unstructured":"Xiang Deng Jeff Da Edwin Pan Yannis\u00a0Yiming He Charles Ide Kanak Garg Niklas Lauffer Andrew Park Nitin Pasari Chetan Rane Karmini Sampath Maya Krishnan Srivatsa Kundurthy Sean Hendryx Zifan Wang Vijay Bharadwaj Jeff Holm Raja Aluri Chen Bo\u00a0Calvin Zhang Noah Jacobson Bing Liu and Brad Kenstler. 2025. SWE-Bench Pro: Can AI Agents Solve Long-Horizon Software Engineering Tasks? arXiv preprint (2025). arxiv:https:\/\/arXiv.org\/abs\/2509.1694110.48550\/arXiv.2509.16941","DOI":"10.48550\/arXiv.2509.16941"},{"key":"e_1_3_3_1_7_2","unstructured":"Naman Jain King Han Alex Gu Wen-Ding Li Fanjia Yan Tianjun Zhang Sida Wang Armando Solar-Lezama Koushik Sen and Ion Stoica. 2024. LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code. arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2403.07974https:\/\/arxiv.org\/abs\/2403.07974"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Carlos\u00a0E Jimenez John Yang Alexander Wettig Shunyu Yao Kexin Pei Ofir Press and Karthik Narasimhan. 2024. SWE-bench: Can Language Models Resolve Real-World GitHub Issues? arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2310.0677010.48550\/arXiv.2310.06770","DOI":"10.48550\/arXiv.2310.06770"},{"key":"e_1_3_3_1_9_2","unstructured":"Jing\u00a0Yu Koh Robert Lo Lawrence Jang Vikram Duvvur Ming\u00a0Chong Lim Po-Yu Huang Graham Neubig Shuyan Zhou Ruslan Salakhutdinov and Daniel Fried. 2024. VisualWebArena: Evaluating Multimodal Agents on Realistic Visual Web Tasks. arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2401.13649https:\/\/arxiv.org\/abs\/2401.13649"},{"key":"e_1_3_3_1_10_2","unstructured":"Lovable. 2025. Lovable. https:\/\/lovable.dev. Accessed: 2026-03-02."},{"key":"e_1_3_3_1_11_2","unstructured":"Mike\u00a0A. Merrill et\u00a0al. 2025. Terminal-Bench: Benchmarking Agents on Hard Realistic Tasks in Command Line Interfaces. arXiv preprint (2025). arxiv:https:\/\/arXiv.org\/abs\/2601.11868https:\/\/arxiv.org\/abs\/2601.11868"},{"key":"e_1_3_3_1_12_2","unstructured":"Evan Miller. 2024. Adding Error Bars to Evals: A Statistical Approach to Language Model Evaluations. arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2411.00640https:\/\/arxiv.org\/abs\/2411.00640"},{"key":"e_1_3_3_1_13_2","unstructured":"Samuel Miserendino Michele Wang Tejal Patwardhan and Johannes Heidecke. 2025. SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering? arXiv preprint (2025). arxiv:https:\/\/arXiv.org\/abs\/2502.12115https:\/\/arxiv.org\/abs\/2502.12115"},{"key":"e_1_3_3_1_14_2","unstructured":"Magnus M\u00fcller and Gregor Zuni\u0107. 2024. Browser Use: Enable AI to Control Your Browser. https:\/\/github.com\/browser-use\/browser-use. Accessed: 2026-03-03."},{"key":"e_1_3_3_1_15_2","unstructured":"OpenAI. 2024. Introducing SWE-bench Verified. https:\/\/openai.com\/index\/introducing-swe-bench-verified\/. Accessed: 2026-03-03."},{"key":"e_1_3_3_1_16_2","unstructured":"OpenAI. 2025. Codex. https:\/\/developers.openai.com\/codex. Accessed: 2026-03-02."},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","unstructured":"Sida Peng Eirini Kalliamvakou Peter Cihon and Mert Demirer. 2023. The Impact of AI on Developer Productivity: Evidence from GitHub Copilot. arXiv preprint (2023). arxiv:https:\/\/arXiv.org\/abs\/2302.0659010.48550\/arXiv.2302.06590","DOI":"10.48550\/arXiv.2302.06590"},{"key":"e_1_3_3_1_18_2","unstructured":"Replit. 2025. Replit Agent. https:\/\/replit.com\/products\/agent. Accessed: 2026-03-02."},{"key":"e_1_3_3_1_19_2","unstructured":"Stack Overflow. 2024. Developer Survey 2024. https:\/\/survey.stackoverflow.co\/2024."},{"key":"e_1_3_3_1_20_2","unstructured":"Supabase. 2024. Supabase. https:\/\/supabase.com. Accessed: 2026-02-27."},{"key":"e_1_3_3_1_21_2","unstructured":"SWE-Bench. 2026. SWE-Bench Leaderboard. https:\/\/www.swebench.com. Accessed: 2026-02-27."},{"key":"e_1_3_3_1_22_2","unstructured":"Xingyao Wang Boxuan Li Yufan Song Frank\u00a0F. Xu Xiangru Tang Mingchen Zhuge Jiayi Pan Yueqi Song Bowen Li Jasber Singh Hoang\u00a0H. Tran Fuqiang Li Ren Ma Mingzhang Zheng Bill Qian Yanjun Shao Niklas Muennighoff Ziyang Zhang Botian Jiang Yongliang Shen Weiming Lu Stephanie Lin Yuqing Du Wenhu Chen and Graham Neubig. 2024. OpenHands: An Open Platform for AI Software Developers as Generalist Agents. arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2407.16741https:\/\/arxiv.org\/abs\/2407.16741"},{"key":"e_1_3_3_1_23_2","unstructured":"Chunqiu\u00a0Steven Xia Yinlin Deng Soren Dunn and Lingming Zhang. 2024. Agentless: Demystifying LLM-based Software Engineering Agents. arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2407.01489https:\/\/arxiv.org\/abs\/2407.01489"},{"key":"e_1_3_3_1_24_2","unstructured":"John Yang Carlos\u00a0E Jimenez Alexander Wettig Kilian Lieret Shunyu Yao Karthik Narasimhan and Ofir Press. 2024. SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering. arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2405.15793https:\/\/arxiv.org\/abs\/2405.15793"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","unstructured":"Shuyan Zhou Frank\u00a0F. Xu Hao Zhu Xuhui Zhou Robert Lo Abishek Sridhar Xianyi Cheng Yonatan Bisk Daniel Fried Uri Alon and Graham Neubig. 2024. WebArena: A Realistic Web Environment for Building Autonomous Agents. arXiv preprint (2024). arxiv:https:\/\/arXiv.org\/abs\/2307.1385410.48550\/arXiv.2307.13854","DOI":"10.48550\/arXiv.2307.13854"}],"event":{"name":"CAIS '26: ACM Conference on AI and Agentic Systems","location":"San Jose CA USA","acronym":"CAIS '26"},"container-title":["Proceedings of the ACM Conference on AI and Agentic Systems"],"original-title":[],"deposited":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T03:20:54Z","timestamp":1779420054000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3786335.3813180"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,26]]},"references-count":24,"alternative-id":["10.1145\/3786335.3813180","10.1145\/3786335"],"URL":"https:\/\/doi.org\/10.1145\/3786335.3813180","relation":{},"subject":[],"published":{"date-parts":[[2026,5,26]]},"assertion":[{"value":"2026-05-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}