{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T04:06:24Z","timestamp":1779422784867,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,5,26]]},"DOI":"10.1145\/3786335.3813132","type":"proceedings-article","created":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T03:16:22Z","timestamp":1779419782000},"page":"674-688","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["DraftNEPABench: A Benchmark for Drafting NEPA Document Sections with Coding Agents"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3883-5287","authenticated-orcid":false,"given":"Anurag","family":"Acharya","sequence":"first","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8234-4389","authenticated-orcid":false,"given":"Bishal","family":"Lakha","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7650-338X","authenticated-orcid":false,"given":"Rounak","family":"Meyur","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, 0009-0001-8234-4389, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2031-8189","authenticated-orcid":false,"given":"Rohan","family":"Nuttall","sequence":"additional","affiliation":[{"name":"OpenAI, 0009-0001-8234-4389, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5649-0745","authenticated-orcid":false,"given":"Sarthak","family":"Chaturvedi","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, 0009-0001-8234-4389, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3165-8660","authenticated-orcid":false,"given":"Anika","family":"Halappanavar","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4156-5846","authenticated-orcid":false,"given":"Leah","family":"Hare","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2204-6275","authenticated-orcid":false,"given":"Lin","family":"Zeng","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1086-570X","authenticated-orcid":false,"given":"Mike","family":"Parker","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1205-7405","authenticated-orcid":false,"given":"Sai","family":"Munikoti","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0327-3819","authenticated-orcid":false,"given":"Sameera","family":"Horawalavithana","sequence":"additional","affiliation":[{"name":"Pacific Northwest National Laboratory, Richland, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,5,26]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Shubham Agarwal Gaurav Sahu Abhay Puri Issam\u00a0Hadj Laradji Krishnamurthy\u00a0DJ Dvijotham Jason Stanley Laurent Charlin and Christopher Pal. 2024. LitLLMs LLMs for Literature Review: Are we there yet? Transactions on Machine Learning Research 2025 (2024). https:\/\/api.semanticscholar.org\/CorpusID:274965768"},{"key":"e_1_3_3_2_3_2","unstructured":"Anthropic. 2025. Claude Code. https:\/\/claude.com\/product\/claude-code."},{"key":"e_1_3_3_2_4_2","unstructured":"Anthropic. 2025. Claude Sonnet 4.5. https:\/\/www.anthropic.com\/claude\/sonnet."},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"crossref","unstructured":"Christian Basile Stefan\u00a0D. Anker and Gianluigi Savarese. 2025. Large language models to write scientific manuscripts: To be considered but not trusted. Global Cardiology 3 (2025). Issue 2. https:\/\/api.semanticscholar.org\/CorpusID:280244156","DOI":"10.4081\/cardio.2025.74"},{"key":"e_1_3_3_2_6_2","unstructured":"Hengxing Cai Xiaochen Cai Junhan Chang Sihang Li Lin Yao Changxin Wang Zhifeng Gao Hongshuai Wang Yongge Li Mujie Lin Shuwen Yang Jiankun Wang Yuqi Yin Yaqi Li Linfeng Zhang and Guolin Ke. 2024. SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.01976 (2024). https:\/\/api.semanticscholar.org\/CorpusID:268248146"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Andr\u00e9s Castellanos-G\u00f3mez. 2023. Good Practices for Scientific Article Writing with ChatGPT and Other Artificial Intelligence Language Models. Nanomanufacturing 3 (2023) 135\u2013138. Issue 2. https:\/\/api.semanticscholar.org\/CorpusID:258148838","DOI":"10.3390\/nanomanufacturing3020009"},{"key":"e_1_3_3_2_8_2","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et\u00a0al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.06261 (2025)."},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Tu\u00a0Anh Dinh Carlos Mullov Leonard B\u00e4rmann Zhaolin Li Danni Liu Simon Rei\u00df Jueun Lee Nathan Lerzer Fabian Ternava Jianfeng Gao et\u00a0al. 2024. SciEx: Benchmarking large language models on scientific exams with human expert grading and automatic grading. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.10421 (2024).","DOI":"10.18653\/v1\/2024.emnlp-main.647"},{"key":"e_1_3_3_2_10_2","unstructured":"Zhiwei Fei Xiaoyu Shen Dawei Zhu Fengzhe Zhou Zhuo Han Songyang Zhang Kai Chen Zongwen Shen and Jidong Ge. 2023. LawBench: Benchmarking legal knowledge of large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.16289 (2023)."},{"key":"e_1_3_3_2_11_2","unstructured":"Shubham Gandhi Dhruv Shah Manasi\u00a0S. Patwardhan Lovekesh Vig and Gautam\u00a0M. Shroff. 2025. ResearchCodeAgent: An LLM Multi-Agent System for Automated Codification of Research Methodologies. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.20117 (2025). https:\/\/api.semanticscholar.org\/CorpusID:278171454"},{"key":"e_1_3_3_2_12_2","volume-title":"Annual Meeting of the Association for Computational Linguistics","author":"Gao Fan","year":"2023","unstructured":"Fan Gao, Hang Jiang, Rui Yang, Qingcheng Zeng, Jinghui Lu, Moritz Blum, Dairui Liu, Tianwei She, Yuang Jiang, and Irene Li. 2023. Evaluating Large Language Models on Wikipedia-Style Survey Generation. In Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:267783118"},{"key":"e_1_3_3_2_13_2","unstructured":"Krishna Garg Firoz Shaik Sambaran Bandyopadhyay and Cornelia Caragea. 2025. Let\u2019s Use ChatGPT To Write Our Paper! Benchmarking LLMs To Write the Introduction of a Research Paper. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.14273 (2025). https:\/\/api.semanticscholar.org\/CorpusID:280691788"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Alireza Ghafarollahi and Markus\u00a0J. Buehler. 2024. SciAgents: Automating Scientific Discovery Through Bioinspired Multi\u2010Agent Intelligent Graph Reasoning. Advanced Materials 37 (2024) 2413523. Issue 22. https:\/\/api.semanticscholar.org\/CorpusID:274858957","DOI":"10.1002\/adma.202413523"},{"key":"e_1_3_3_2_15_2","unstructured":"Google. 2025. Gemini CLI. https:\/\/github.com\/google-gemini\/gemini-cli."},{"key":"e_1_3_3_2_16_2","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu Yuanzhuo Wang and Jian Guo. 2024. A Survey on LLM-as-a-Judge. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.15594 (2024). https:\/\/api.semanticscholar.org\/CorpusId:274234014"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Neel Guha Julian Nyarko Daniel\u00a0E. Ho Christopher R\u00e9 Adam Chilton Aditya Narayana Alex Chohlas-Wood Austin M.\u00a0K. Peters Brandon Waldon Daniel\u00a0N. Rockmore Diego\u00a0A. Zambrano Dmitry Talisman Enam Hoque Faiz Surani Frank Fagan Galit Sarfaty Gregory\u00a0M. Dickinson Haggai Porat Jason Hegland Jessica Wu Joe Nudell Joel Niklaus John\u00a0J. Nay Jonathan\u00a0H. Choi Kevin\u00a0Patrick Tobia Margaret Hagan Megan Ma Michael\u00a0A. Livermore Nikon Rasumov-Rahe Nils Holzenberger Noam Kolt Peter Henderson Sean Rehaag Sharad Goel Shangsheng Gao Spencer Williams Sunny\u00a0G. Gandhi Tomer Zur Varun\u00a0J. Iyer and Zehua Li. 2023. LegalBench: A Collaboratively Built Benchmark for Measuring Legal Reasoning in Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.11462 (2023). https:\/\/api.semanticscholar.org\/CorpusID:261064672","DOI":"10.2139\/ssrn.4583531"},{"key":"e_1_3_3_2_18_2","unstructured":"Siyuan Guo Cheng Deng Ying Wen Hechang Chen Yi Chang and Jun Wang. 2024. DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.17453 (2024). https:\/\/api.semanticscholar.org\/CorpusId:268033675"},{"key":"e_1_3_3_2_19_2","unstructured":"Taicheng Guo Kehan Guo Bozhao Nan Zhenwen Liang Zhichun Guo Nitesh\u00a0V. Chawla Olaf Wiest and Xiangliang Zhang. 2023. What indeed can GPT models do in chemistry? A comprehensive benchmark on eight tasks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.18365v1 (2023)."},{"key":"e_1_3_3_2_20_2","unstructured":"Dan Hendrycks Collin Burns Anya Chen and Spencer Ball. 2021. CUAD: An expert-annotated NLP dataset for legal contract review. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2103.06268 (2021)."},{"key":"e_1_3_3_2_21_2","unstructured":"Dan Hendrycks Collin Burns Saurav Kadavath Akul Arora Steven Basart Eric Tang Dawn Song and Jacob Steinhardt. 2021. Measuring Mathematical Problem Solving With the MATH Dataset. NeurIPS (2021)."},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Lekang Jiang and Stephan Goetz. 2024. Natural language processing in the patent domain: A survey. Artificial Intelligence Review 58 (2024) 214. https:\/\/api.semanticscholar.org\/CorpusID:268264536","DOI":"10.1007\/s10462-025-11168-z"},{"key":"e_1_3_3_2_23_2","unstructured":"Abhinav Joshi Shounak Paul Akshat Sharma Pawan Goyal Saptarshi Ghosh and Ashutosh Modi. 2024. IL-TUR: Benchmark for Indian legal text understanding and reasoning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.05399 (2024)."},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"crossref","unstructured":"Yuta Koreeda and Christopher\u00a0D Manning. 2021. ContractNLI: A dataset for document-level natural language inference for contracts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2110.01799 (2021).","DOI":"10.18653\/v1\/2021.findings-emnlp.164"},{"key":"e_1_3_3_2_25_2","unstructured":"Jinqi Lai Wensheng Gan Jiayang Wu Zhenlian Qi and Philip\u00a0S. Yu. 2023. Large Language Models in Law: A Survey. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.03718 (2023). https:\/\/api.semanticscholar.org\/CorpusID:266054920"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Haitao Li You Chen Qingyao Ai Yueyue Wu Ruizhe Zhang and Yiqun Liu. 2024. LexEval: A comprehensive Chinese legal benchmark for evaluating large language models. Advances in Neural Information Processing Systems 37 (2024) 25061\u201325094.","DOI":"10.52202\/079017-0790"},{"key":"e_1_3_3_2_27_2","unstructured":"Haitao Li Jiaying Ye Yiran Hu Jia Chen Qingyao Ai Yueyue Wu Junjie Chen Yifan Chen Cheng Luo Quan Zhou et\u00a0al. 2025. CaseGen: A benchmark for multi-stage legal case documents generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.17943 (2025)."},{"key":"e_1_3_3_2_28_2","unstructured":"Junwei Liu Kaixin Wang Yixuan Chen Xin Peng Zhenpeng Chen Lingming Zhang and Yiling Lou. 2024. Large Language Model-Based Agents for Software Engineering: A Survey. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.02977 (2024). https:\/\/api.semanticscholar.org\/CorpusId:272423732"},{"key":"e_1_3_3_2_29_2","unstructured":"Yuxuan Liu Tianchi Yang Shaohan Huang Zihan Zhang Haizhen Huang Furu Wei Weiwei Deng Feng Sun and Qi Zhang. 2024. HD-Eval: Aligning large language model evaluators through hierarchical criteria decomposition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.15754 (2024)."},{"key":"e_1_3_3_2_30_2","unstructured":"Rounak Meyur Hung\u00a0D. Phan Koby\u00a0B. Hayashi Ian Stewart Shivam Sharma Sarthak Chaturvedi Mike Parker Daniel\u00a0M. Nally Sadie\u00a0A. Montgomery Karl Pazdernik Ali Jannesari Mahantesh\u00a0M. Halappanavar Sai Munikoti Sameera Horawalavithana and Anurag Acharya. 2025. Benchmarking LLMs for Environmental Review and Permitting. Workshop on Large Language Models for Scientific and Societal Advances (SciSoc LLM) KDD 2025. https:\/\/kdd25scisocllm.github.io\/accepted\/ Non-archival workshop paper; listed on workshop website."},{"key":"e_1_3_3_2_31_2","unstructured":"Meredith\u00a0Ringel Morris. 2023. Scientists\u2019 Perspectives on the Potential for Generative AI in their Fields. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.01420 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257921814"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"crossref","unstructured":"Savinay Narendra Kaushal Shetty and Adwait Ratnaparkhi. 2024. Enhancing Contract Negotiations with LLM-Based Legal Document Comparison. Proceedings of the Natural Legal Language Processing Workshop 2024 (2024). https:\/\/api.semanticscholar.org\/CorpusID:273901486","DOI":"10.18653\/v1\/2024.nllp-1.11"},{"key":"e_1_3_3_2_33_2","unstructured":"OpenAI. 2025. Codex CLI. https:\/\/developers.openai.com\/codex\/cli."},{"key":"e_1_3_3_2_34_2","unstructured":"OpenAI. 2025. GPT-5. https:\/\/openai.com\/gpt-5."},{"key":"e_1_3_3_2_35_2","unstructured":"OpenAI. 2025. text-embedding-3-small. https:\/\/platform.openai.com\/docs\/models\/text-embedding-3-small."},{"key":"e_1_3_3_2_36_2","unstructured":"Naseela Pervez and Alexander\u00a0J. Titus. 2024. Inclusivity in Large Language Models: Personality Traits and Gender Bias in Scientific Abstracts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.19497 (2024). https:\/\/api.semanticscholar.org\/CorpusID:270845479"},{"key":"e_1_3_3_2_37_2","unstructured":"Ramon Pires Roseval\u00a0Malaquias Junior and Rodrigo Nogueira. 2025. Automatic Legal Writing Evaluation of LLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.21202 (2025). https:\/\/api.semanticscholar.org\/CorpusId:278207672"},{"key":"e_1_3_3_2_38_2","volume-title":"Annual Meeting of the Association for Computational Linguistics","author":"Qian Cheng","year":"2023","unstructured":"Cheng Qian, Wei Liu, Hongzhang Liu, Nuo Chen, Yufan Dang, Jiahao Li, Cheng Yang, Weize Chen, Yusheng Su, Xin Cong, Juyuan Xu, Dahai Li, Zhiyuan Liu, and Maosong Sun. 2023. ChatDev: Communicative Agents for Software Development. In Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusId:270257715"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"Jaromir Savelka and Kevin\u00a0D. Ashley. 2023. The unreasonable effectiveness of large language models in zero-shot semantic annotation of legal texts. Frontiers in Artificial Intelligence 6 (2023). https:\/\/api.semanticscholar.org\/CorpusID:265336249","DOI":"10.3389\/frai.2023.1279794"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"crossref","unstructured":"Ryan Shea and Zhou Yu. 2025. AutoSpec: An Agentic Framework for Automatically Drafting Patent Specification. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.19640 (2025). https:\/\/api.semanticscholar.org\/CorpusID:281505421","DOI":"10.18653\/v1\/2025.findings-emnlp.687"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Wenqi Shi Ran Xu Yuchen Zhuang Yue Yu Jieyu Zhang Hang Wu Yuanda Zhu Joyce\u00a0C. Ho Carl Yang and M.\u00a0D. Wang. 2024. EHRAgent: Code Empowers Large Language Models for Few-shot Complex Tabular Reasoning on Electronic Health Records. Proceedings of the Conference on Empirical Methods in Natural Language Processing. Conference on Empirical Methods in Natural Language Processing 2024 (2024) 22315\u201322339. https:\/\/api.semanticscholar.org\/CorpusId:266998862","DOI":"10.18653\/v1\/2024.emnlp-main.1245"},{"key":"e_1_3_3_2_42_2","unstructured":"Aarohi Srivastava et\u00a0al. 2022. Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2206.04615 (2022). https:\/\/api.semanticscholar.org\/CorpusID:263625818"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730295"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29872"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"crossref","unstructured":"Amulya Suravarjhula Rashi\u00a0Chandrashekhar Agrawal Sakshi\u00a0Jayesh Patel and Rahul Gupta. 2025. Retrieval-Augmented Multi-Agent System for Rapid Statement of Work Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.07569 (2025). https:\/\/api.semanticscholar.org\/CorpusID:280566188","DOI":"10.1109\/ICMLA66185.2025.00218"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"crossref","unstructured":"Arianna Trozze Toby Davies and Bennett Kleinberg. 2024. Large language models in cryptocurrency securities cases: Can a GPT model meaningfully assist lawyers? Artificial Intelligence and Law 33 (2024) 691\u2013737. https:\/\/api.semanticscholar.org\/CorpusID:267783155","DOI":"10.1007\/s10506-024-09399-6"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.642"},{"key":"e_1_3_3_2_48_2","unstructured":"Steven\u00a0H. Wang Maksim Zubkov Kexin Fan Sarah Harrell Yuyang Sun Wei Chen Andreas\u00a0Lindhardt Plesner and Roger Wattenhofer. 2025. ACORD: An Expert-Annotated Retrieval Dataset for Legal Contract Drafting. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.06582 (2025). https:\/\/api.semanticscholar.org\/CorpusID:275471563"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"crossref","unstructured":"Yubo Wang Xueguang Ma Ge Zhang Yuansheng Ni Abhranil Chandra Shiguang Guo Weiming Ren Aaran Arulraj Xuan He Ziyan Jiang et\u00a0al. 2024. MMLU-Pro: A more robust and challenging multi-task language understanding benchmark. Advances in Neural Information Processing Systems 37 (2024) 95266\u201395290.","DOI":"10.52202\/079017-3018"},{"key":"e_1_3_3_2_50_2","unstructured":"Yue Wu Yewen Fan So\u00a0Yeon Min Shrimai Prabhumoye Stephen McAleer Yonatan Bisk Ruslan Salakhutdinov Yuanzhi Li and Tom Mitchell. 2024. AgentKit: Structured LLM Reasoning with Dynamic Graphs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.11483 (2024). https:\/\/arxiv.org\/abs\/2404.11483"},{"key":"e_1_3_3_2_51_2","unstructured":"John Yang Carlos\u00a0E. Jimenez Alexander Wettig Kilian\u00a0Adriano Lieret Shunyu Yao Karthik Narasimhan and Ofir Press. 2024. SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.15793 (2024). https:\/\/arxiv.org\/pdf\/2405.15793.pdf"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"crossref","unstructured":"Ziming You Yumiao Zhang Dexuan Xu Yiwei Lou Yandong Yan Wei Wang Huaming Zhang and Yu Huang. 2025. DatawiseAgent: A Notebook-Centric LLM Agent Framework for Automated Data Science. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.07044 (2025). https:\/\/api.semanticscholar.org\/CorpusId:276902751","DOI":"10.18653\/v1\/2025.emnlp-main.58"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2025\/34"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.737"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3650212.3680384"},{"key":"e_1_3_3_2_56_2","unstructured":"Jianing Zhao Peng Gao Jiannong Cao Zhiyuan Wen Chen Chen Jianing Yin Ruosong Yang and Bo Yuan. 2025. CodeEdu: A Multi-Agent Collaborative Platform for Personalized Coding Education. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.13814 (2025). https:\/\/api.semanticscholar.org\/CorpusId:280048553"},{"key":"e_1_3_3_2_57_2","unstructured":"Yuxuan Zhou Xien Liu Chen Ning and Ji Wu. 2024. MultifacetEval: Multifaceted evaluation to probe LLMs in mastering medical knowledge. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.02919 (2024)."}],"event":{"name":"CAIS '26: ACM Conference on AI and Agentic Systems","location":"San Jose CA USA","acronym":"CAIS '26"},"container-title":["Proceedings of the ACM Conference on AI and Agentic Systems"],"original-title":[],"deposited":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T03:18:33Z","timestamp":1779419913000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3786335.3813132"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,26]]},"references-count":56,"alternative-id":["10.1145\/3786335.3813132","10.1145\/3786335"],"URL":"https:\/\/doi.org\/10.1145\/3786335.3813132","relation":{},"subject":[],"published":{"date-parts":[[2026,5,26]]},"assertion":[{"value":"2026-05-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}