{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T22:59:16Z","timestamp":1780873156325,"version":"3.54.1"},"reference-count":112,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,13]],"date-time":"2026-05-13T00:00:00Z","timestamp":1778630400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information and Software Technology"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.infsof.2026.108185","type":"journal-article","created":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T03:26:42Z","timestamp":1778556402000},"page":"108185","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["An overview of evaluation and enhancement methods for code generation by large language models"],"prefix":"10.1016","volume":"197","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2653-3286","authenticated-orcid":false,"given":"Jacob","family":"Truong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5838-3409","authenticated-orcid":false,"given":"Van","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9709-1663","authenticated-orcid":false,"given":"Thanh Thi","family":"Nguyen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"8","key":"10.1016\/j.infsof.2026.108185_b1","doi-asserted-by":"crossref","first-page":"165","DOI":"10.1145\/960118.808380","article-title":"The history of fortran i, ii, and iii","volume":"13","author":"Backus","year":"1978","journal-title":"ACM SIGPLAN Not."},{"key":"10.1016\/j.infsof.2026.108185_b2","series-title":"Compilers: Principles, Techniques, & Tools","year":"2007"},{"key":"10.1016\/j.infsof.2026.108185_b3","series-title":"IBM system\/360 Operating System Assembler Language","author":"IBM","year":"1967"},{"issue":"3","key":"10.1016\/j.infsof.2026.108185_b4","doi-asserted-by":"crossref","first-page":"201","DOI":"10.1145\/155360.155580","article-title":"The development of the C language","volume":"28","author":"Ritchie","year":"1993","journal-title":"ACM SIGPLAN Not."},{"key":"10.1016\/j.infsof.2026.108185_b5","article-title":"Yacc: yet another compiler-compiler","author":"Johnson","year":"1978"},{"key":"10.1016\/j.infsof.2026.108185_b6","article-title":"Smalltalk-80: the interactive programming environment","author":"Goldberg","year":"1984"},{"key":"10.1016\/j.infsof.2026.108185_b7","series-title":"Wiley Encyclopedia of Electrical and Electronics Engineering","year":"2000"},{"key":"10.1016\/j.infsof.2026.108185_b8","series-title":"MDA Guide Version 1.0.1","author":"Miller","year":"2003"},{"key":"10.1016\/j.infsof.2026.108185_b9","series-title":"MDA explained: the model driven architecture: practice and promise","author":"Kleppe","year":"2003"},{"key":"10.1016\/j.infsof.2026.108185_b10","first-page":"50","article-title":"Inductive programming: a survey of program synthesis techniques","volume":"vol. 5812","author":"Kitzelmann","year":"2010"},{"key":"10.1016\/j.infsof.2026.108185_b11","article-title":"Your wish is my command: programming by example","author":"Lieberman","year":"2001"},{"issue":"6","key":"10.1016\/j.infsof.2026.108185_b12","doi-asserted-by":"crossref","first-page":"419","DOI":"10.1145\/2666356.2594321","article-title":"Code completion with statistical language models","volume":"49","author":"Raychev","year":"2014","journal-title":"ACM SIGPLAN Not."},{"key":"10.1016\/j.infsof.2026.108185_b13","series-title":"Proceedings of the 2013 9th Joint Meeting on Foundations of Software Engineering","first-page":"532","article-title":"A statistical semantic language model for source code","author":"Nguyen","year":"2013"},{"key":"10.1016\/j.infsof.2026.108185_b14","first-page":"1","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.infsof.2026.108185_b15","series-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.infsof.2026.108185_b16","series-title":"Codexglue: a machine learning benchmark dataset for code understanding and generation","author":"Lu","year":"2021"},{"key":"10.1016\/j.infsof.2026.108185_b17","series-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"key":"10.1016\/j.infsof.2026.108185_b18","first-page":"1877","article-title":"Language models are few-shot learners","volume":"vol. 33","author":"Brown","year":"2020"},{"key":"10.1016\/j.infsof.2026.108185_b19","series-title":"Introducing GitHub copilot: your AI pair programmer","author":"Friedman","year":"2021"},{"key":"10.1016\/j.infsof.2026.108185_b20","series-title":"Early LLM-based tools for enterprise information workers likely provide meaningful boosts to productivity","author":"Cambon","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b21","doi-asserted-by":"crossref","DOI":"10.1145\/3772721","article-title":"A survey on large language models for code generation","author":"Jiang","year":"2025","journal-title":"ACM Trans. Softw. Eng. Methodol."},{"key":"10.1016\/j.infsof.2026.108185_b22","series-title":"2024 IEEE International Conference on Big Data (BigData)","first-page":"5402","article-title":"Code LLMs: A taxonomy-based survey","author":"Raihan","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b23","series-title":"Large language models for code generation: A comprehensive survey of challenges, techniques, evaluation, and applications","author":"Huynh","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b24","series-title":"Systems and Software Engineering \u2014 Systems and Software Quality Requirements and Evaluation (Square) \u2014 Product Quality Model","author":"for Standardization","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b25","series-title":"Do we need improved code quality metrics?","author":"Sharma","year":"2020"},{"issue":"4","key":"10.1016\/j.infsof.2026.108185_b26","doi-asserted-by":"crossref","first-page":"308","DOI":"10.1109\/TSE.1976.233837","article-title":"A complexity measure","volume":"SE-2","author":"McCabe","year":"1976","journal-title":"IEEE Trans. Softw. Eng."},{"key":"10.1016\/j.infsof.2026.108185_b27","series-title":"Elements of Software Science (Operating and programming systems series)","author":"Halstead","year":"1977"},{"issue":"6","key":"10.1016\/j.infsof.2026.108185_b28","doi-asserted-by":"crossref","first-page":"476","DOI":"10.1109\/32.295895","article-title":"A metrics suite for object oriented design","volume":"20","author":"Chidamber","year":"1994","journal-title":"IEEE Trans. Softw. Eng."},{"key":"10.1016\/j.infsof.2026.108185_b29","unstructured":"F. Brito, R. dos Santos Carapu\u00e7a, et al., Object-oriented software engineering: measuring and controlling the development process, in: 4th International Conference on Software Quality, 1994, pp. 1\u20138."},{"key":"10.1016\/j.infsof.2026.108185_b30","series-title":"Designing Object-Oriented C++ Applications","author":"Martin","year":"1995"},{"key":"10.1016\/j.infsof.2026.108185_b31","series-title":"2010 7th IEEE Working Conference on Mining Software Repositories (MSR 2010)","first-page":"31","article-title":"An extensive comparison of bug prediction approaches","author":"D\u2019Ambros","year":"2010"},{"issue":"2","key":"10.1016\/j.infsof.2026.108185_b32","doi-asserted-by":"crossref","first-page":"111","DOI":"10.1016\/0164-1212(93)90077-B","article-title":"Object-oriented metrics that predict maintainability","volume":"23","author":"Li","year":"1993","journal-title":"J. Syst. Softw."},{"key":"10.1016\/j.infsof.2026.108185_b33","series-title":"Object-Oriented Metrics in Practice: Using Software Metrics to Characterize, Evaluate, and Improve the Design of Object-Oriented Systems","author":"Lanza","year":"2007"},{"key":"10.1016\/j.infsof.2026.108185_b34","series-title":"From vulnerabilities to remediation: a systematic literature review of LLMs in code security","author":"Basic","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b35","series-title":"Unveiling inefficiencies in llm-generated code: toward a comprehensive taxonomy","author":"Abbassi","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b36","first-page":"1","article-title":"SPoC: search-based pseudocode to code","volume":"vol. 32","author":"Kulal","year":"2019"},{"key":"10.1016\/j.infsof.2026.108185_b37","series-title":"Program synthesis with large language models","author":"Austin","year":"2021"},{"key":"10.1016\/j.infsof.2026.108185_b38","series-title":"Findings of the Association for Computational Linguistics ACL 2024","first-page":"3603","article-title":"DevEval: a manually-annotated code generation benchmark aligned with real-world code repositories","author":"Li","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b39","series-title":"Codebleu: a method for automatic evaluation of code synthesis","author":"Ren","year":"2020"},{"key":"10.1016\/j.infsof.2026.108185_b40","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2023","first-page":"1271","article-title":"Execution-based evaluation for open-domain code generation","author":"Wang","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b41","series-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing","first-page":"13921","article-title":"CodeBERTScore: Evaluating code generation with pretrained models of code","author":"Zhou","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b42","series-title":"Proceedings of the 1st International Workshop on Mining Software Repositories Applications for Privacy and Security","first-page":"29","article-title":"SecurityEval dataset: mining vulnerability examples to evaluate machine learning-based code generation techniques","author":"Siddiq","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b43","series-title":"2023 IEEE\/ACM 20th International Conference on Mining Software Repositories (MSR)","first-page":"588","article-title":"Llmseceval: a dataset of natural language prompts for security evaluations","author":"Tony","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b44","series-title":"2024 IEEE Conference on Secure and Trustworthy Machine Learning (SaTML)","first-page":"684","article-title":"Codelmsec benchmark: systematically evaluating and finding security vulnerabilities in black-box code language models","author":"Hajipour","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b45","series-title":"The Thirty-Ninth Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track","article-title":"SECODEPLT: A unified benchmark for evaluating the security risks and capabilities of code genAI","author":"Nie","year":"2025"},{"issue":"2","key":"10.1016\/j.infsof.2026.108185_b46","doi-asserted-by":"crossref","first-page":"96","DOI":"10.1145\/3610721","article-title":"Asleep at the keyboard? Assessing the security of GitHub copilot\u2019s code contributions","volume":"68","author":"Pearce","year":"2025","journal-title":"Commun. ACM"},{"key":"10.1016\/j.infsof.2026.108185_b47","series-title":"Purple llama cyberseceval: a secure coding benchmark for language models","author":"Bhatt","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b48","series-title":"Security and quality in llm-generated code: a multi-language, multi-model analysis","author":"Kharma","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b49","series-title":"Constrained decoding for secure code generation","author":"Fu","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b50","first-page":"11506","article-title":"Effibench: benchmarking the efficiency of automatically generated code","volume":"vol. 37","author":"Huang","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b51","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"How efficient is LLM-generated code? A rigorous & high-standard benchmark","author":"Qiu","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b52","series-title":"2025 ACM\/IEEE International Symposium on Empirical Software Engineering and Measurement (ESEM)","first-page":"151","article-title":"Is LLM-generated code more maintainable & reliable than human-written code?","author":"Molison","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b53","series-title":"Comparing human and LLM generated code: the jury is still out!","author":"Licorish","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b54","series-title":"Proceedings of the 37th IEEE\/ACM International Conference on Automated Software Engineering","first-page":"1","article-title":"How readable is model-generated code? Examining readability and visual inspection of github copilot","author":"Al Madi","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b55","series-title":"Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","first-page":"5673","article-title":"Codegeex: a pre-trained model for code generation with multilingual benchmarking on humaneval-x","author":"Zheng","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b56","series-title":"The Eleventh International Conference on Learning Representations","article-title":"Multi-lingual evaluation of code generation models","author":"Athiwaratkun","year":"2023"},{"issue":"7","key":"10.1016\/j.infsof.2026.108185_b57","doi-asserted-by":"crossref","first-page":"3675","DOI":"10.1109\/TSE.2023.3267446","article-title":"Multipl-e: a scalable and polyglot approach to benchmarking neural code generation","volume":"49","author":"Cassano","year":"2023","journal-title":"IEEE Trans. Softw. Eng."},{"key":"10.1016\/j.infsof.2026.108185_b58","series-title":"Thirty-Fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track","article-title":"Measuring coding challenge competence with apps","author":"Hendrycks","year":"2021"},{"key":"10.1016\/j.infsof.2026.108185_b59","series-title":"Proceedings of the 40th International Conference on Machine Learning","article-title":"DS-1000: a natural and reliable benchmark for data science code generation","author":"Lai","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b60","series-title":"Training and evaluating a jupyter notebook data science assistant","author":"Chandel","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b61","series-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","article-title":"Is your code generated by ChatGPT really correct? rigorous evaluation of large language models for code generation","author":"Liu","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b62","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"Bigcodebench: benchmarking code generation with diverse function calls and complex instructions","author":"Zhuo","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b63","series-title":"ClassEval: a manually-crafted benchmark for evaluating llms on class-level code generation","author":"Du","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b64","series-title":"Proceedings of the IEEE\/ACM 46th International Conference on Software Engineering","first-page":"1","article-title":"CoderEval: a benchmark of pragmatic code generation with generative pre-trained models","author":"Yu","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b65","first-page":"57619","article-title":"EvoCodeBench: an evolving code generation benchmark with domain-specific evaluations","volume":"vol. 37","author":"Li","year":"2024"},{"issue":"6624","key":"10.1016\/j.infsof.2026.108185_b66","doi-asserted-by":"crossref","first-page":"1092","DOI":"10.1126\/science.abq1158","article-title":"Competition-level code generation with AlphaCode","volume":"378","author":"Li","year":"2022","journal-title":"Science"},{"key":"10.1016\/j.infsof.2026.108185_b67","series-title":"2025 IEEE\/ACM International Workshop on Large Language Models for Code (LLM4Code)","first-page":"33","article-title":"CWEval: Outcome-driven evaluation on functionality and security of LLM code generation","author":"Peng","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b68","series-title":"Evaluation of software product quality metrics","author":"Molnar","year":"2020"},{"key":"10.1016\/j.infsof.2026.108185_b69","series-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","article-title":"CodeRL: mastering code generation through pretrained models and deep reinforcement learning","author":"Le","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b70","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TBDATA.2024.3524104","article-title":"Aligning crowd-sourced human feedback for reinforcement learning on code generation by large language models","author":"Wong","year":"2024","journal-title":"IEEE Trans. Big Data"},{"key":"10.1016\/j.infsof.2026.108185_b71","article-title":"Execution-based code generation using deep reinforcement learning","author":"Shojaee","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.infsof.2026.108185_b72","series-title":"Rlef: grounding code llms in execution feedback with reinforcement learning","author":"Gehring","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b73","article-title":"Rltf: reinforcement learning from unit test feedback","author":"Liu","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.infsof.2026.108185_b74","series-title":"The Twelfth International Conference on Learning Representations","article-title":"Wizardcoder: empowering code large language models with evol-instruct","author":"Luo","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b75","series-title":"Forty-First International Conference on Machine Learning","article-title":"Magicoder: empowering code generation with oss-instruct","author":"Wei","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b76","series-title":"Forty-First International Conference on Machine Learning","article-title":"Instruction tuning for secure code generation","author":"He","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b77","doi-asserted-by":"crossref","DOI":"10.1145\/3714461","article-title":"Exploring parameter-efficient fine-tuning techniques for code generation with large language models","author":"Weyssow","year":"2025","journal-title":"ACM Trans. Softw. Eng. Methodol."},{"key":"10.1016\/j.infsof.2026.108185_b78","series-title":"Proceedings of the 2024 IEEE\/ACM First International Conference on AI Foundation Models and Software Engineering","first-page":"103","article-title":"On evaluating the efficiency of source code generated by llms","author":"Niu","year":"2024"},{"issue":"9","key":"10.1016\/j.infsof.2026.108185_b79","doi-asserted-by":"crossref","first-page":"2437","DOI":"10.1109\/TSE.2024.3440503","article-title":"Chain-of-thought in neural code generation: from and for lightweight language models","volume":"50","author":"Yang","year":"2024","journal-title":"IEEE Trans. Softw. Eng."},{"key":"10.1016\/j.infsof.2026.108185_b80","first-page":"84482","article-title":"Effilearner: enhancing efficiency of generated code via self-optimization","volume":"vol. 37","author":"Huang","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b81","series-title":"Extended Abstracts of the 2021 CHI Conference on Human Factors in Computing Systems","first-page":"1","article-title":"Prompt programming for large language models: beyond the few-shot paradigm","author":"Reynolds","year":"2021"},{"key":"10.1016\/j.infsof.2026.108185_b82","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume":"vol. 35","author":"Wei","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b83","first-page":"109","article-title":"Catastrophic interference in connectionist networks: The sequential learning problem","volume":"vol. 24","author":"McCloskey","year":"1989"},{"key":"10.1016\/j.infsof.2026.108185_b84","series-title":"2024 IEEE 40th International Conference on Data Engineering","first-page":"3696","article-title":"Cost-effective in-context learning for entity resolution: A design space exploration","author":"Fan","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b85","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024","first-page":"3144","article-title":"Large language models are limited in out-of-context knowledge reasoning","author":"Hu","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b86","series-title":"Proceedings of the 31st International Conference on Computational Linguistics","first-page":"1880","article-title":"Efficient solutions for an intriguing failure of LLMs: Long context window does not mean LLMs can analyze long sequences flawlessly","author":"Hosseini","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b87","series-title":"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"328","article-title":"Universal language model fine-tuning for text classification","author":"Howard","year":"2018"},{"key":"10.1016\/j.infsof.2026.108185_b88","series-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"3470","article-title":"Cross-task generalization via natural language crowdsourcing instructions","author":"Mishra","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b89","series-title":"International Conference on Learning Representations","article-title":"Finetuned language models are zero-shot learners","author":"Wei","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b90","series-title":"International Conference on Learning Representations","article-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2022"},{"key":"10.1016\/j.infsof.2026.108185_b91","series-title":"Thirty-Seventh Conference on Neural Information Processing Systems","article-title":"QLoRA: Efficient finetuning of quantized LLMs","author":"Dettmers","year":"2023"},{"issue":"3","key":"10.1016\/j.infsof.2026.108185_b92","doi-asserted-by":"crossref","first-page":"220","DOI":"10.1038\/s42256-023-00626-4","article-title":"Parameter-efficient fine-tuning of large-scale pre-trained language models","volume":"5","author":"Ding","year":"2023","journal-title":"Nat. Mach. Intell."},{"key":"10.1016\/j.infsof.2026.108185_b93","series-title":"Forty-Second International Conference on Machine Learning","article-title":"SFT memorizes, RL generalizes: A comparative study of foundation model post-training","author":"Chu","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b94","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"7765","article-title":"Does fine-tuning LLMs on new knowledge encourage hallucinations?","author":"Gekhman","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b95","series-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","first-page":"4302","article-title":"Deep reinforcement learning from human preferences","author":"Christiano","year":"2017"},{"key":"10.1016\/j.infsof.2026.108185_b96","series-title":"Fine-tuning language models from human preferences","author":"Ziegler","year":"2019"},{"key":"10.1016\/j.infsof.2026.108185_b97","series-title":"2025 5th Intelligent Cybersecurity Conference","first-page":"272","article-title":"Reward hacking in reinforcement learning and RLHF: A multidisciplinary examination of vulnerabilities, mitigation strategies, and alignment challenges","author":"Hu","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b98","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"580","article-title":"Mitigating the alignment tax of RLHF","author":"Lin","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b99","first-page":"50528","article-title":"SWE-agent: Agent-computer interfaces enable automated software engineering","volume":"vol. 37","author":"Yang","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b100","series-title":"First Conference on Language Modeling","article-title":"AutoGen: Enabling next-gen LLM applications via multi-agent conversations","author":"Wu","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b101","series-title":"The Twelfth International Conference on Learning Representations","article-title":"MetaGPT: Meta programming for a multi-agent collaborative framework","author":"Hong","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b102","series-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","article-title":"SELF-REFINE: iterative refinement with self-feedback","author":"Madaan","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b103","series-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","article-title":"Reflexion: language agents with verbal reinforcement learning","author":"Shinn","year":"2023"},{"key":"10.1016\/j.infsof.2026.108185_b104","series-title":"ICLR 2025 Workshop on Foundation Models in the Wild","article-title":"AgentTaxo: Dissecting and benchmarking token distribution of LLM multi-agent systems","author":"Wang","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b105","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"Cut the crap: An economical communication pipeline for LLM-based multi-agent systems","author":"Zhang","year":"2025"},{"key":"10.1016\/j.infsof.2026.108185_b106","series-title":"Software Engineering: a Practitioners Approach","author":"Pressman","year":"2019"},{"key":"10.1016\/j.infsof.2026.108185_b107","article-title":"Managing technical debt: reducing friction in software development","author":"Kruchten","year":"2019"},{"key":"10.1016\/j.infsof.2026.108185_b108","series-title":"Performance Solutions: a Practical Guide to Creating Responsive, Scalable Software","author":"Smith","year":"2002"},{"issue":"6","key":"10.1016\/j.infsof.2026.108185_b109","doi-asserted-by":"crossref","first-page":"1498","DOI":"10.1016\/j.jss.2012.12.052","article-title":"An exploration of technical debt","volume":"86","author":"Tom","year":"2013","journal-title":"J. Syst. Softw."},{"key":"10.1016\/j.infsof.2026.108185_b110","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"10122","article-title":"Optimizing language models with fair and stable reward composition in reinforcement learning","author":"Li","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b111","series-title":"The Twelfth International Conference on Learning Representations","article-title":"Confronting reward model overoptimization with constrained RLHF","author":"Moskovitz","year":"2024"},{"key":"10.1016\/j.infsof.2026.108185_b112","series-title":"A survey on code generation with LLM-based agents","author":"Dong","year":"2025"}],"container-title":["Information and Software Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950584926001746?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950584926001746?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T22:36:34Z","timestamp":1780871794000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950584926001746"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":112,"alternative-id":["S0950584926001746"],"URL":"https:\/\/doi.org\/10.1016\/j.infsof.2026.108185","relation":{},"ISSN":["0950-5849"],"issn-type":[{"value":"0950-5849","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"An overview of evaluation and enhancement methods for code generation by large language models","name":"articletitle","label":"Article Title"},{"value":"Information and Software Technology","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.infsof.2026.108185","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"108185"}}