{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T22:27:53Z","timestamp":1781562473617,"version":"3.54.5"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T00:00:00Z","timestamp":1772668800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T00:00:00Z","timestamp":1776038400000},"content-version":"vor","delay-in-days":39,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100004770","name":"Universit\u00e0 degli Studi di Parma","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004770","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Discov Artif Intell"],"DOI":"10.1007\/s44163-026-01009-5","type":"journal-article","created":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T07:15:48Z","timestamp":1772694948000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Helping LLMs improve code generation using feedback from testing and static analysis"],"prefix":"10.1007","volume":"6","author":[{"given":"Greta","family":"Dolcetti","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vincenzo","family":"Arceri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eleonora","family":"Iotti","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sergio","family":"Maffeis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Agostino","family":"Cortesi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Enea","family":"Zaffanella","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,5]]},"reference":[{"key":"1009_CR1","doi-asserted-by":"publisher","unstructured":"Zhao S, Jia M, Tuan LA, Wen J. Universal vulnerabilities in large language models: in-context learning backdoor attacks. CoRR 2024 https:\/\/doi.org\/10.48550\/ARXIV.2401.05949, arXiv:abs\/2401.05949.","DOI":"10.48550\/ARXIV.2401.05949"},{"key":"1009_CR2","doi-asserted-by":"publisher","unstructured":"Zhang B, Liang P, Zhou X, Ahmad A, Waseem M. Practices and challenges of using github copilot: An empirical study. In: Chang, S. (ed.) The 35th International Conference on Software Engineering and Knowledge Engineering, SEKE 2023, KSIR Virtual Conference Center, USA, July 1\u201310, 2023, pp. 124\u2013129. KSI Research Inc., 2023. https:\/\/doi.org\/10.18293\/SEKE2023-077.","DOI":"10.18293\/SEKE2023-077"},{"issue":"OOPSLA1","key":"1009_CR3","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1145\/3586030","volume":"7","author":"S Barke","year":"2023","unstructured":"Barke S, James MB, Polikarpova N. Grounded copilot: how programmers interact with code-generating models. Proc ACM Program Lang. 2023;7(OOPSLA1):85\u2013111. https:\/\/doi.org\/10.1145\/3586030.","journal-title":"Proc ACM Program Lang"},{"key":"1009_CR4","unstructured":"Gartner: gartner hype cycle shows ai practices and platform engineering will reach mainstream adoption in software engineering in two to five years 2024. https:\/\/www.gartner.com\/en\/newsroom\/press-releases\/2023-11-28-gartner-hype-cycle-shows-ai-practices-and-platform-engineering-will-reach-mainstream-adoption-in-software-engineering-in-two-to-five-years."},{"key":"1009_CR5","unstructured":"Team ML. https:\/\/ai.meta.com\/blog\/meta-llama-3\/ 2024. https:\/\/ai.meta.com\/blog\/meta-llama-3\/."},{"key":"1009_CR6","doi-asserted-by":"publisher","unstructured":"Mesnard T, Hardin C, Dadashi R, Bhupatiraju S, Pathak S, Sifre L, Rivi\u00e8re M, Kale MS, Love J, Tafti P, Hussenot L, Chowdhery A, Roberts A, Barua A, Botev A, Castro-Ros A, Slone A, H\u00e9liou A, Tacchetti A, Bulanova A, Paterson A, Tsai B, Shahriari B, Lan CL, Choquette-Choo CA, Crepy C, Cer D, Ippolito D, Reid D, Buchatskaya E, Ni E, Noland E, Yan G, Tucker G, Muraru G, Rozhdestvenskiy G, Michalewski H, Tenney I, Grishchenko I, Austin J, Keeling J, Labanowski J, Lespiau J, Stanway J, Brennan J, Chen J, Ferret J, Chiu J, et al. Gemma: Open models based on gemini research and technology. 2024 https:\/\/doi.org\/10.48550\/ARXIV.2403.08295, CoRR arXiv:abs\/2403.08295.","DOI":"10.48550\/ARXIV.2403.08295"},{"key":"1009_CR7","doi-asserted-by":"publisher","unstructured":"Jiang AQ, Sablayrolles A, Roux A, Mensch A, Savary B, Bamford C, Chaplot DS, Las Casas D, Hanna EB, Bressand F, Lengyel G, Bour G, Lample G, Lavaud LR, Saulnier L, Lachaux M, Stock P, Subramanian S, Yang S, Antoniak S, Scao TL, Gervet T, Lavril T, Wang T, Lacroix T, Sayed WE. Mixtral of experts. https:\/\/doi.org\/10.48550\/ARXIV.2401.04088, 2024 CoRR arXiv:abs\/2401.04088.","DOI":"10.48550\/ARXIV.2401.04088"},{"key":"1009_CR8","doi-asserted-by":"publisher","unstructured":"Calcagno C, Distefano D. Infer: An automatic program verifier for memory safety of C programs. In: Bobaru MG, Havelund K, Holzmann GJ, Joshi R (eds.) NASA Formal Methods - Third International Symposium, NFM 2011, Pasadena, CA, USA, April 18\u201320, 2011. Proceedings. Lecture Notes in Computer Science, vol. 6617, pp. 459\u2013465. Springer, 2011 https:\/\/doi.org\/10.1007\/978-3-642-20398-5_33.","DOI":"10.1007\/978-3-642-20398-5_33"},{"key":"1009_CR9","unstructured":"Austin J, Odena A, Nye MI, Bosma M, Michalewski H, Dohan D, Jiang E, Cai CJ, Terry M, Le QV, Sutton C. Program synthesis with large language models. 2021 CoRR arXiv:abs\/2108.07732."},{"key":"1009_CR10","doi-asserted-by":"crossref","unstructured":"Xu R, Cao J, Lu Y, Wen M, Lin H, Han X, He B, Cheung S-C, Sun L. CRUXEval-X: a benchmark for multilingual code reasoning, understanding and execution 2025. https:\/\/arxiv.org\/abs\/2408.13001.","DOI":"10.18653\/v1\/2025.acl-long.1158"},{"key":"1009_CR11","unstructured":"Chen M, Tworek J, Jun H, Yuan Q, Oliveira Pinto HP, Kaplan J, Edwards H, Burda Y, Joseph N, Brockman G, Ray A, Puri R, Krueger G, Petrov M, Khlaaf H, Sastry G, Mishkin P, Chan B, Gray S, Ryder N, Pavlov M, Power A, Kaiser L, Bavarian M, Winter C, Tillet P, Such FP, Cummings D, Plappert M, Chantzis F, Barnes E, Herbert-Voss A, Guss WH, Nichol A, Paino A, Tezak N, Tang J, Babuschkin I, Balaji S, Jain S, Saunders W, Hesse C, Carr AN, Leike J, Achiam J, Misra V, Morikawa E, Radford A, Knight M, Brundage M, Murati M, Mayer K, Welinder P, McGrew B, Amodei D, McCandlish S, Sutskever I, Zaremba W. Evaluating large language models trained on code 2021 arXiv:2107.03374 [cs.LG]."},{"key":"1009_CR12","doi-asserted-by":"publisher","unstructured":"Jiang AQ, Sablayrolles A, Mensch A, Bamford C, Chaplot DS, Las Casas D, Bressand F, Lengyel G, Lample G, Saulnier L, Lavaud LR, Lachaux M, Stock P, Scao TL, Lavril T, Wang T, Lacroix T, Sayed WE. Mistral 7b. 2023 https:\/\/doi.org\/10.48550\/ARXIV.2310.06825, CoRR arXiv:abs\/2310.06825.","DOI":"10.48550\/ARXIV.2310.06825"},{"key":"1009_CR13","doi-asserted-by":"publisher","unstructured":"Schulhoff S, Ilie M, Balepur N, Kahadze K, Liu A, Si C, Li Y, Gupta A, Han H, Schulhoff S, Dulepet PS, Vidyadhara S, Ki D, Agrawal S, Pham C, Kroiz GC, Li F, Tao H, Srivastava A, Costa HD, Gupta S, Rogers ML, Goncearenco I, Sarli G, Galynker I, Peskoff D, Carpuat M, White J, Anadkat S, Hoyle AM, Resnik P. The prompt report: A systematic survey of prompting techniques. https:\/\/doi.org\/10.48550\/ARXIV.2406.06608, CoRR arXiv:abs\/2406.06608 (2024).","DOI":"10.48550\/ARXIV.2406.06608"},{"key":"1009_CR14","unstructured":"Wei J, Wang X, Schuurmans D, Bosma M, Ichter B, Xia F, Chi EH, Le QV, Zhou D. Chain-of-thought prompting elicits reasoning in large language models. In: Koyejo S, Mohamed S, Agarwal A, Belgrave D, Cho K, Oh A. (eds.) Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022 2022. http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/9d5609613524ecf4f15af0f7b31abca4-Abstract-Conference.html."},{"key":"1009_CR15","unstructured":"Zhou Y, Muresanu AI, Han Z, Paster K, Pitis S, Chan H, Ba J. Large language models are human-level prompt engineers. In: The Eleventh International Conference on Learning Representations, ICLR 2023, Kigali, Rwanda, May 1\u20135, 2023. OpenReview.net, 2023. https:\/\/openreview.net\/forum?id=92gvk82DE-."},{"key":"1009_CR16","doi-asserted-by":"publisher","unstructured":"Cousot P, Cousot R. Abstract interpretation: A unified lattice model for static analysis of programs by construction or approximation of fixpoints. In: Graham RM, Harrison MA, Sethi R. (eds.) Conference Record of the Fourth ACM Symposium on Principles of Programming Languages, Los Angeles, California, USA, January 1977, pp. 238\u2013252. ACM, 1977. https:\/\/doi.org\/10.1145\/512950.512973.","DOI":"10.1145\/512950.512973"},{"key":"1009_CR17","doi-asserted-by":"publisher","unstructured":"Tihanyi N, Bisztray T, Jain R, Ferrag MA, Cordeiro LC, Mavroeidis V. The formai dataset: Generative AI in software security through the lens of formal verification. In: McIntosh S, Choi E, Herbold S. (eds.) Proceedings of the 19th International Conference on Predictive Models and Data Analytics in Software Engineering, PROMISE 2023, San Francisco, CA, USA, 8 December 2023, pp. 33\u201343. ACM, 2023. https:\/\/doi.org\/10.1145\/3617555.3617874.","DOI":"10.1145\/3617555.3617874"},{"key":"1009_CR18","doi-asserted-by":"publisher","unstructured":"Gadelha MYR, Monteiro FR, Morse J, Cordeiro LC, Fischer B, Nicole DA. ESBMC 5.0: an industrial-strength C model checker. In: Huchard M, K\u00e4stner C, Fraser G. (eds.) Proceedings of the 33rd ACM\/IEEE International Conference on Automated Software Engineering, ASE 2018, Montpellier, France, September 3\u20137, 2018, pp. 888\u2013891. ACM, 2018. https:\/\/doi.org\/10.1145\/3238147.3240481","DOI":"10.1145\/3238147.3240481"},{"issue":"2","key":"1009_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s10664-024-10590-1","volume":"30","author":"N Tihanyi","year":"2025","unstructured":"Tihanyi N, Bisztray T, Ferrag MA, Jain R, Cordeiro LC. How secure is ai-generated code: a large-scale comparison of large language models. Empir Softw Eng. 2025;30(2):1\u201342.","journal-title":"Empir Softw Eng"},{"key":"1009_CR20","doi-asserted-by":"crossref","unstructured":"Pearce H, Ahmad B, Tan B, Dolan-Gavitt B, Karri R. Asleep at the keyboard? assessing the security of github copilot\u2019s code contributions. In: IEEE Symposium on Security and Privacy, S&P 2022, pp. 754\u2013768 2022. IEEE.","DOI":"10.1109\/SP46214.2022.9833571"},{"key":"1009_CR21","doi-asserted-by":"crossref","unstructured":"Nazzal M, Khalil I, Khreishah A, Phan N. Promsec: Prompt optimization for secure generation of functional source code with large language models (llms). In: Proceedings of the 2024 on ACM SIGSAC Conference on Computer and Communications Security, pp. 2266\u20132280, 2024.","DOI":"10.1145\/3658644.3690298"},{"key":"1009_CR22","doi-asserted-by":"crossref","unstructured":"He J, Vechev M. Large language models for code: Security hardening and adversarial testing. In: Proceedings of the 2023 ACM SIGSAC Conference on Computer and Communications Security, pp. 1865\u20131879, 2023.","DOI":"10.1145\/3576915.3623175"},{"key":"1009_CR23","doi-asserted-by":"crossref","unstructured":"Li D, Yan M, Zhang Y, Liu Z, Liu C, Zhang X, Chen T, Lo D. Cosec: On-the-fly security hardening of code llms via supervised co-decoding. In: Proceedings of the 33rd ACM SIGSOFT International Symposium on Software Testing and Analysis, pp. 1428\u20131439, 2024.","DOI":"10.1145\/3650212.3680371"},{"key":"1009_CR24","doi-asserted-by":"crossref","unstructured":"Chapman PJ, Rubio-Gonz\u00e1lez C, Thakur AV. Interleaving static analysis and LLM prompting. In: Proceedings of the 13th ACM SIGPLAN International Workshop on the State Of the Art in Program Analysis, SOAP 2024, pp. 9\u201317, 2024.","DOI":"10.1145\/3652588.3663317"},{"key":"1009_CR25","unstructured":"Li Z, Dutta S, Naik M. LLM-assisted static analysis for detecting security vulnerabilities. arXiv preprint, 2024 arXiv:2405.17238."},{"key":"1009_CR26","doi-asserted-by":"crossref","unstructured":"Ullah S, Han M, Pujar S, Pearce H, Coskun A, Stringhini G. LLMs cannot reliably identify and reason about security vulnerabilities (yet?): A comprehensive evaluation, framework, and benchmarks. In: IEEE Symposium on Security and Privacy, S&P 2024, 2024.","DOI":"10.1109\/SP54263.2024.00210"},{"key":"1009_CR27","doi-asserted-by":"publisher","DOI":"10.1016\/j.jss.2024.112031","volume":"212","author":"G Lu","year":"2024","unstructured":"Lu G, Ju X, Chen X, Pei W, Cai Z. Grace: empowering llm-based software vulnerability detection with graph structure and in-context learning. J Syst Softw. 2024;212:112031.","journal-title":"J Syst Softw"},{"key":"1009_CR28","doi-asserted-by":"crossref","unstructured":"Wen X-C, Gao C, Gao S, Xiao Y, Lyu MR. Scale: Constructing structured natural language comment trees for software vulnerability detection. In: Proceedings of the 33rd ACM SIGSOFT International Symposium on Software Testing and Analysis, pp. 235\u2013247, 2024.","DOI":"10.1145\/3650212.3652124"},{"key":"1009_CR29","doi-asserted-by":"publisher","unstructured":"Charalambous Y, Tihanyi N, Jain R, Sun Y, Ferrag MA, Cordeiro LC. A new era in software security: Towards self-healing software via large language models and formal verification, 2023. https:\/\/doi.org\/10.48550\/ARXIV.2305.14752, CoRR arXiv:abs\/2305.14752.","DOI":"10.48550\/ARXIV.2305.14752"},{"key":"1009_CR30","doi-asserted-by":"publisher","unstructured":"Jin M, Shahriar S, Tufano M, Shi X, Lu S, Sundaresan N, Svyatkovskiy A. Inferfix: End-to-end program repair with llms. In: Chandra S, Blincoe K, Tonella P. (eds.) Proceedings of the 31st ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering, ESEC\/FSE 2023, San Francisco, CA, USA, December 3\u20139, 2023, pp. 1646\u20131656. ACM, 2023. https:\/\/doi.org\/10.1145\/3611643.3613892.","DOI":"10.1145\/3611643.3613892"},{"key":"1009_CR31","doi-asserted-by":"publisher","unstructured":"Jan\u00dfen C, Richter C, Wehrheim H. Can chatgpt support software verification?, 2023. https:\/\/doi.org\/10.48550\/ARXIV.2311.02433, CoRR arXiv:abs\/2311.02433.","DOI":"10.48550\/ARXIV.2311.02433"},{"key":"1009_CR32","doi-asserted-by":"publisher","unstructured":"Li H, Hao Y, Zhai Y, Qian Z. Assisting static analysis with large language models: A chatgpt experiment. In: Chandra S, Blincoe K, Tonella P. (eds.) Proceedings of the 31st ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering, ESEC\/FSE 2023, San Francisco, CA, USA, December 3\u20139, 2023, pp. 2107\u20132111. ACM, 2023. https:\/\/doi.org\/10.1145\/3611643.3613078.","DOI":"10.1145\/3611643.3613078"},{"key":"1009_CR33","unstructured":"Jain N, Han K, Gu A, Li W-D, Yan F, Zhang T, Wang S, Solar-Lezama A, Sen K, Stoica I: Livecodebench: Holistic and contamination free evaluation of large language models for code, 2024. arXiv preprint arXiv:2403.07974."},{"key":"1009_CR34","unstructured":"Olausson TX, Inala JP, Wang C, Gao J, Solar-Lezama A. Is self-repair a silver bullet for code generation? In: The Twelfth International Conference on Learning Representations, 2023."},{"key":"1009_CR35","unstructured":"Islam NT, Khoury J, Seong A, Bou-Harb E, Najafirad P. Enhancing source code security with llms: demystifying the challenges and generating reliable repairs. In: Network and Distributed System Security (NDSS) Symposium 2024, 2024."},{"key":"1009_CR36","unstructured":"OpenAI: GPT-3.5-turbo, 2022. Accessed March 2024 . https:\/\/platform.openai.com\/docs\/models\/gpt-3-5-turbo."},{"key":"1009_CR37","doi-asserted-by":"publisher","unstructured":"Bhatt M, Chennabasappa S, Nikolaidis C. Wan S, Evtimov I, Gabi D, Song D, Ahmad F, Aschermann C, Fontana L, Frolov S, Giri RP, Kapil D, Kozyrakis Y, LeBlanc D, Milazzo J, Straumann A, Synnaeve G, Vontimitta V, Whitman S, Saxe J. Purple llama cyberseceval: A secure coding benchmark for language models. 2023. https:\/\/doi.org\/10.48550\/ARXIV.2312.04724, CoRR arXiv:abs\/2312.04724.","DOI":"10.48550\/ARXIV.2312.04724"},{"key":"1009_CR38","unstructured":"He J, Vero M, Krasnopolska G, Vechev MT. Instruction tuning for secure code generation. In: Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21\u201327, 2024. OpenReview.net, 2024. https:\/\/openreview.net\/forum?id=MgTzMaYHvG."},{"key":"1009_CR39","doi-asserted-by":"crossref","unstructured":"Pearce H, Tan B, Ahmad B, Karri R, Dolan-Gavitt B. Examining zero-shot vulnerability repair with large language models. In: IEEE Symposium on Security and Privacy, S&P 2023, pp. 2339\u20132356, 2023. IEEE.","DOI":"10.1109\/SP46215.2023.10179324"},{"key":"1009_CR40","doi-asserted-by":"publisher","unstructured":"Plein L, Ou\u00e9draogo WC, Klein J, Bissyand\u00e9 TF. Automatic generation of test cases based on bug reports: a feasibility study with large language models. In: Proceedings of the 2024 IEEE\/ACM 46th International Conference on Software Engineering: Companion Proceedings, ICSE Companion 2024, Lisbon, Portugal, April 14\u201320, 2024, pp. 360\u2013361. ACM, 2024. https:\/\/doi.org\/10.1145\/3639478.3643119..","DOI":"10.1145\/3639478.3643119."}],"container-title":["Discover Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s44163-026-01009-5","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44163-026-01009-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44163-026-01009-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T10:45:15Z","timestamp":1776077115000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s44163-026-01009-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,5]]},"references-count":40,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,12]]}},"alternative-id":["1009"],"URL":"https:\/\/doi.org\/10.1007\/s44163-026-01009-5","relation":{},"ISSN":["2731-0809"],"issn-type":[{"value":"2731-0809","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,5]]},"assertion":[{"value":"23 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to publish"}}],"article-number":"314"}}