{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T11:51:54Z","timestamp":1781610714134,"version":"3.54.5"},"reference-count":76,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T00:00:00Z","timestamp":1774656000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T00:00:00Z","timestamp":1774656000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100010663","name":"H2020 European Research Council","doi-asserted-by":"publisher","award":["949014"],"award-info":[{"award-number":["949014"]}],"id":[{"id":"10.13039\/100010663","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001866","name":"Fonds National de la Recherche Luxembourg","doi-asserted-by":"publisher","award":["17185670"],"award-info":[{"award-number":["17185670"]}],"id":[{"id":"10.13039\/501100001866","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Empir Software Eng"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s10664-026-10840-4","type":"journal-article","created":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T08:34:45Z","timestamp":1774686885000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Prompt engineering in LLMs for automated unit test generation: A large-scale study"],"prefix":"10.1007","volume":"31","author":[{"given":"Wendk\u00fbuni C.","family":"Ou\u00e9draogo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abdoul Kader","family":"Kabor\u00e9","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1390-0393","authenticated-orcid":false,"given":"Yinghua","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haoye","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Anil","family":"Koyuncu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jacques","family":"Klein","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"David","family":"Lo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tegawend\u00e9 F.","family":"Bissyand\u00e9","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,28]]},"reference":[{"issue":"116","key":"10840_CR1","first-page":"30","volume":"1","author":"AM Abdullin","year":"2022","unstructured":"Abdullin AM, Itsykson VM (2022) Kex: A platform for analysis of jvm programs. Inform Contr Syst 1(116):30\u201343","journal-title":"Inform Contr Syst"},{"key":"10840_CR2","unstructured":"Achiam J, Adler S, Agarwal S, Ahmad L, Akkaya I, Aleman FL, Almeida D, Altenschmidt J, Altman S, Anadkat S et al (2023) Gpt-4 technical report. arXiv:2303.08774"},{"key":"10840_CR3","doi-asserted-by":"crossref","unstructured":"Almasi MM, Hemmati H, Fraser G, Arcuri A, Benefelds J (2017) An industrial evaluation of unit test generation: Finding real faults in a financial application. In: 2017 IEEE\/ACM 39th International Conference on Software Engineering: Software Engineering in Practice Track (ICSE-SEIP), IEEE, pp 263\u2013272","DOI":"10.1109\/ICSE-SEIP.2017.27"},{"key":"10840_CR4","doi-asserted-by":"crossref","unstructured":"Alshahwan N, Chheda J, Finogenova A, Gokkaya B, Harman M, Harper I, Marginean A, Sengupta S, Wang E (2024) Automated unit test improvement using large language models at meta. In: Companion Proceedings of the 32nd ACM International Conference on the Foundations of Software Engineering, pp 185\u2013196","DOI":"10.1145\/3663529.3663839"},{"key":"10840_CR5","unstructured":"Amatriain X (2024) Prompt design and engineering: Introduction and advanced methods. arXiv preprint arXiv:2401.14423"},{"key":"10840_CR6","doi-asserted-by":"publisher","first-page":"594","DOI":"10.1007\/s10664-013-9249-9","volume":"18","author":"A Arcuri","year":"2013","unstructured":"Arcuri A, Fraser G (2013) Parameter tuning or default values? an empirical investigation in search-based software engineering. Empir Softw Eng 18:594\u2013623","journal-title":"Empir Softw Eng"},{"key":"10840_CR7","doi-asserted-by":"crossref","unstructured":"Arcuri A, Fraser G, Just R (2017) Private api access and functional mocking in automated unit test generation. In: 2017 IEEE international conference on software testing, verification and validation (ICST), IEEE, pp 126\u2013137","DOI":"10.1109\/ICST.2017.19"},{"key":"10840_CR8","unstructured":"Beck K (2000) Extreme programming explained: embrace change. addison-wesley professional"},{"key":"10840_CR9","first-page":"162","volume-title":"2025 IEEE Conference on Software Testing","author":"M Biagiola","year":"2025","unstructured":"Biagiola M, Ghislotti G, Tonella P (2025) Improving the readability of automatically generated tests using large language models. 2025 IEEE Conference on Software Testing. Verification and Validation (ICST), IEEE, pp 162\u2013173"},{"key":"10840_CR10","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown T, Mann B, Ryder N, Subbiah M, Kaplan JD, Dhariwal P, Neelakantan A, Shyam P, Sastry G, Askell A et al (2020) Language models are few-shot learners. Adv Neural Inf Process Syst 33:1877\u20131901","journal-title":"Adv Neural Inf Process Syst"},{"issue":"4","key":"10840_CR11","doi-asserted-by":"publisher","first-page":"546","DOI":"10.1109\/TSE.2009.70","volume":"36","author":"RP Buse","year":"2009","unstructured":"Buse RP, Weimer WR (2009) Learning a metric for code readability. IEEE Trans Software Eng 36(4):546\u2013558","journal-title":"IEEE Trans Software Eng"},{"key":"10840_CR12","doi-asserted-by":"crossref","unstructured":"Chen Y, Hu Z, Zhi C, Han J, Deng S, Yin J (2023) Chatunitest: A framework for llm-based test generation. arXiv e-prints pp arXiv-2305","DOI":"10.1145\/3663529.3663801"},{"key":"10840_CR13","doi-asserted-by":"crossref","unstructured":"Chen Y, Hu Z, Zhi C, Han J, Deng S, Yin J (2024) Chatunitest: A framework for llm-based test generation. In: Companion Proceedings of the 32nd ACM International Conference on the Foundations of Software Engineering, pp 572\u2013576","DOI":"10.1145\/3663529.3663801"},{"key":"10840_CR14","doi-asserted-by":"crossref","unstructured":"Daka E, Fraser G (2014) A survey on unit testing practices and problems. In: 2014 IEEE 25th International Symposium on Software Reliability Engineering, IEEE, pp 201\u2013211","DOI":"10.1109\/ISSRE.2014.11"},{"key":"10840_CR15","doi-asserted-by":"crossref","unstructured":"Daka E, Campos J, Fraser G, Dorn J, Weimer W (2015) Modeling readability to improve unit tests. In: Proceedings of the 2015 10th Joint Meeting on Foundations of Software Engineering, pp 107\u2013118","DOI":"10.1145\/2786805.2786838"},{"key":"10840_CR16","doi-asserted-by":"publisher","first-page":"107468","DOI":"10.1016\/j.infsof.2024.107468","volume":"171","author":"AM Dakhel","year":"2024","unstructured":"Dakhel AM, Nikanjam A, Majdinasab V, Khomh F, Desmarais MC (2024) Effective test generation using pre-trained large language models and mutation testing. Inf Softw Technol 171:107468","journal-title":"Inf Softw Technol"},{"key":"10840_CR17","unstructured":"Dantas CEC, Maia MA (2021) Readability and understandability scores for snippet assessment: An exploratory study. arXiv preprint arXiv:2108.09181"},{"key":"10840_CR18","doi-asserted-by":"crossref","unstructured":"Deljouyi A, Koohestani R, Izadi M, Zaidman A (2024) Leveraging large language models for enhancing the understandability of generated unit tests. arXiv preprint arXiv:2408.11710","DOI":"10.1109\/ICSE55347.2025.00032"},{"key":"10840_CR19","doi-asserted-by":"crossref","unstructured":"Deljouyi A, Koohestani R, Izadi M, Zaidman A (2025) Leveraging large language models for enhancing the understandability of generated unit tests. In: 2025 ieee\/acm 47th international conference on software engineering (icse). Los Alamitos, CA, USA pp 392\u2013404","DOI":"10.1109\/ICSE55347.2025.00032"},{"key":"10840_CR20","unstructured":"Dorn J (2012) A general software readability model. MCS Thesis available from (http:\/\/www.cs.virginia.edu\/~weimer\/students\/dorn-mcs-paper.pdf) 5:11\u201314"},{"key":"10840_CR21","doi-asserted-by":"crossref","unstructured":"Fan A, Gokkaya B, Harman M, Lyubarskiy M, Sengupta S, Yoo S, Zhang JM (2023) Large language models for software engineering: Survey and open problems. arXiv preprint arXiv:2310.03533","DOI":"10.1109\/ICSE-FoSE59343.2023.00008"},{"key":"10840_CR22","doi-asserted-by":"crossref","unstructured":"Fraser G, Arcuri A (2011) Evosuite: automatic test suite generation for object-oriented software. In: Proceedings of the 19th ACM SIGSOFT symposium and the 13th European conference on Foundations of software engineering, pp 416\u2013419","DOI":"10.1145\/2025113.2025179"},{"key":"10840_CR23","doi-asserted-by":"crossref","unstructured":"Fraser G, Arcuri A (2013) Evosuite: On the challenges of test case generation in the real world. In: 2013 IEEE sixth international conference on software testing, verification and validation, IEEE, pp 362\u2013369","DOI":"10.1109\/ICST.2013.51"},{"issue":"4","key":"10840_CR24","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2699688","volume":"24","author":"G Fraser","year":"2015","unstructured":"Fraser G, Staats M, McMinn P, Arcuri A, Padberg F (2015) Does automated unit test generation really help software testers? a controlled empirical study. ACM Trans Softw Eng Methodol (TOSEM) 24(4):1\u201349","journal-title":"ACM Trans Softw Eng Methodol (TOSEM)"},{"key":"10840_CR25","doi-asserted-by":"crossref","unstructured":"Grano G, Scalabrino S, Gall HC, Oliveto R (2018) An empirical investigation on the readability of manual and generated test cases. In: Proceedings of the 26th Conference on Program Comprehension, pp 348\u2013351","DOI":"10.1145\/3196321.3196363"},{"key":"10840_CR26","unstructured":"Gu S, Fang C, Zhang Q, Tian F, Chen Z (2024) Testart: Improving llm-based unit test via co-evolution of automated generation and repair iteration. arXiv e-prints pp arXiv-2408"},{"key":"10840_CR27","unstructured":"Hossain SB, Dwyer M (2024) Togll: Correct and strong test oracle generation with llms. arXiv preprint arXiv:2405.03786"},{"key":"10840_CR28","doi-asserted-by":"crossref","unstructured":"Jahangirova G, Terragni V (2023) Sbft tool competition 2023-java test case generation track. In: 2023 IEEE\/ACM International Workshop on Search-Based and Fuzz Testing (SBFT), IEEE, pp 61\u201364","DOI":"10.1109\/SBFT59156.2023.00025"},{"key":"10840_CR29","unstructured":"Jiang AQ, Sablayrolles A, Mensch A, Bamford C, Chaplot DS, Casas Ddl, Bressand F, Lengyel G, Lample G, Saulnier L et al (2023) Mistral 7b. arXiv preprint arXiv:2310.06825"},{"key":"10840_CR30","unstructured":"Jiang AQ, Sablayrolles A, Roux A, Mensch A, Savary B, Bamford C, Chaplot DS, Casas Ddl, Hanna EB, Bressand F et al (2024) Mixtral of experts. arXiv preprint arXiv:2401.04088"},{"key":"10840_CR31","first-page":"22199","volume":"35","author":"T Kojima","year":"2022","unstructured":"Kojima T, Gu SS, Reid M, Matsuo Y, Iwasawa Y (2022) Large language models are zero-shot reasoners. Adv Neural Inf Process Syst 35:22199\u201322213","journal-title":"Adv Neural Inf Process Syst"},{"key":"10840_CR32","unstructured":"Kumar A, Haiduc S, Das PP, Chakrabarti PP (2024) Llms as evaluators: A novel approach to evaluate bug report summarization. arXiv preprint arXiv:2409.00630"},{"issue":"2","key":"10840_CR33","first-page":"1","volume":"34","author":"J Li","year":"2025","unstructured":"Li J, Li G, Li Y, Jin Z (2025) Structured chain-of-thought prompting for code generation. ACM Trans Softw Eng Methodol 34(2):1\u201323","journal-title":"ACM Trans Softw Eng Methodol"},{"key":"10840_CR34","unstructured":"Long J (2023) Large language model guided tree-of-thought. arXiv preprint arXiv:2305.08291"},{"key":"10840_CR35","doi-asserted-by":"crossref","unstructured":"Macedo M, Tian Y, Cogo FR, Adams B (2024) Exploring the impact of the output format on the evaluation of large language models for code translation. arXiv preprint arXiv:2403.17214","DOI":"10.1145\/3650105.3652301"},{"key":"10840_CR36","doi-asserted-by":"publisher","first-page":"111454","DOI":"10.1016\/j.jss.2022.111454","volume":"193","author":"Q Mi","year":"2022","unstructured":"Mi Q, Hao Y, Ou L, Ma W (2022) Towards using visual, semantic and structural features to improve code readability classification. J Syst Softw 193:111454","journal-title":"J Syst Softw"},{"issue":"5","key":"10840_CR37","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3715107","volume":"34","author":"F Molina","year":"2025","unstructured":"Molina F, Gorla A, d\u2019Amorim M (2025) Test oracle automation in the era of llms. ACM Trans Softw Eng Methodol 34(5):1\u201324","journal-title":"ACM Trans Softw Eng Methodol"},{"key":"10840_CR38","unstructured":"Naveed H, Khan AU, Qiu S, Saqib M, Anwar S, Usman M, Barnes N, Mian A (2023) A comprehensive overview of large language models. arXiv preprint arXiv:2307.06435"},{"key":"10840_CR39","doi-asserted-by":"crossref","unstructured":"Oliveira D, Bruno R, Madeiral F, Masuhara H, Castor F (2022) A systematic literature review on the impact of formatting elements on program understandability. Available at SSRN 4182156","DOI":"10.2139\/ssrn.4182156"},{"key":"10840_CR40","unstructured":"OpenAI (2023) Gpt-3.5-turbo. https:\/\/platformopenai.com\/docs\/models\/gpt-3-5-turbo\/. Accessed 21 May 2024"},{"issue":"3","key":"10840_CR41","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1007\/s10664-025-10635-z","volume":"30","author":"WC Ou\u00e9draogo","year":"2025","unstructured":"Ou\u00e9draogo WC, Plein L, Kabore K, Habib A, Klein J, Lo D, Bissyand\u00e9 TF (2025) Enriching automatic test case generation by extracting relevant test inputs from bug reports. Empir Softw Eng 30(3):85","journal-title":"Empir Softw Eng"},{"key":"10840_CR42","doi-asserted-by":"crossref","unstructured":"Pacheco C, Ernst MD (2007) Randoop: feedback-directed random testing for java. In: Companion to the 22nd ACM SIGPLAN conference on Object-oriented programming systems and applications companion, pp 815\u2013816","DOI":"10.1145\/1297846.1297902"},{"key":"10840_CR43","doi-asserted-by":"crossref","unstructured":"Palomba F, Di Nucci D, Panichella A, Oliveto R, De Lucia A (2016) On the diffusion of test smells in automatically generated test code: An empirical study. In: Proceedings of the 9th international workshop on search-based software testing, pp 5\u201314","DOI":"10.1145\/2897010.2897016"},{"issue":"2","key":"10840_CR44","doi-asserted-by":"publisher","first-page":"122","DOI":"10.1109\/TSE.2017.2663435","volume":"44","author":"A Panichella","year":"2017","unstructured":"Panichella A, Kifetew FM, Tonella P (2017) Automated test case generation as a many-objective optimisation problem with dynamic selection of the targets. IEEE Trans Software Eng 44(2):122\u2013158","journal-title":"IEEE Trans Software Eng"},{"key":"10840_CR45","doi-asserted-by":"crossref","unstructured":"Panichella A, Panichella S, Fraser G, Sawant AA, Hellendoorn VJ (2020) Revisiting test smells in automatically generated tests: limitations, pitfalls, and opportunities. In: 2020 IEEE international conference on software maintenance and evolution (ICSME), IEEE, pp 523\u2013533","DOI":"10.1109\/ICSME46990.2020.00056"},{"issue":"7","key":"10840_CR46","doi-asserted-by":"publisher","first-page":"170","DOI":"10.1007\/s10664-022-10207-5","volume":"27","author":"A Panichella","year":"2022","unstructured":"Panichella A, Panichella S, Fraser G, Sawant AA, Hellendoorn VJ (2022) Test smells 20 years later: detectability, validity, and reliability. Empir Softw Eng 27(7):170","journal-title":"Empir Softw Eng"},{"key":"10840_CR47","unstructured":"Peruma A, Almalki KS, Newman CD, Mkaouer MW, Ouni A, Palomba F (2019) On the distribution of test smells in open source android applications: An exploratory study. (2019). Citado na p 13"},{"key":"10840_CR48","doi-asserted-by":"crossref","unstructured":"Peruma A, Almalki K, Newman CD, Mkaouer MW, Ouni A, Palomba F (2020) Tsdetect: An open source test smells detection tool. In: Proceedings of the 28th ACM joint meeting on european software engineering conference and symposium on the foundations of software engineering, pp 1650\u20131654","DOI":"10.1145\/3368089.3417921"},{"key":"10840_CR49","doi-asserted-by":"crossref","unstructured":"Pinto GH, Vergilio SR (2010) A multi-objective genetic algorithm to test data generation. In: 2010 22nd IEEE International Conference on Tools with Artificial Intelligence, IEEE, vol 1, pp 129\u2013134","DOI":"10.1109\/ICTAI.2010.26"},{"key":"10840_CR50","doi-asserted-by":"crossref","unstructured":"Posnett D, Hindle A, Devanbu P (2011) A simpler model of software readability. In: Proceedings of the 8th working conference on mining software repositories, pp 73\u201382","DOI":"10.1145\/1985441.1985454"},{"key":"10840_CR51","doi-asserted-by":"crossref","unstructured":"Rojas JM, Fraser G, Arcuri A (2015) Automated unit test generation during software development: A controlled experiment and think-aloud observations. In: Proceedings of the 2015 international symposium on software testing and analysis, pp 338\u2013349","DOI":"10.1145\/2771783.2771801"},{"key":"10840_CR52","unstructured":"Sahoo P, Singh AK, Saha S, Jain V, Mondal S, Chadha A (2024) A systematic survey of prompt engineering in large language models: Techniques and applications. arXiv preprint arXiv:2402.07927"},{"key":"10840_CR53","unstructured":"Sallou J, Durieux T, Panichella A (2023) Breaking the silence: the threats of using llms in software engineering. arXiv preprint arXiv:2312.08055"},{"issue":"6","key":"10840_CR54","doi-asserted-by":"publisher","first-page":"e1958","DOI":"10.1002\/smr.1958","volume":"30","author":"S Scalabrino","year":"2018","unstructured":"Scalabrino S, Linares-V\u00e1squez M, Oliveto R, Poshyvanyk D (2018) A comprehensive model for code readability. J Softw Evol Process 30(6):e1958","journal-title":"J Softw Evol Process"},{"key":"10840_CR55","doi-asserted-by":"crossref","unstructured":"Sch\u00e4fer M, Nadi S, Eghbali A, Tip F (2023) An empirical evaluation of using large language models for automated unit test generation. IEEE Trans Softw Eng","DOI":"10.1109\/TSE.2023.3334955"},{"issue":"5","key":"10840_CR56","doi-asserted-by":"publisher","first-page":"263","DOI":"10.1145\/1095430.1081750","volume":"30","author":"K Sen","year":"2005","unstructured":"Sen K, Marinov D, Agha G (2005) Cute: A concolic unit testing engine for c. ACM SIGSOFT Softw Eng Notes 30(5):263\u2013272","journal-title":"ACM SIGSOFT Softw Eng Notes"},{"key":"10840_CR57","doi-asserted-by":"crossref","unstructured":"Sergeyuk A, Lvova O, Titov S, Serova A, Bagirov F, Kirillova E, Bryksin T (2024) Reassessing java code readability models with a human-centered approach. In: Proceedings of the 32nd IEEE\/ACM International Conference on Program Comprehension, pp 225\u2013235","DOI":"10.1145\/3643916.3644435"},{"key":"10840_CR58","doi-asserted-by":"crossref","unstructured":"Shamshiri S, Just R, Rojas JM, Fraser G, McMinn P, Arcuri A (2015) Do automatically generated unit tests find real faults? an empirical study of effectiveness and challenges (t). In: 2015 30th IEEE\/ACM International Conference on Automated Software Engineering (ASE), IEEE, pp 201\u2013211","DOI":"10.1109\/ASE.2015.86"},{"key":"10840_CR59","doi-asserted-by":"crossref","unstructured":"Shamshiri S, Rojas JM, Galeotti JP, Walkinshaw N, Fraser G (2018) How do automatically generated unit tests influence software maintenance? In: 2018 IEEE 11th international conference on software testing, verification and validation (ICST), IEEE, pp 250\u2013261","DOI":"10.1109\/ICST.2018.00033"},{"key":"10840_CR60","unstructured":"Si C, Gan Z, Yang Z, Wang S, Wang J, Boyd-Graber J, Wang L (2022) Prompting gpt-3 to be reliable. arXiv preprint arXiv:2210.09150"},{"key":"10840_CR61","doi-asserted-by":"crossref","unstructured":"Siddiq ML, Da Silva Santos JC, Tanvir RH, Ulfat N, Al Rifat F, Carvalho Lopes V (2024a) Using large language models to generate junit tests: An empirical study. In: Proceedings of the 28th International Conference on Evaluation and Assessment in Software Engineering, pp 313\u2013322","DOI":"10.1145\/3661167.3661216"},{"key":"10840_CR62","unstructured":"Siddiq ML, Dristi S, Saha J, Santos J (2024b) Quality assessment of prompts used in code generation. arXiv preprint arXiv:2404.10155"},{"key":"10840_CR63","doi-asserted-by":"crossref","unstructured":"Sun W, Miao Y, Li Y, Zhang H, Fang C, Liu Y, Deng G, Liu Y, Chen Z (2024) Source code summarization in the era of large language models. arXiv preprint arXiv:2407.07959","DOI":"10.1109\/ICSE55347.2025.00034"},{"key":"10840_CR64","doi-asserted-by":"crossref","unstructured":"Tang Y, Liu Z, Zhou Z, Luo X (2024) Chatgpt vs sbst: A comparative assessment of unit test suite generation. IEEE Trans Softw Eng","DOI":"10.1109\/TSE.2024.3382365"},{"key":"10840_CR65","doi-asserted-by":"crossref","unstructured":"Wang J, Huang Y, Chen C, Liu Z, Wang S, Wang Q (2024a) Software testing with large language models: Survey, landscape, and vision. IEEE Trans Softw Eng","DOI":"10.1109\/TSE.2024.3368208"},{"key":"10840_CR66","unstructured":"Wang T, Zhou N, Chen Z (2024b) Enhancing computer programming education with llms: A study on effective prompt engineering for python code generation. arXiv preprint arXiv:2407.05437"},{"key":"10840_CR67","unstructured":"Wang X, Wei J, Schuurmans D, Le Q, Chi E, Narang S, Chowdhery A, Zhou D (2022) Self-consistency improves chain of thought reasoning in language models. arXiv preprint arXiv:2203.11171"},{"key":"10840_CR68","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei J, Wang X, Schuurmans D, Bosma M, Xia F, Chi E, Le QV, Zhou D et al (2022) Chain-of-thought prompting elicits reasoning in large language models. Adv Neural Inf Process Syst 35:24824\u201324837","journal-title":"Adv Neural Inf Process Syst"},{"issue":"2","key":"10840_CR69","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1007\/s10664-023-10390-z","volume":"29","author":"D Winkler","year":"2024","unstructured":"Winkler D, Urbanke P, Ramler R (2024) Investigating the readability of test code. Empir Softw Eng 29(2):53","journal-title":"Empir Softw Eng"},{"key":"10840_CR70","doi-asserted-by":"crossref","unstructured":"Yang L, Yang C, Gao S, Wang W, Wang B, Zhu Q, Chu X, Zhou J, Liang G, Wang Q et al (2024) On the evaluation of large language models in unit test generation. In: Proceedings of the 39th IEEE\/ACM International Conference on Automated Software Engineering, pp 1607\u20131619","DOI":"10.1145\/3691620.3695529"},{"key":"10840_CR71","doi-asserted-by":"crossref","unstructured":"Yuan Z, Lou Y, Liu M, Ding S, Wang K, Chen Y, Peng X (2023) No more manual tests? evaluating and improving chatgpt for unit test generation. arXiv preprint arXiv:2305.04207","DOI":"10.1145\/3660783"},{"key":"10840_CR72","doi-asserted-by":"crossref","unstructured":"Yuan Z, Liu M, Ding S, Wang K, Chen Y, Peng X, Lou Y (2024) Evaluating and improving chatgpt for unit test generation. Proceed ACM Softw Eng 1(FSE):1703\u20131726","DOI":"10.1145\/3660783"},{"key":"10840_CR73","unstructured":"Zeller A, Gopinath R, B\u00f6hme M, Fraser G, Holler C (2019) The fuzzing book"},{"issue":"6","key":"10840_CR74","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1007\/s10664-024-10553-6","volume":"29","author":"X Zhang","year":"2024","unstructured":"Zhang X, Hou X, Qiao X, Song W (2024a) A review of automatic source code summarization. Empir Softw Eng 29(6):162","journal-title":"Empir Softw Eng"},{"key":"10840_CR75","unstructured":"Zhang Y, Lu Q, Liu K, Dou W, Zhu J, Qian L, Zhang C, Lin Z, Wei J (2025) Citywalk: Enhancing llm-based c++ unit test generation via project-dependency awareness and language-specific knowledge. arXiv preprint arXiv:2501.16155"},{"key":"10840_CR76","unstructured":"Zhang Z, Wang Y, Wang C, Chen J, Zheng Z (2024b) Llm hallucinations in practical code generation: Phenomena, mechanism, and mitigation. arXiv preprint arXiv:2409.20550"}],"container-title":["Empirical Software Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10840-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10664-026-10840-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10840-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T11:01:04Z","timestamp":1781607664000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10664-026-10840-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,28]]},"references-count":76,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["10840"],"URL":"https:\/\/doi.org\/10.1007\/s10664-026-10840-4","relation":{},"ISSN":["1382-3256","1573-7616"],"issn-type":[{"value":"1382-3256","type":"print"},{"value":"1573-7616","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,28]]},"assertion":[{"value":"31 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"The authors declare that they have no conflict of interest.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed Consent"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Clinical Trial Number"}}],"article-number":"103"}}