{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T19:47:36Z","timestamp":1782157656718,"version":"3.54.5"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T00:00:00Z","timestamp":1782086400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T00:00:00Z","timestamp":1782086400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Fondation UCA"},{"name":"MIAI Cluster, France 2030"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Software Qual J"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1007\/s11219-026-09767-2","type":"journal-article","created":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T18:50:24Z","timestamp":1782154224000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Reliable execution of natural language test cases for GUI applications using LLM agents"],"prefix":"10.1007","volume":"34","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2432-9097","authenticated-orcid":false,"given":"S\u00e9bastien","family":"Salva","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,22]]},"reference":[{"issue":"6","key":"9767_CR1","doi-asserted-by":"publisher","first-page":"2937","DOI":"10.1007\/s10664-016-9497-6","volume":"22","author":"E Al\u00e9groth","year":"2017","unstructured":"Al\u00e9groth, E., & Feldt, R. (2017). On the long-term use of visual gui testing in industrial practice: a case study. Empirical Software Engineering, 22(6), 2937\u20132971. https:\/\/doi.org\/10.1007\/s10664-016-9497-6","journal-title":"Empirical Software Engineering"},{"key":"9767_CR2","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1145\/3212695","volume":"4","author":"M Allamanis","year":"2018","unstructured":"Allamanis, M., Barr, E. T., Devanbu, P., & Sutton, C. (2018). A survey of machine learning for big code and naturalness. ACM Computing Surveys, 4, 51. https:\/\/doi.org\/10.1145\/3212695","journal-title":"ACM Computing Surveys"},{"key":"9767_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.scico.2019.102377","volume":"189","author":"F Arruda","year":"2020","unstructured":"Arruda, F., Barros, F., & Sampaio, A. (2020). Automation and consistency analysis of test cases written in natural language: An industrial context. Science of Computer Programming, 189, Article 102377. https:\/\/doi.org\/10.1016\/j.scico.2019.102377","journal-title":"Science of Computer Programming"},{"key":"9767_CR4","doi-asserted-by":"publisher","unstructured":"Augusto, C., Mor\u00e1n, J., Bertolino, A., Riva, C., & Tuya, J. (2024). Software system testing assisted by large language models: An exploratory study. In Testing Software and Systems: 36th IFIP WG 6.1 International Conference, ICTSS 2024, London, UK, October 30 \u2013 November 1, 2024, Proceedings, pp. 239\u2013255. Springer, Berlin, Heidelberg. https:\/\/doi.org\/10.1007\/978-3-031-80889-0_17.","DOI":"10.1007\/978-3-031-80889-0_17"},{"key":"9767_CR5","unstructured":"Austin, J., Odena, A., Nye, M., Bosma, M., Michalewski, H., Dohan, D., Jiang, E., Cai, C., Terry, M., Le, Q., & Sutton, C. (2021). Program Synthesis with Large Language Models."},{"key":"9767_CR6","doi-asserted-by":"publisher","unstructured":"Ayenew, H., & Wagaw, M. (2024). Software test case generation using natural language processing (nlp): A systematic literature review. Artificial Intelligence Evolution, 5(1), 1\u201310.https:\/\/doi.org\/10.37256\/aie.5120243220. [Online; cited 2026-01-07].","DOI":"10.37256\/aie.5120243220"},{"key":"9767_CR7","doi-asserted-by":"publisher","unstructured":"Carvalho, G., Falc\u00e3o, D., Barros, F., Sampaio, A., Mota, A., Motta, L., & Blackburn, M. (2014). Nat2testscr: Test case generation from natural language requirements based on scr specifications. Science of Computer Programming,\u00a095(275\u2013297), 2013. https:\/\/doi.org\/10.1016\/j.scico.2014.06.007. Special Section: ACM SAC-SVT 2013 + Bytecode","DOI":"10.1016\/j.scico.2014.06.007"},{"issue":"09","key":"9767_CR8","doi-asserted-by":"publisher","first-page":"1943","DOI":"10.1109\/TSE.2019.2940179","volume":"47","author":"Z Chen","year":"2021","unstructured":"Chen, Z., Kommrusch, S., Tufano, M., Pouchet, L., Poshyvanyk, D., & Monperrus, M. (2021). Sequencer: Sequence-to-sequence learning for end-to-end program repair. IEEE Transactions on Software Engineering, 47(09), 1943\u20131959. https:\/\/doi.org\/10.1109\/TSE.2019.2940179","journal-title":"IEEE Transactions on Software Engineering"},{"key":"9767_CR9","doi-asserted-by":"publisher","unstructured":"Chen, Y., Hu, Z., Zhi, C., Han, J., Deng, S., & Yin, J. (2024). Chatunitest: A framework for llm-based test generation. In Companion Proceedings of the 32nd ACM International Conference on the Foundations of Software Engineering. FSE 2024, pp. 572\u2013576. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3663529.3663801.","DOI":"10.1145\/3663529.3663801"},{"key":"9767_CR10","doi-asserted-by":"publisher","first-page":"383","DOI":"10.1007\/978-3-031-37703-7_18","volume-title":"Computer Aided Verification","author":"M Cosler","year":"2023","unstructured":"Cosler, M., Hahn, C., Mendoza, D., Schmitt, F., & Trippel, C. (2023). nl2spec: Interactively translating unstructured natural language to temporal logics with large language models. In C. Enea & A. Lal (Eds.), Computer Aided Verification (pp. 383\u2013396). Cham: Springer."},{"key":"9767_CR11","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1016\/j.infsof.2018.12.007","volume":"110","author":"F Dalpiaz","year":"2019","unstructured":"Dalpiaz, F., van der Schalk, I., Brinkkemper, S., Aydemir, F. B., & Lucassen, G. (2019). Detecting terminological ambiguity in user stories: Tool and experimentation. Information and Software Technology, 110, 3\u201316. https:\/\/doi.org\/10.1016\/j.infsof.2018.12.007","journal-title":"Information and Software Technology"},{"key":"9767_CR12","doi-asserted-by":"publisher","unstructured":"Deng, Y., Xia, C.S., Peng, H., Yang, C., & Zhang, L. (2023). Large language models are zero-shot fuzzers: Fuzzing deep-learning libraries via large language models. In Proceedings of the 32nd ACM SIGSOFT International Symposium on Software Testing and Analysis. ISSTA 2023, pp. 423\u2013435. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3597926.3598067.","DOI":"10.1145\/3597926.3598067"},{"key":"9767_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.infsof.2025.107928","volume":"189","author":"S Di Meglio","year":"2026","unstructured":"Di Meglio, S., Starace, L. L. L., Pontillo, V., Opdebeeck, R., De Roover, C., & Di Martino, S. (2026). Investigating the adoption and maintenance of web gui testing: Insights from github repositories. Information and Software Technology, 189, Article 107928. https:\/\/doi.org\/10.1016\/j.infsof.2025.107928","journal-title":"Information and Software Technology"},{"key":"9767_CR14","doi-asserted-by":"publisher","unstructured":"Endres, M., Fakhoury, S., Chakraborty, S., & Lahiri, S.K. (2024). Can large language models transform natural language intent into formal method postconditions?. In Proceedings of the ACM on Software Engineering 1(FSE). https:\/\/doi.org\/10.1145\/3660791.","DOI":"10.1145\/3660791"},{"key":"9767_CR15","doi-asserted-by":"crossref","unstructured":"Fan, Z., Gao, X., Mirchev, M., Roychoudhury, A., & Tan, S. H. (2023). Automated Repair of Programs from Large Language Models.","DOI":"10.1109\/ICSE48619.2023.00128"},{"key":"9767_CR16","first-page":"1","volume":"1","author":"V Helander","year":"2024","unstructured":"Helander, V., Ekedahl, H., Bucaioni, A., & Nguyen, T. P. (2024). Programming with chatgpt: How far can we go? Machine Learning with Applications, 1, 1\u201334.","journal-title":"Machine Learning with Applications"},{"key":"9767_CR17","doi-asserted-by":"publisher","unstructured":"Hou, X., Zhao, Y., Liu, Y., Yang, Z., Wang, K., Li, L., Luo, X., Lo, D., Grundy, J., & Wang, H. (2024). Large language models for software engineering: A systematic literature review. ACM Transactions on Software Engineering and Methodology, 33(8) https:\/\/doi.org\/10.1145\/3695988.","DOI":"10.1145\/3695988"},{"key":"9767_CR18","doi-asserted-by":"publisher","unstructured":"Jard, C., J\u00e9ron, T., & Morel, P. (2000). In H. Ural, R.L. Probert, & G. Bochmann (Eds.), Verification of Test Suites, pp. 3\u201318. Springer, Boston, MA. https:\/\/doi.org\/10.1007\/978-0-387-35516-0_1.","DOI":"10.1007\/978-0-387-35516-0_1"},{"key":"9767_CR19","doi-asserted-by":"publisher","unstructured":"Jesse, K., Ahmed, T., Devanbu, P.T., & Morgan, E. (2023). Large language models and simple, stupid bugs. In 2023 IEEE\/ACM 20th International Conference on Mining Software Repositories (MSR), pp. 563\u2013575. https:\/\/doi.org\/10.1109\/MSR59073.2023.00082.","DOI":"10.1109\/MSR59073.2023.00082"},{"key":"9767_CR20","doi-asserted-by":"crossref","unstructured":"Jiang, N., Liu, K., Lutellier, T., & Tan, L. (2023). Impact of Code Language Models on Automated Program Repair.","DOI":"10.1109\/ICSE48619.2023.00125"},{"key":"9767_CR21","doi-asserted-by":"publisher","unstructured":"Just, R., Jalali, D., & Ernst, M.D. (2014). Defects4j: a database of existing faults to enable controlled testing studies for java programs. In Proceedings of the 2014 International Symposium on Software Testing and Analysis. ISSTA 2014, pp. 437\u2013440. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/2610384.2628055.","DOI":"10.1145\/2610384.2628055"},{"key":"9767_CR22","unstructured":"Kanade, A., Maniatis, P., Balakrishnan, G., & Shi, K. (2020). Learning and evaluating contextual embedding of source code. In III, H.D., Singh, A. (Eds.), Proceedings of the 37th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 119, pp. 5110\u20135121. PMLR, online. https:\/\/proceedings.mlr.press\/v119\/kanade20a.html."},{"key":"9767_CR23","doi-asserted-by":"publisher","unstructured":"Liu, Z., Chen, C., Wang, J., Chen, M., Wu, B., Che, X., Wang, D., & Wang, Q. (2024). Make llm a testing expert: Bringing human-like interaction to mobile gui testing via functionality-aware decisions. In Proceedings of the IEEE\/ACM 46th International Conference on Software Engineering. ICSE \u201924. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3597503.3639180.","DOI":"10.1145\/3597503.3639180"},{"key":"9767_CR24","doi-asserted-by":"publisher","unstructured":"Lu, Y., Yao, B., Gu, H., Huang, J., Wang, J., Li, L., Gesi, J., He, Q., Li, T.J.-J., & Wang, D. (2025). Uxagent: An llm agent-based usability testing framework for web design. In Extended Abstracts of the CHI Conference on Human Factors in Computing Systems (CHI EA \u201925). ACM, Yokohama, Japan. https:\/\/doi.org\/10.1145\/3706599.3719729.","DOI":"10.1145\/3706599.3719729"},{"key":"9767_CR25","doi-asserted-by":"publisher","unstructured":"Luo, Q., Hariri, F., Eloussi, L., & Marinov, D. (2014). An empirical analysis of flaky tests. In Proceedings of the 22nd ACM SIGSOFT International Symposium on Foundations of Software Engineering. FSE 2014, pp. 643\u2013653. Association for Computing Machinery, New York, NY, USA.https:\/\/doi.org\/10.1145\/2635868.2635920.","DOI":"10.1145\/2635868.2635920"},{"key":"9767_CR26","doi-asserted-by":"publisher","unstructured":"Ma, W., Wu, D., Sun, Y., Wang, T., Liu, S., Zhang, J., Xue, Y., & Liu, Y. (2025). Combining fine-tuning and llm-based agents for intuitive smart contract auditing with justifications. In Proceedings - 2025 IEEE\/ACM 47th International Conference on Software Engineering, ICSE 2025, pp. 1742\u20131754. IEEE Computer Society, United States. https:\/\/doi.org\/10.1109\/ICSE55347.2025.00027. Publisher Copyright: 2025 IEEE.; 47th IEEE\/ACM International Conference on Software Engineering, ICSE 2025 ; Conference date: 26-04-2025 Through 06-05-2025.","DOI":"10.1109\/ICSE55347.2025.00027"},{"key":"9767_CR27","unstructured":"Maniatis, P., & Tarlow, D. (2023). Large Sequence Models for Software Development Activities. Google Brain. https:\/\/blog.research.google\/2023\/05\/large-sequence-models-for-software.html."},{"key":"9767_CR28","unstructured":"Salva, S. (2026). NLTestCaseRunner, A Natural Language Test Case Executor Using LLM Agents. https:\/\/github.com\/FondationUCA-Chair-LLM\/NL-test-case-runnerv2."},{"key":"9767_CR29","unstructured":"Salva, S., & Zafimiharisoa, S.R. (2014). Model reverse-engineering of mobile applications with exploration strategies. In 9th International Conference on Software Engineering Advances, ICSEA 2014, Nice, France. https:\/\/uca.hal.science\/hal-02019705."},{"key":"9767_CR30","doi-asserted-by":"publisher","unstructured":"Salva, S., & Sue, J. (2025). Dynamic mitigation of restful service failures using llms. In: Mecella, M., Rensink, A., Maciaszek, L.A. (Eds.), Proceedings of the 20th International Conference on Software Technologies, ICSOFT 2025, June 10-12, 2025, pp. 27\u201338. SCITEPRESS, Bilbao, Spain. https:\/\/doi.org\/10.5220\/0013460700003964.","DOI":"10.5220\/0013460700003964"},{"key":"9767_CR31","doi-asserted-by":"crossref","unstructured":"Salva, S., & Taguelmimt, R. (2025). On the Soundness and Consistency of LLM Agents for Executing Test Cases Written in Natural Language arXiv:2509.19136.","DOI":"10.5220\/0014980700004015"},{"key":"9767_CR32","doi-asserted-by":"publisher","unstructured":"Sapozhnikov, A., Olsthoorn, M., Panichella, A., Kovalenko, V., & Derakhshanfar, P. (2024). Testspark: Intellij idea\u2019s ultimate test generation companion. In Proceedings of the 2024 IEEE\/ACM 46th International Conference on Software Engineering: Companion Proceedings, ICSE Companion 2024, April 14-20, 2024, pp. 30\u201334. ACM, Lisbon, Portugal. https:\/\/doi.org\/10.1145\/3639478.3640024.","DOI":"10.1145\/3639478.3640024"},{"key":"9767_CR33","unstructured":"Schuster, R., Song, C., Tromer, E., & Shmatikov, V. (2021). You autocomplete me: Poisoning vulnerabilities in neural code completion. In 30th USENIX Security Symposium (USENIX Security 21), pp. 1559\u20131575. USENIX Association, online. https:\/\/www.usenix.org\/conference\/usenixsecurity21\/presentation\/schuster."},{"key":"9767_CR34","doi-asserted-by":"publisher","unstructured":"Sinha, A., Sutton, S. M., & Paradkar, A. (2010). Text2test: Automated inspection of natural language use cases. 2010 3rd International Conference on Software Testing, Verification and Validation (pp. 155\u2013164). https:\/\/doi.org\/10.1109\/ICST.2010.19.","DOI":"10.1109\/ICST.2010.19"},{"key":"9767_CR35","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2025.115177","volume":"335","author":"W Su","year":"2026","unstructured":"Su, W., Wu, X., & Zhao, Y. (2026). Nl2acsl: Interactively translating natural language to ansi c specification language with large language models. Knowledge-Based Systems, 335, Article 115177. https:\/\/doi.org\/10.1016\/j.knosys.2025.115177","journal-title":"Knowledge-Based Systems"},{"key":"9767_CR36","doi-asserted-by":"publisher","unstructured":"Tomic, S., Al\u00e9groth, E., & Isaac, M. (2025). Evaluation of the choice of llm in a multi-agent solution for gui-test generation. 2025 IEEE Conference on Software Testing, Verification and Validation (ICST) (pp. 487\u2013497). https:\/\/doi.org\/10.1109\/ICST62969.2025.10989038.","DOI":"10.1109\/ICST62969.2025.10989038"},{"issue":"3","key":"9767_CR37","first-page":"103","volume":"17","author":"J Tretmans","year":"1996","unstructured":"Tretmans, J. (1996). Test generation with inputs, outputs and repetitive quiescence. Software - Concepts and Tools, 17(3), 103\u2013120.","journal-title":"Software - Concepts and Tools"},{"issue":"13\u201314","key":"9767_CR38","doi-asserted-by":"publisher","first-page":"1526","DOI":"10.1080\/14783363.2016.1150173","volume":"28","author":"M Uluskan","year":"2017","unstructured":"Uluskan, M., Godfrey, A. B., & Joines, J. A. (2017). Integration of six sigma to traditional quality management theory: an empirical study on organisational performance. Total Quality Management & Business Excellence, 28(13\u201314), 1526\u20131543. https:\/\/doi.org\/10.1080\/14783363.2016.1150173","journal-title":"Total Quality Management & Business Excellence"},{"key":"9767_CR39","doi-asserted-by":"publisher","unstructured":"Wang, C., Pastore, F., Goknil, A., Briand, L. C., & Iqbal, Z. (2015). Umtg: a toolset to automatically generate system test cases from use case specifications. In Proceedings of the 2015 10th Joint Meeting on Foundations of Software Engineering. ESEC\/FSE 2015 (pp. 942\u2013945). New York, NY, USA: Association for Computing Machinery. https:\/\/doi.org\/10.1145\/2786805.2803187.","DOI":"10.1145\/2786805.2803187"},{"key":"9767_CR40","doi-asserted-by":"crossref","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E. H., Le, Q. V., & Zhou, D. (2022). Chain-of-thought prompting elicits reasoning in large language models. Proceedings of the 36th International Conference on Neural Information Processing Systems. NIPS \u201922. Red Hook, NY, USA: Curran Associates Inc.","DOI":"10.52202\/068431-1800"},{"key":"9767_CR41","unstructured":"White, J., Fu, Q., Hays, S., Sandborn, M., Olea, C., Gilbert, H., Elnashar, A., Spencer-Smith, J., & Schmidt, D. C. (2023). A prompt pattern catalog to enhance prompt engineering with chatgpt. In Proceedings of the 30th Conference on Pattern Languages of Programs. PLoP \u201923. The Hillside Group. USA."},{"key":"9767_CR42","doi-asserted-by":"crossref","unstructured":"Wu, Y., Jiang, A. Q., Li, W., Rabe, M. N., Staats, C., Jamnik, M., & Szegedy, C. (2022). Autoformalization with large language models arXiv:2205.12615.","DOI":"10.52202\/068431-2344"},{"key":"9767_CR43","unstructured":"Yasunaga, M., & Liang, P. (2020). Graph-based, self-supervised program repair from diagnostic feedback. In Proceedings of the 37th International Conference on Machine Learning. ICML\u201920. JMLR.org, online"},{"key":"9767_CR44","doi-asserted-by":"publisher","unstructured":"Yoon, J., Feldt, R., & Yoo, S. (2024). Intent-Driven Mobile GUI Testing with Autonomous Large Language Model Agents . In 2024 IEEE Conference on Software Testing, Verification and Validation (ICST), pp. 129\u2013139. IEEE Computer Society, Los Alamitos, CA, USA. https:\/\/doi.org\/10.1109\/ICST60714.2024.00020, https:\/\/doi.ieeecomputersociety.org\/10.1109\/ICST60714.2024.00020.","DOI":"10.1109\/ICST60714.2024.00020"}],"container-title":["Software Quality Journal"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11219-026-09767-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11219-026-09767-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11219-026-09767-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T18:50:34Z","timestamp":1782154234000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11219-026-09767-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,22]]},"references-count":44,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,9]]}},"alternative-id":["9767"],"URL":"https:\/\/doi.org\/10.1007\/s11219-026-09767-2","relation":{},"ISSN":["0963-9314","1573-1367"],"issn-type":[{"value":"0963-9314","type":"print"},{"value":"1573-1367","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,22]]},"assertion":[{"value":"17 March 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 June 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}},{"value":"Not applicable","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics Approval and Consent to Participate"}},{"value":"All authors consent for publication.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}}],"article-number":"33"}}