{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T07:01:40Z","timestamp":1784530900020,"version":"3.55.0"},"reference-count":97,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T00:00:00Z","timestamp":1776988800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":52,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"name":"European Union \u2013 NextGenerationEU"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-026-11571-0","type":"journal-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T14:53:48Z","timestamp":1777042428000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["From benchmarks to deployment: a comprehensive review of agentic AI evaluation"],"prefix":"10.1007","volume":"59","author":[{"given":"Tanzila","family":"Kehkashan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Muhammad","family":"Abdullah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ahmad Sami","family":"Al-Shamayleh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nikola","family":"Ivkovi\u0107","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nor Azman","family":"Ismail","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sharifah Sakinah Syed","family":"Ahmad","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abdul","family":"Rehman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8370-9290","authenticated-orcid":false,"given":"Adnan","family":"Akhunzada","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,4,24]]},"reference":[{"key":"11571_CR1","doi-asserted-by":"crossref","unstructured":"Acharya DB, Kuppan K, Divya B (2025) Agentic ai: Autonomous intelligence for complex goals\u2013a comprehensive survey. IEEe Access","DOI":"10.1109\/ACCESS.2025.3532853"},{"key":"11571_CR2","doi-asserted-by":"crossref","unstructured":"Akshathala S, Adnan B, Ramesh M, Vaidhyanathan K, Muhammed B, Parthasarathy K (2025) Beyond task completion: An assessment framework for evaluating agentic ai systems. arXiv preprint arXiv:2512.12791","DOI":"10.1145\/3786167.3788414"},{"key":"11571_CR3","doi-asserted-by":"publisher","first-page":"101677","DOI":"10.1016\/j.softx.2024.101677","volume":"26","author":"Y Almeida","year":"2024","unstructured":"Almeida Y, Albuquerque D, Dantas Filho E, Muniz F, Farias Santos K, Perkusich M, Almeida H, Perkusich A (2024) Aicodereview: Advancing code quality with ai-enhanced reviews. SoftwareX 26:101677","journal-title":"SoftwareX"},{"key":"11571_CR4","doi-asserted-by":"publisher","first-page":"143815","DOI":"10.1109\/ACCESS.2023.3343252","volume":"11","author":"R Anwar","year":"2023","unstructured":"Anwar R, Bashir MB (2023) A systematic literature review of ai-based software requirements prioritization techniques. IEEE Access 11:143815\u2013143860","journal-title":"IEEE Access"},{"issue":"5","key":"11571_CR5","first-page":"52","volume":"14","author":"A Arabiat","year":"2024","unstructured":"Arabiat A, Altayeb M (2024) Enhancing internet of things security: evaluating machine learning classifiers for attack prediction. Int J Elect Comput Eng 14(5):52","journal-title":"Int J Elect Comput Eng"},{"issue":"8938","key":"11571_CR6","first-page":"4423","volume":"2252","author":"A Arabiat","year":"2023","unstructured":"Arabiat A, Hassan M, Almomani O (2023) Weka-based machine learning for traffic congestion prediction in amman city. Int J Artif Intell ISSN 2252(8938):4423","journal-title":"Int J Artif Intell ISSN"},{"key":"11571_CR7","doi-asserted-by":"crossref","unstructured":"Baack S (2024) A critical analysis of the largest source for generative ai training data: Common crawl. In: Proceedings of the 2024 ACM Conference on Fairness, Accountability, and Transparency, pp. 2199\u20132208","DOI":"10.1145\/3630106.3659033"},{"issue":"9","key":"11571_CR8","doi-asserted-by":"publisher","first-page":"404","DOI":"10.3390\/fi17090404","volume":"17","author":"A Bandi","year":"2025","unstructured":"Bandi A, Kongari B, Naguru R, Pasnoor S, Vilipala SV (2025) The rise of agentic ai: a review of definitions, frameworks, architectures, applications, evaluation metrics, and challenges. Fut Intern 17(9):404","journal-title":"Fut Intern"},{"issue":"8","key":"11571_CR9","doi-asserted-by":"publisher","first-page":"2939","DOI":"10.1007\/s00371-021-02166-7","volume":"38","author":"K Bayoudh","year":"2022","unstructured":"Bayoudh K, Knani R, Hamdaoui F, Mtibaa A (2022) A survey on deep multimodal learning for computer vision: advances, trends, applications, and datasets. Vis Comput 38(8):2939\u20132970","journal-title":"Vis Comput"},{"key":"11571_CR10","unstructured":"Borazio F, Croce D, Basili R (2025) Unitor at bioasq 2025: modular biomedical qa with synthetic snippets and multiple task answer generation. In: CLEF"},{"key":"11571_CR11","unstructured":"Brohan A, Chebotar Y, Finn C, Hausman K, Herzog A, Ho D, Ibarz J, Irpan A, Jang E, Julian R, et al (2023) Do as i can, not as i say: Grounding language in robotic affordances. In: Conference on Robot Learning, pp. 287\u2013318. PMLR"},{"issue":"11","key":"11571_CR12","doi-asserted-by":"publisher","first-page":"4948","DOI":"10.3390\/app11114948","volume":"11","author":"L Canese","year":"2021","unstructured":"Canese L, Cardarilli GC, Di Nunzio L, Fazzolari R, Giardino D, Re M, Span\u00f2 S (2021) Multi-agent reinforcement learning: a review of challenges and applications. Appl Sci 11(11):4948","journal-title":"Appl Sci"},{"issue":"3","key":"11571_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3641289","volume":"15","author":"Y Chang","year":"2024","unstructured":"Chang Y, Wang X, Wang J, Wu Y, Yang L, Zhu K, Chen H, Yi X, Wang C, Wang Y et al (2024) A survey on evaluation of large language models. ACM Trans Intell Syst Technol 15(3):1\u201345","journal-title":"ACM Trans Intell Syst Technol"},{"issue":"1","key":"11571_CR14","doi-asserted-by":"publisher","first-page":"111102","DOI":"10.1007\/s11432-023-4127-5","volume":"68","author":"X Chen","year":"2025","unstructured":"Chen X, Hu X, Huang Y, Jiang H, Ji W, Jiang Y, Jiang Y, Liu B, Liu H, Li X et al (2025) Deep learning-based software engineering: progress, challenges, and opportunities. Sci China Inf Sci 68(1):111102","journal-title":"Sci China Inf Sci"},{"key":"11571_CR15","doi-asserted-by":"crossref","unstructured":"Chen Y, Xie H, Ma M, Kang Y, Gao X, Shi L, Cao Y, Gao X, Fan H, Wen M, et al (2024) Automatic root cause analysis via large language models for cloud incidents. In: Proceedings of the Nineteenth European Conference on Computer Systems, pp. 674\u2013688","DOI":"10.1145\/3627703.3629553"},{"key":"11571_CR16","unstructured":"Chen M (2021) Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374"},{"key":"11571_CR17","doi-asserted-by":"crossref","unstructured":"Cihon P, Stein M, Bansal G, Manning S, Xu K (2025) Measuring ai agent autonomy: Towards a scalable approach with code inspection. arXiv preprint arXiv:2502.15212","DOI":"10.70777\/si.v2i3.15295"},{"key":"11571_CR18","doi-asserted-by":"publisher","first-page":"107468","DOI":"10.1016\/j.infsof.2024.107468","volume":"171","author":"AM Dakhel","year":"2024","unstructured":"Dakhel AM, Nikanjam A, Majdinasab V, Khomh F, Desmarais MC (2024) Effective test generation using pre-trained large language models and mutation testing. Inf Softw Technol 171:107468","journal-title":"Inf Softw Technol"},{"key":"11571_CR19","doi-asserted-by":"publisher","first-page":"28091","DOI":"10.52202\/075280-1220","volume":"36","author":"X Deng","year":"2023","unstructured":"Deng X, Gu Y, Zheng B, Chen S, Stevens S, Wang B, Sun H, Su Y (2023) Mind2web: towards a generalist agent for the web. Adv Neural Inf Process Syst 36:28091\u201328114","journal-title":"Adv Neural Inf Process Syst"},{"issue":"3","key":"11571_CR20","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3695991","volume":"34","author":"Y Dong","year":"2025","unstructured":"Dong Y, Ding J, Jiang X, Li G, Li Z, Jin Z (2025) Codescore: evaluating code generation by learning code execution. ACM Trans Softw Eng Methodol 34(3):1\u201322","journal-title":"ACM Trans Softw Eng Methodol"},{"issue":"7","key":"11571_CR21","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3672459","volume":"33","author":"Y Dong","year":"2024","unstructured":"Dong Y, Jiang X, Jin Z, Li G (2024) Self-collaboration code generation via chatgpt. ACM Trans Softw Eng Methodol 33(7):1\u201338","journal-title":"ACM Trans Softw Eng Methodol"},{"issue":"5","key":"11571_CR22","doi-asserted-by":"publisher","first-page":"3215","DOI":"10.1007\/s10462-020-09938-y","volume":"54","author":"W Du","year":"2021","unstructured":"Du W, Ding S (2021) A survey on multi-agent deep reinforcement learning: from the perspective of challenges and applications. Artif Intell Rev 54(5):3215\u20133238","journal-title":"Artif Intell Rev"},{"issue":"6","key":"11571_CR23","doi-asserted-by":"publisher","first-page":"138","DOI":"10.1007\/s10664-022-10184-9","volume":"27","author":"NU Eisty","year":"2022","unstructured":"Eisty NU, Carver JC (2022) Testing research software: a survey. Empir Softw Eng 27(6):138","journal-title":"Empir Softw Eng"},{"key":"11571_CR24","unstructured":"Esposito G (2025) Llms in the siem loop: A contract-based framework for threat detection with an evaluation on windows telemetry and mitre att&ck mapping. PhD thesis, Politecnico di Torino"},{"key":"11571_CR25","doi-asserted-by":"publisher","first-page":"111741","DOI":"10.1016\/j.jss.2023.111741","volume":"203","author":"M Evtikhiev","year":"2023","unstructured":"Evtikhiev M, Bogomolov E, Sokolov Y, Bryksin T (2023) Out of the bleu: how should we assess quality of the code generation models? J Syst Softw 203:111741","journal-title":"J Syst Softw"},{"issue":"4","key":"11571_CR26","doi-asserted-by":"publisher","first-page":"271","DOI":"10.3233\/AIC-220113","volume":"35","author":"I Gemp","year":"2022","unstructured":"Gemp I, Anthony T, Bachrach Y, Bhoopchand A, Bullard K, Connor J, Dasagi V, De Vylder B, Duenez-Guzman EA, Elie R et al (2022) Developing, evaluating and scaling learning agents in multi-agent environments. AI Commun 35(4):271\u2013284","journal-title":"AI Commun"},{"key":"11571_CR27","first-page":"346","volume":"9","author":"M Geva","year":"2021","unstructured":"Geva M, Khashabi D, Segal E, Khot T, Roth D, Berant J (2021) Did aristotle use a laptop? A question answering benchmark with implicit reasoning strategies. Trans Assoc Comput Ling 9:346\u2013361","journal-title":"Trans Assoc Comput Ling"},{"key":"11571_CR28","unstructured":"Golchin S, Surdeanu M (2023) Time travel in llms: Tracing data contamination in large language models. arXiv preprint arXiv:2308.08493"},{"key":"11571_CR29","first-page":"381","volume-title":"AI Agents on the CLI","author":"A Gull\u00ed","year":"2025","unstructured":"Gull\u00ed A (2025) AI Agents on the CLI. Springer, Cham, pp 381\u2013386"},{"key":"11571_CR30","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1016\/j.future.2021.06.014","volume":"126","author":"OE Gundersen","year":"2022","unstructured":"Gundersen OE, Shamsaliei S, Isdahl RJ (2022) Do machine learning platforms provide out-of-the-box reproducibility? Futur Gener Comput Syst 126:34\u201347","journal-title":"Futur Gener Comput Syst"},{"issue":"1","key":"11571_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3611664","volume":"33","author":"H Guo","year":"2023","unstructured":"Guo H, Chen X, Huang Y, Wang Y, Ding X, Zheng Z, Zhou X, Dai H-N (2023) Snippet comment generation based on code context expansion. ACM Trans Softw Eng Methodol 33(1):1\u201330","journal-title":"ACM Trans Softw Eng Methodol"},{"issue":"5","key":"11571_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3712003","volume":"34","author":"J He","year":"2025","unstructured":"He J, Treude C, Lo D (2025) Llm-based multi-agent systems for software engineering: literature review, vision, and the road ahead. ACM Trans Softw Eng Methodol 34(5):1\u201330","journal-title":"ACM Trans Softw Eng Methodol"},{"key":"11571_CR33","unstructured":"Hendrycks D, Carlini N, Schulman J, Steinhardt J (2021) Unsolved problems in ml safety. arXiv preprint arXiv:2109.13916"},{"issue":"8","key":"11571_CR34","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3695988","volume":"33","author":"X Hou","year":"2024","unstructured":"Hou X, Zhao Y, Liu Y, Yang Z, Wang K, Li L, Luo X, Lo D, Grundy J, Wang H (2024) Large language models for software engineering: A systematic literature review. ACM Trans Softw Eng Methodol 33(8):1\u201379","journal-title":"ACM Trans Softw Eng Methodol"},{"key":"11571_CR35","doi-asserted-by":"publisher","first-page":"128068","DOI":"10.1016\/j.neucom.2024.128068","volume":"599","author":"K Hu","year":"2024","unstructured":"Hu K, Li M, Song Z, Xu K, Xia Q, Sun N, Zhou P, Xia M (2024) A review of research on reinforcement learning algorithms for multi-agents. Neurocomputing 599:128068","journal-title":"Neurocomputing"},{"issue":"1","key":"11571_CR36","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1007\/s10515-022-00374-6","volume":"30","author":"Y Huang","year":"2023","unstructured":"Huang Y, Huang J, Chen X, He K, Zhou X (2023) Bcgen: a comment generation method for bytecode. Autom Softw Eng 30(1):5","journal-title":"Autom Softw Eng"},{"key":"11571_CR37","unstructured":"Hui DY-T, Chevalier-Boisvert M, Bahdanau D, Bengio Y (2020) Babyai 1.1. arXiv preprint arXiv:2007.12770"},{"key":"11571_CR38","unstructured":"Jimenez CE, Yang J, Wettig A, Yao S, Pei K, Press O, Narasimhan K (2023) Swe-bench: Can language models resolve real-world github issues? arXiv preprint arXiv:2310.06770"},{"key":"11571_CR39","unstructured":"Kamalloo E, Gontier N, Lu XH, Dziri N, Murty S, Lacoste A (2025) Proceedings of the 1st workshop for research on agent language models (realm 2025). In: Proceedings of the 1st Workshop for Research on Agent Language Models (REALM 2025)"},{"issue":"9","key":"11571_CR40","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1016\/j.patter.2023.100804","volume":"4","author":"S Kapoor","year":"2023","unstructured":"Kapoor S, Narayanan A (2023) Leakage and the reproducibility crisis in machine-learning-based science. Patterns 4(9):123","journal-title":"Patterns"},{"key":"11571_CR41","unstructured":"Kazdan J, Schaeffer R, Allouah Y, Sullivan C, Yu K, Levi N, Koyejo S (2025) Efficient prediction of pass@ k scaling in large language models. arXiv preprint arXiv:2510.05197"},{"key":"11571_CR42","unstructured":"Kenton Z, Everitt T, Weidinger L, Gabriel I, Mikulik V, Irving G (2021) Alignment of language agents. arXiv preprint arXiv:2103.14659"},{"key":"11571_CR43","doi-asserted-by":"crossref","unstructured":"Kim J, Shin B, Chung J, Rhu M (2025) The cost of dynamic reasoning: Demystifying ai agents and test-time scaling from an ai infrastructure perspective. arXiv preprint arXiv:2506.04301","DOI":"10.1109\/HPCA68181.2026.11408569"},{"key":"11571_CR44","unstructured":"Kio G (2025) Swe-bench-secret: Automating ai agent evaluation for software engineering tasks"},{"issue":"3","key":"11571_CR45","doi-asserted-by":"publisher","first-page":"1501","DOI":"10.25300\/MISQ\/2021\/16564","volume":"45","author":"S Lebovitz","year":"2021","unstructured":"Lebovitz S, Levina N, Lifshitz-Assaf H (2021) Is ai ground truth really true? the dangers of training and evaluating ai tools based on experts\u2019 know-what. MIS Q 45(3):1501\u20131526","journal-title":"MIS Q"},{"key":"11571_CR46","doi-asserted-by":"publisher","first-page":"31199","DOI":"10.52202\/068431-2262","volume":"35","author":"S Li","year":"2022","unstructured":"Li S, Puig X, Paxton C, Du Y, Wang C, Fan L, Chen T, Huang D-A, Aky\u00fcrek E, Anandkumar A et al (2022) Pre-trained language models for interactive decision-making. Adv Neural Inf Process Syst 35:31199\u201331212","journal-title":"Adv Neural Inf Process Syst"},{"key":"11571_CR47","doi-asserted-by":"crossref","unstructured":"Li M, Zhao Y, Yu B, Song F, Li H, Yu H, Li Z, Huang F, Li Y (2023) Api-bank: A comprehensive benchmark for tool-augmented llms. arXiv preprint arXiv:2304.08244","DOI":"10.18653\/v1\/2023.emnlp-main.187"},{"key":"11571_CR48","unstructured":"Li D, Murr L (2024) Humaneval on latest gpt models\u20132024. arXiv preprint arXiv:2402.14852"},{"key":"11571_CR49","unstructured":"Li Z, Li Y, Zhang Y, et al (2024) Adaptive environmental modeling for task-oriented language agents"},{"key":"11571_CR50","unstructured":"Li Z, Zhao X, Wu D-D, Cui J, Shen Z (2025) A frustratingly simple yet highly effective attack baseline: Over 90% success rate against the strong black-box models of gpt-4.5\/4o\/o1. arXiv preprint arXiv:2503.10635"},{"key":"11571_CR51","unstructured":"Liang P, Bommasani R, Lee T, Tsipras D, Soylu D, Yasunaga M, Zhang Y, Narayanan D, Wu Y, Kumar A, et al (2022) Holistic evaluation of language models. arXiv preprint arXiv:2211.09110"},{"key":"11571_CR52","unstructured":"Liu X, Yu H, Zhang H, Xu Y, Lei X, Lai H, Gu Y, Ding H, Men K, Yang K, et al (2023) Agentbench: Evaluating llms as agents. arXiv preprint arXiv:2308.03688"},{"issue":"8","key":"11571_CR53","doi-asserted-by":"publisher","first-page":"198341","DOI":"10.1007\/s11704-024-40415-9","volume":"19","author":"Z Lyu","year":"2025","unstructured":"Lyu Z, Li X, Xie Z, Li M (2025) Top pass: improve code generation by pass@ k-maximized code ranking. Front Comp Sci 19(8):198341","journal-title":"Front Comp Sci"},{"issue":"2","key":"11571_CR54","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3487043","volume":"31","author":"S Mart\u00ednez-Fern\u00e1ndez","year":"2022","unstructured":"Mart\u00ednez-Fern\u00e1ndez S, Bogner J, Franch X, Oriol M, Siebert J, Trendowicz A, Vollmer AM, Wagner S (2022) Software engineering for ai-based systems: a survey. ACM Trans Softw Eng Methodol (TOSEM) 31(2):1\u201359","journal-title":"ACM Trans Softw Eng Methodol (TOSEM)"},{"key":"11571_CR55","unstructured":"Merrill MA, Shaw AG, Carlini N, Li B, Raj H, Bercovich I, Shi L, Shin JY, Walshe T, Buchanan EK, et al (2026) Terminal-bench: Benchmarking agents on hard, realistic tasks in command line interfaces. arXiv preprint arXiv:2601.11868"},{"key":"11571_CR56","unstructured":"Mialon G, Fourrier C, Wolf T, LeCun Y, Scialom T (2023) Gaia: a benchmark for general ai assistants. In: The Twelfth International Conference on Learning Representations"},{"key":"11571_CR57","doi-asserted-by":"crossref","unstructured":"Mohammadi M, Li Y, Lo J, Yip W (2025) Evaluation and benchmarking of llm agents: A survey. In: Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 2, pp. 6129\u20136139","DOI":"10.1145\/3711896.3736570"},{"key":"11571_CR58","doi-asserted-by":"publisher","first-page":"1167","DOI":"10.1613\/jair.1.15703","volume":"79","author":"A Mohan","year":"2024","unstructured":"Mohan A, Zhang A, Lindauer M (2024) Structure in deep reinforcement learning: a survey and open problems. J Artif Intell Res 79:1167\u20131236","journal-title":"J Artif Intell Res"},{"issue":"4","key":"11571_CR59","doi-asserted-by":"publisher","first-page":"916","DOI":"10.1016\/j.ijrobp.2023.11.012","volume":"118","author":"AC Moreno","year":"2024","unstructured":"Moreno AC, Bitterman DS (2024) Toward clinical-grade evaluation of large language models. Int J Radiat Oncol Biol Phys 118(4):916\u2013920","journal-title":"Int J Radiat Oncol Biol Phys"},{"issue":"11","key":"11571_CR60","doi-asserted-by":"publisher","first-page":"15051","DOI":"10.1109\/TNNLS.2023.3283523","volume":"35","author":"S Munikoti","year":"2023","unstructured":"Munikoti S, Agarwal D, Das L, Halappanavar M, Natarajan B (2023) Challenges and opportunities in deep reinforcement learning with graph neural networks: a comprehensive review of algorithms and applications. IEEE Trans Neural Netw Learn Syst 35(11):15051\u201315071","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"issue":"1","key":"11571_CR61","doi-asserted-by":"publisher","first-page":"5928","DOI":"10.1038\/s41598-023-32234-y","volume":"13","author":"SS Nair","year":"2023","unstructured":"Nair SS, Muddapu VR, Vigneswaran C, Balasubramani PP, Ramanathan DS, Mishra J, Chakravarthy VS (2023) A generalized reinforcement learning based deep neural network agent model for diverse cognitive constructs. Sci Rep 13(1):5928","journal-title":"Sci Rep"},{"key":"11571_CR62","doi-asserted-by":"crossref","unstructured":"Nashaat M, Miller J (2024) Towards efficient fine-tuning of language models with organizational data for automated software review. IEEE Transactions on Software Engineering","DOI":"10.1109\/TSE.2024.3428324"},{"key":"11571_CR63","unstructured":"Nathani D, Madaan L, Roberts N, Bashlykov N, Menon A, Moens V, Budhiraja A, Magka D, Vorotilov V, Chaurasia G, et al (2025) Mlgym: A new framework and benchmark for advancing ai research agents. arXiv preprint arXiv:2502.14499"},{"key":"11571_CR64","doi-asserted-by":"publisher","first-page":"111524","DOI":"10.1016\/j.jss.2022.111524","volume":"195","author":"IG Ndukwe","year":"2023","unstructured":"Ndukwe IG, Licorish SA, Tahir A, MacDonell SG (2023) How have views on software quality differed over time? Research and practice viewpoints. J Syst Softw 195:111524","journal-title":"J Syst Softw"},{"issue":"9","key":"11571_CR65","doi-asserted-by":"publisher","first-page":"3826","DOI":"10.1109\/TCYB.2020.2977374","volume":"50","author":"TT Nguyen","year":"2020","unstructured":"Nguyen TT, Nguyen ND, Nahavandi S (2020) Deep reinforcement learning for multiagent systems: a review of challenges, solutions, and applications. IEEE Trans Cybern 50(9):3826\u20133839","journal-title":"IEEE Trans Cybern"},{"issue":"1","key":"11571_CR66","doi-asserted-by":"publisher","first-page":"6793","DOI":"10.1038\/s41467-022-34591-0","volume":"13","author":"S Ott","year":"2022","unstructured":"Ott S, Barbosa-Silva A, Blagec K, Brauner J, Samwald M (2022) Mapping global dynamics of benchmark creation and saturation in artificial intelligence. Nat Commun 13(1):6793","journal-title":"Nat Commun"},{"issue":"164","key":"11571_CR67","first-page":"1","volume":"22","author":"J Pineau","year":"2021","unstructured":"Pineau J, Vincent-Lamarre P, Sinha K, Larivi\u00e8re V, Beygelzimer A, d\u2019Alch\u00e9-Buc F, Fox E, Larochelle H (2021) Improving reproducibility in machine learning research (a report from the neurips 2019 reproducibility program). J Mach Learn Res 22(164):1\u201320","journal-title":"J Mach Learn Res"},{"key":"11571_CR68","first-page":"120","volume":"32","author":"E Raff","year":"2019","unstructured":"Raff E (2019) A step toward quantifying independently reproducible machine learning research. Adv Neural Inf Process Syst 32:120","journal-title":"Adv Neural Inf Process Syst"},{"key":"11571_CR69","doi-asserted-by":"publisher","first-page":"133001","DOI":"10.1109\/ACCESS.2022.3230637","volume":"10","author":"H Rahman","year":"2022","unstructured":"Rahman H, D\u2019Cruze RS, Ahmed MU, Sohlberg R, Sakao T, Funk P (2022) Artificial intelligence-based life cycle engineering in industrial production: a systematic literature review. IEEE Access 10:133001\u2013133015","journal-title":"IEEE Access"},{"key":"11571_CR70","doi-asserted-by":"crossref","unstructured":"Sankhe MP, Patil N, Ghorpade MM, Prasad MP, Linkesh MM (2025) Empirical analysis of ai-assisted code generation tools: Impact on code quality, security and developer productivity","DOI":"10.36948\/ijfmr.2025.v07i06.61350"},{"issue":"1","key":"11571_CR71","doi-asserted-by":"publisher","first-page":"380","DOI":"10.32996\/jcsts.2025.7.1.28","volume":"7","author":"A Satav","year":"2025","unstructured":"Satav A (2025) Enterprise api & platform strategy in the era of agentic ai. J Comput Sci Technol Stud 7(1):380\u2013385","journal-title":"J Comput Sci Technol Stud"},{"issue":"1","key":"11571_CR72","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1109\/TSE.2023.3334955","volume":"50","author":"M Sch\u00e4fer","year":"2023","unstructured":"Sch\u00e4fer M, Nadi S, Eghbali A, Tip F (2023) An empirical evaluation of using large language models for automated unit test generation. IEEE Trans Software Eng 50(1):85\u2013105","journal-title":"IEEE Trans Software Eng"},{"key":"11571_CR73","doi-asserted-by":"publisher","first-page":"140896","DOI":"10.1109\/ACCESS.2021.3119746","volume":"9","author":"S Shafiq","year":"2021","unstructured":"Shafiq S, Mashkoor A, Mayr-Dorn C, Egyed A (2021) A literature review of using machine learning in software development life cycle stages. IEEe Access 9:140896\u2013140920","journal-title":"IEEe Access"},{"key":"11571_CR74","doi-asserted-by":"crossref","unstructured":"Shridhar M, Thomason J, Gordon D, Bisk Y, Han W, Mottaghi R, Zettlemoyer L, Fox D (2020) Alfred: A benchmark for interpreting grounded instructions for everyday tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10740\u201310749","DOI":"10.1109\/CVPR42600.2020.01075"},{"key":"11571_CR75","doi-asserted-by":"publisher","first-page":"103825","DOI":"10.1016\/j.aei.2025.103825","volume":"69","author":"AK Singh","year":"2026","unstructured":"Singh AK, Hsieh S-H (2026) Multi-llm-based augmentation and synthetic data generation of construction schedules and task descriptions with slm-as-a-judge assessment. Adv Eng Inform 69:103825","journal-title":"Adv Eng Inform"},{"issue":"3","key":"11571_CR76","doi-asserted-by":"publisher","first-page":"1512","DOI":"10.3390\/en16031512","volume":"16","author":"K Sivamayil","year":"2023","unstructured":"Sivamayil K, Rajasekar E, Aljafari B, Nikolovski S, Vairavasundaram S, Vairavasundaram I (2023) A systematic study on reinforcement learning based applications. Energies 16(3):1512","journal-title":"Energies"},{"key":"11571_CR77","unstructured":"Starace G, Jaffe O, Sherburn D, Aung J, Chan JS, Maksin L, Dias R, Mays E, Kinsella B, Thompson W, et al (2025) Paperbench: Evaluating ai\u2019s ability to replicate ai research. arXiv preprint arXiv:2504.01848"},{"key":"11571_CR78","unstructured":"Sumers T, Yao S, Narasimhan K, Griffiths T (2023) Cognitive architectures for language agents. Transactions on Machine Learning Research"},{"key":"11571_CR79","unstructured":"Tarasova N, Balp-Straffon E, Iancheruk A, Sielskyi Y, Kozodoi N, Byrne LH, Butler J, Czelej M, Ang A, Shah Y, et al (2025) Swe-infrabench: Evaluating language models on cloud infrastructure code. In: NeurIPS 2025 Workshop on Evaluating the Evolving LLM Lifecycle: Benchmarks, Emergent Abilities, and Scaling"},{"issue":"5","key":"11571_CR80","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3715003","volume":"34","author":"V Terragni","year":"2025","unstructured":"Terragni V, Vella A, Roop P, Blincoe K (2025) The future of ai-driven software engineering. ACM Trans Softw Eng Methodol 34(5):1\u201320","journal-title":"ACM Trans Softw Eng Methodol"},{"key":"11571_CR81","unstructured":"Thakkar M, Chapados N, Pal C, et al (2025) Webarena verified: Reliable evaluation for web agents. In: Workshop on Scaling Environments for Agents"},{"key":"11571_CR82","doi-asserted-by":"publisher","first-page":"1393","DOI":"10.1016\/j.procir.2024.10.257","volume":"130","author":"C Wachter","year":"2024","unstructured":"Wachter C, Beckschulte S, Hinrichs MP, Sohnius F, Schmitt RH (2024) Strategies for resilient manufacturing: a systematic literature review of failure management in production. Procedia CIRP 130:1393\u20131402","journal-title":"Procedia CIRP"},{"issue":"4","key":"11571_CR83","doi-asserted-by":"publisher","first-page":"911","DOI":"10.1109\/TSE.2024.3368208","volume":"50","author":"J Wang","year":"2024","unstructured":"Wang J, Huang Y, Chen C, Liu Z, Wang S, Wang Q (2024) Software testing with large language models: Survey, landscape, and vision. IEEE Trans Software Eng 50(4):911\u2013936","journal-title":"IEEE Trans Software Eng"},{"issue":"6","key":"11571_CR84","doi-asserted-by":"publisher","first-page":"186345","DOI":"10.1007\/s11704-024-40231-1","volume":"18","author":"L Wang","year":"2024","unstructured":"Wang L, Ma C, Feng X, Zhang Z, Yang H, Zhang J, Chen Z, Tang J, Chen X, Lin Y et al (2024) A survey on large language model based autonomous agents. Front Comp Sci 18(6):186345","journal-title":"Front Comp Sci"},{"key":"11571_CR85","doi-asserted-by":"publisher","first-page":"95266","DOI":"10.52202\/079017-3018","volume":"37","author":"Y Wang","year":"2024","unstructured":"Wang Y, Ma X, Zhang G, Ni Y, Chandra A, Guo S, Ren W, Arulraj A, He X, Jiang Z et al (2024) Mmlu-pro: a more robust and challenging multi-task language understanding benchmark. Adv Neural Inf Process Syst 37:95266\u201395290","journal-title":"Adv Neural Inf Process Syst"},{"key":"11571_CR86","unstructured":"Wang G, Liu J, Zhou M, Chen X, Zhang L, Sun Z (2025) Toolbench 2.0: Evaluating long-horizon and multi-step tool use in llms"},{"key":"11571_CR87","unstructured":"WebChoreArena (2025) WebArena: Webchorearena: Evaluating web browsing agents on realistic tedious web tasks. arXiv preprint"},{"issue":"2","key":"11571_CR88","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3487569","volume":"31","author":"FF Xu","year":"2022","unstructured":"Xu FF, Vasilescu B, Neubig G (2022) In-ide code generation from natural language: Promise and challenges. ACM Trans Softw Eng Methodol (TOSEM) 31(2):1\u201347","journal-title":"ACM Trans Softw Eng Methodol (TOSEM)"},{"key":"11571_CR89","doi-asserted-by":"crossref","unstructured":"Xu F, Medappa PK, Tunc MM (2024) Does ai technology deployment benefit the owner of the technology? impact of github copilot release on microsoft. Impact of GitHub Copilot Release on Microsoft (June 15, 2024)","DOI":"10.2139\/ssrn.4881226"},{"issue":"1","key":"11571_CR90","doi-asserted-by":"publisher","first-page":"23133","DOI":"10.1038\/s41598-025-05192-w","volume":"15","author":"Y Yan","year":"2025","unstructured":"Yan Y, Li J, Zaggia C (2025) The analysis of deep reinforcement learning for dynamic graphical games under artificial intelligence. Sci Rep 15(1):23133","journal-title":"Sci Rep"},{"issue":"4","key":"11571_CR91","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1109\/MRL.2025.3626482","volume":"2","author":"G-Y Yang","year":"2025","unstructured":"Yang G-Y, Wang F (2025) Taming silent failures: a framework for verifiable ai reliability. IEEE Reliabil Magaz 2(4):46\u201355","journal-title":"IEEE Reliabil Magaz"},{"key":"11571_CR92","doi-asserted-by":"crossref","unstructured":"Yang Z, Qi P, Zhang S, Bengio Y, Cohen W, Salakhutdinov R, Manning CD (2018) Hotpotqa: A dataset for diverse, explainable multi-hop question answering. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 2369\u20132380","DOI":"10.18653\/v1\/D18-1259"},{"issue":"1","key":"11571_CR93","doi-asserted-by":"publisher","first-page":"106","DOI":"10.4218\/etrij.2023-0357","volume":"46","author":"S Yeo","year":"2024","unstructured":"Yeo S, Ma Y-S, Kim SC, Jun H, Kim T (2024) Framework for evaluating code generation ability of large language models. ETRI J 46(1):106\u2013117","journal-title":"ETRI J"},{"issue":"1","key":"11571_CR94","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3477600","volume":"55","author":"C Yu","year":"2021","unstructured":"Yu C, Liu J, Nemati S, Yin G (2021) Reinforcement learning in healthcare: a survey. ACM Comput Surv (CSUR) 55(1):1\u201336","journal-title":"ACM Comput Surv (CSUR)"},{"key":"11571_CR95","doi-asserted-by":"crossref","unstructured":"Yu Z, Zhao Y, Cohan A, Zhang X-P (2024) Humaneval pro and mbpp pro: Evaluating large language models on self-invoking code generation. arXiv preprint arXiv:2412.21199","DOI":"10.18653\/v1\/2025.findings-acl.686"},{"issue":"3","key":"11571_CR96","doi-asserted-by":"publisher","first-page":"172988142110073","DOI":"10.1177\/17298814211007305","volume":"18","author":"T Zhang","year":"2021","unstructured":"Zhang T, Mo H (2021) Reinforcement learning for robot research: A comprehensive review and open issues. Int J Adv Rob Syst 18(3):17298814211007304","journal-title":"Int J Adv Rob Syst"},{"key":"11571_CR97","unstructured":"Zhou S, Xu FF, Zhu H, Zhou X, Lo R, Sridhar A, Cheng X, Ou T, Bisk Y, Fried D, et al (2023) Webarena: A realistic web environment for building autonomous agents. arXiv preprint arXiv:2307.13854"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-026-11571-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-026-11571-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-026-11571-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T06:02:19Z","timestamp":1784527339000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-026-11571-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,24]]},"references-count":97,"journal-issue":{"issue":"8","published-online":{"date-parts":[[2026,8]]}},"alternative-id":["11571"],"URL":"https:\/\/doi.org\/10.1007\/s10462-026-11571-0","relation":{},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,24]]},"assertion":[{"value":"12 January 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no Conflict of interest.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"167"}}