{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T20:18:00Z","timestamp":1783196280363,"version":"3.54.6"},"reference-count":208,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,7]],"date-time":"2026-05-07T00:00:00Z","timestamp":1778112000000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100008102","name":"University of Cincinnati","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100008102","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100018693","name":"Horizon Europe","doi-asserted-by":"publisher","award":["101214389"],"award-info":[{"award-number":["101214389"]}],"id":[{"id":"10.13039\/100018693","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Fusion"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.inffus.2026.104444","type":"journal-article","created":{"date-parts":[[2026,5,9]],"date-time":"2026-05-09T15:38:01Z","timestamp":1778341081000},"page":"104444","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Evaluating and regulating agentic AI: A study of benchmarks, metrics, and regulation"],"prefix":"10.1016","volume":"136","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7867-2546","authenticated-orcid":false,"given":"Azib","family":"Farooq","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1061-5845","authenticated-orcid":false,"given":"Shaina","family":"Raza","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5522-4456","authenticated-orcid":false,"given":"Nazmul","family":"Karim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2162-3367","authenticated-orcid":false,"given":"Hasan","family":"Iqbal","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1902-9877","authenticated-orcid":false,"given":"Athanasios V.","family":"Vasilakos","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4335-6915","authenticated-orcid":false,"given":"Christos","family":"Emmanouilidis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.inffus.2026.104444_bib0001","doi-asserted-by":"crossref","unstructured":"A. Yehudai, L. Eden, A. Li, G. Uziel, Y. Zhao, R. Bar-Haim, A. Cohan, M. Shmueli-Scheuer, Survey on evaluation of llm-based agents, arXiv: 2503.16416(2025).","DOI":"10.18653\/v1\/2026.findings-acl.1330"},{"key":"10.1016\/j.inffus.2026.104444_bib0002","unstructured":"Achim T., Best A., Der K., F\u00e9d\u00e9rico M., Gukov S., Halpern-Leister D., Henningsgard K., Kudryashov Y., Meiburg A., Michelsen M., et al., Aristotle: imo-level automated theorem proving, arXiv: 2510.01346, 2025."},{"key":"10.1016\/j.inffus.2026.104444_bib0003","unstructured":"K. Feng, D.W. McDonald, A.X. Zhang, Levels of autonomy for AI agents, 2025, (Essays and Scholarship, 25-15 Knight First Amend. Inst.). Accessed: September 30, 2025, https:\/\/knightcolumbia.org\/content\/levels-of-autonomy-for-ai-agents-1."},{"key":"10.1016\/j.inffus.2026.104444_bib0004","doi-asserted-by":"crossref","first-page":"141","DOI":"10.1016\/j.cogsys.2006.07.004","article-title":"Cognitive architectures: research issues and challenges","volume":"10","author":"Langley","year":"2009","journal-title":"Cogn. Syst. Res."},{"issue":"1","key":"10.1016\/j.inffus.2026.104444_bib0005","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1080\/0952813X.2012.661236","article-title":"Psychologically realistic cognitive agents: taking human cognition seriously","volume":"25","author":"Sun","year":"2013","journal-title":"J. Exp. Theor. Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104444_bib0006","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1016\/j.cogsys.2020.08.010","article-title":"Toward ethical cognitive architectures for the development of artificial moral agents","volume":"64","author":"Cervantes","year":"2020","journal-title":"Cogn. Syst. Res."},{"key":"10.1016\/j.inffus.2026.104444_bib0007","unstructured":"Artacho B., Savakis A., Omnipose: a multi-scale framework for multi-person pose estimation, arXiv: 2103.10180, 2021."},{"key":"10.1016\/j.inffus.2026.104444_bib0008","unstructured":"Chen L., Gu J., Huang L., Huang W., Jiang Z., Jie A., Jin X., Jin X., Li C., Ma K., et al., Seed-prover: deep and broad reasoning for automated theorem proving, arXiv: 2507.23726, 2025."},{"key":"10.1016\/j.inffus.2026.104444_sbref0004a","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"Scienceagentbench: toward rigorous assessment of language agents for data-driven scientific discovery","author":"Chen","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0005a","series-title":"The Fourteenth International Conference on Learning Representations","article-title":"Memory, benchmark & robots: a benchmark for solving complex tasks with reinforcement learning","author":"Cherepanov","year":"2026"},{"key":"10.1016\/j.inffus.2026.104444_sbref0006a","series-title":"39th Conference on Neural Information Processing Systems (NeurIPS 2025) Workshop: MATH-AI","article-title":"R-zero: self-evolving reasoning llm from zero data","author":"Huang","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0007a","series-title":"Forty-second International Conference on Machine Learning","article-title":"ITBench: evaluating AI agents across diverse real-world IT automation tasks","author":"Jha","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0008a","series-title":"Finding the Frame: an RLCWorkshop for Examining Conceptual Frameworks","article-title":"The need for a big world simulator: a scientific challenge for continual learning","author":"Kumar","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0014","unstructured":"Li K., Jiang M., Fu D., Wu Y., Hu X., Wang D., Liu P., Datasetresearch: benchmarking agent systems for demand-driven dataset discovery, arXiv: 2508.06960, 2025."},{"key":"10.1016\/j.inffus.2026.104444_sbref0010a","series-title":"Forty-second International Conference on Machine Learning","article-title":"Reflection-bench: evaluating epistemic agency in large language models","author":"Li","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0016","unstructured":"Lin J., Xia Y., Zhang J., Yan K., Lu L., Luo J., Zhang L., Ct-glip: 3d grounded language-image pretraining with ctscans and radiology reports for full-body scenarios, arXiv: 2404.15272, 2024."},{"key":"10.1016\/j.inffus.2026.104444_bib0017","unstructured":"Padigela H., Shah C., Juyal D., Ml-dev-bench: comparative analysis of ai agents on ml development workflows, arXiv: 2502.00964, 2025."},{"key":"10.1016\/j.inffus.2026.104444_sbref0013a","series-title":"The Eleventh International Conference on Learning Representations","article-title":"Evaluating long-term memory in 3d mazes","author":"Pa\u0161ukonis","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_bib0019","unstructured":"Wang H., Zou H., Song H., Feng J., Fang J., Lu J., Liu L., Luo Q., Liang S., Huang S., et al., Ui-tars-2 technical report: Advancing gui agent with multi-turn reinforcement learning, arXiv: 2509.02544, 2025."},{"key":"10.1016\/j.inffus.2026.104444_bib0020","unstructured":"Wang W., Zhang D., Feng T., Wang B., Tang J., Battleagentbench: a benchmark for evaluating cooperation and competition capabilities of language models in multi-agent systems, arXiv: 2408.15971, 2024."},{"key":"10.1016\/j.inffus.2026.104444_sbref0016a","series-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems","article-title":"A-mem: agentic memory for LLM agents","author":"Xu","year":"2026"},{"key":"10.1016\/j.inffus.2026.104444_sbref0017a","series-title":"Forty-second International Conference on Machine Learning","article-title":"Embodiedbench: comprehensive benchmarking multimodal large language models for vision-driven embodied agents","author":"Yang","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0023","unstructured":"Ye D., Zhou F., Lv J., Ma J., Zhang J., Lv J., Li J., Deng M., Yang M., Fu Q., et al., Yan: foundational interactive video generation, arXiv: 2508.08601, 2025."},{"key":"10.1016\/j.inffus.2026.104444_bib0024","unstructured":"Zhang T., Eysenbach B., Salakhutdinov R., Levine S., Gonzalez J.E., C-planning: an automatic curriculum for learning goalreaching tasks, arXiv: 2110.12080, 2021."},{"key":"10.1016\/j.inffus.2026.104444_bib0025","unstructured":"Zhang Z., He Y., Sun Y., Shi J., Liu L., Nie Q., Roboactclip: video-driven pre-training of atomic action understanding for robotics, arXiv: 2504.02069, 2025."},{"key":"10.1016\/j.inffus.2026.104444_bib0026","first-page":"1","article-title":"MIDCA version 1.5: user manual and tutorial for the metacognitive integrated dual-Cycle architecture","author":"Dannenhauer","year":"2020","journal-title":"Tech. Rep. No. COLAB2-TR-5"},{"issue":"3","key":"10.1016\/j.inffus.2026.104444_bib0027","first-page":"3961","article-title":"A review of AI-driven automation technologies: latest taxonomies, existing challenges, and future prospects","volume":"84","author":"Jin","year":"2025","journal-title":"Comput. Mater. Contin."},{"key":"10.1016\/j.inffus.2026.104444_bib0028","unstructured":"D.-A. Team, DeepSeek-R1: incentivizing reasoning capability in LLMs via reinforcement learning, 2025. https:\/\/arxiv.org\/abs\/2501.12948. 2501.12948."},{"key":"10.1016\/j.inffus.2026.104444_bib0029","unstructured":"P. Team, Phi-3 technical report: a highly capable language model locally on your phone, 2024. 2404.14219."},{"key":"10.1016\/j.inffus.2026.104444_bib0030","unstructured":"H. Touvron, T. Lavril, G. Izacard, X. Martinet, M.-A. Lachaux, T. Lacroix, B. Rozi\u00e8re, N. Goyal, E. Hambro, F. Azhar, A. Rodriguez, A. Joulin, E. Grave, G. Lample, LLaMA: open and efficient foundation language models, 2023. 2302.13971."},{"key":"10.1016\/j.inffus.2026.104444_bib0031","unstructured":"H. Yang, S. Yue, Y. He, Auto-gpt for online decision making: Benchmarks and additional opinions, arXiv: 2306.02224(2023)."},{"key":"10.1016\/j.inffus.2026.104444_bib0032","series-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 3: System Demonstrations)","first-page":"185","article-title":"Agentquest: a modular benchmark framework to measure progress and improve llm agents","author":"Gioacchini","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0033","unstructured":"L. Chen, F. Yan, Y. Zhong, S. Chen, Z. Jie, L. Ma, Mindbench: a comprehensive benchmark for mind map structure recognition and analysis, arXiv: 2407.02842(2024)."},{"key":"10.1016\/j.inffus.2026.104444_sbref0014","series-title":"Forty-second International Conference on Machine Learning","article-title":"Minerva: a programmable memory test benchmark for language models","author":"Xia","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0035","unstructured":"D. Rein, J. Becker, A. Deng, S. Nix, C. Canal, D. O\u2019Connel, P. Arnott, R. Bloom, T. Broadley, K. Garcia, et al., HCAST: human-calibrated autonomy software tasks, arXiv: 2503.17354(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0036","doi-asserted-by":"crossref","unstructured":"Ritter F.E., Tehranchi F., Oury J.D., Act-r: a cognitive architecture for modeling cognition, 2019, 10.1002\/wcs.1488.","DOI":"10.1002\/wcs.1488"},{"key":"10.1016\/j.inffus.2026.104444_bib0037","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","volume":"33","author":"Lewis","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"1","key":"10.1016\/j.inffus.2026.104444_bib0038","doi-asserted-by":"crossref","DOI":"10.1155\/int\/5920142","article-title":"Veracity-oriented context-aware large language models\u2013based prompting optimization for fake news detection","volume":"2025","author":"Jin","year":"2025","journal-title":"Int. J. Intell. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0039","series-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing","first-page":"32001","article-title":"RAG+: enhancing retrieval-augmented generation with application-aware reasoning","author":"Wang","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0040","doi-asserted-by":"crossref","unstructured":"F. Wang, B. Chen, K. Xu, B. Tang, F. Xiong, Z. Li, Text2Mem: a unified memory operation language for memory operating system, arXiv: 2509.11145(2025b).","DOI":"10.18653\/v1\/2026.findings-acl.100"},{"issue":"6","key":"10.1016\/j.inffus.2026.104444_bib0041","doi-asserted-by":"crossref","first-page":"140","DOI":"10.1007\/s10994-025-06767-4","article-title":"Developing safe and responsible large language model: can we balance bias reduction and language understanding?","volume":"114","author":"Raza","year":"2025","journal-title":"Mach. Learn."},{"key":"10.1016\/j.inffus.2026.104444_sbref0021","series-title":"NeurIPS 2025 Workshop on Evaluating the Evolving LLM Lifecycle: Benchmarks, Emergent Abilities, and Scaling","article-title":"Human-centric framework for large multimodal models evaluation","author":"Raza","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0043","series-title":"Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 2","first-page":"6129","article-title":"Evaluation and benchmarking of llm agents: a survey","author":"Mohammadi","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0044","unstructured":"J. Haase, S. Pokutta, Beyond static responses: multi-agent LLM systems as a new paradigm for social science research, arXiv: 2506.01839(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0045","doi-asserted-by":"crossref","first-page":"4540","DOI":"10.52202\/079017-0148","article-title":"Taskbench: benchmarking large language models for task automation","volume":"37","author":"Shen","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"1","key":"10.1016\/j.inffus.2026.104444_bib0046","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1007\/s43681-026-00990-y","article-title":"Adaptive monitoring and real-world evaluation of agentic AI systems","volume":"6","author":"Shukla","year":"2026","journal-title":"AI Ethics"},{"key":"10.1016\/j.inffus.2026.104444_bib0047","series-title":"Findings of the Association for Computational Linguistics: NAACL 2025","first-page":"8038","article-title":"Llm-coordination: evaluating and analyzing multi-agent coordination abilities in large language models","author":"Agashe","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0048","unstructured":"S. Yin, X. Pang, Y. Ding, M. Chen, Y. Bi, Y. Xiong, W. Huang, Z. Xiang, J. Shao, S. Chen, SafeAgentBench: a benchmark for safe task planning of embodied LLM agents, 2026. https:\/\/openreview.net\/forum?id=BFb4ACHayj."},{"key":"10.1016\/j.inffus.2026.104444_bib0049","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1613\/jair.1.18675","article-title":"Agentic large language models, a survey","volume":"84","author":"Plaat","year":"2025","journal-title":"J. Artif. Intell. Res."},{"key":"10.1016\/j.inffus.2026.104444_bib0050","series-title":"Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 2","first-page":"6216","article-title":"A survey on trustworthy llm agents: threats and countermeasures","author":"Yu","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0051","doi-asserted-by":"crossref","first-page":"18912","DOI":"10.1109\/ACCESS.2025.3532853","article-title":"Agentic ai: autonomous intelligence for complex goals\u2013a comprehensive survey","volume":"13","author":"Acharya","year":"2025","journal-title":"IEEE Access"},{"key":"10.1016\/j.inffus.2026.104444_bib0052","first-page":"1","article-title":"AI agents and agentic systems: a multi-expert analysis","volume":"65","author":"Hughes","year":"2025","journal-title":"J. Comput. Inf. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0053","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2025.128404","article-title":"AgentAI: a comprehensive survey on autonomous agents in distributed AI for industry 4.0","volume":"291","author":"Piccialli","year":"2025","journal-title":"Expert Syst. Appl."},{"issue":"9","key":"10.1016\/j.inffus.2026.104444_bib0054","doi-asserted-by":"crossref","first-page":"404","DOI":"10.3390\/fi17090404","article-title":"The rise of agentic AI: a review of definitions, frameworks, architectures, applications, evaluation metrics, and challenges","volume":"17","author":"Bandi","year":"2025","journal-title":"Future Internet"},{"key":"10.1016\/j.inffus.2026.104444_bib0055","first-page":"69","article-title":"Agentic AI: the age of reasoning\u2013a review","volume":"5","author":"Nisa","year":"2025","journal-title":"J. Autom. Intel."},{"key":"10.1016\/j.inffus.2026.104444_bib0056","unstructured":"H. Derouiche, Z. Brahmi, H. Mazeni, Agentic AI frameworks: architectures, protocols, and design challenges, arXiv: 2508.10146(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0057","series-title":"Artificial Intelligence: a Modern Approach","author":"Russell","year":"2021"},{"key":"10.1016\/j.inffus.2026.104444_bib0058","series-title":"An Introduction to MultiAgent Systems","author":"Wooldridge","year":"2009"},{"key":"10.1016\/j.inffus.2026.104444_bib0059","unstructured":"E. Sherman, S. Shattuck, N. Singh, I. Eisenberg, From assistant to agent: navigating the governance challenges of increasingly autonomous AI, 2025, (Credo AI Whitepaper). https:\/\/www.credo.ai\/recourseslongform\/from-assistant-to-agent-navigating-the-governance-challenges-of-increasingly-autonomous-ai."},{"key":"10.1016\/j.inffus.2026.104444_sbref0039","article-title":"Agentic AI: a governance wake-up call","author":"Ahmed","year":"2025","journal-title":"Dir. Mag. Online Exclus."},{"key":"10.1016\/j.inffus.2026.104444_bib0061","unstructured":"S. Marks, J. Treutlein, T. Bricken, J. Lindsey, J. Marcus, S. Mishra-Sharma, D. Ziegler, E. Ameisen, J. Batson, T. Belonax, B. Chen, H. Cunningham, C. Denison, S. Golechha, A. Khan, J. Kirchner, J. Leike, A. Meek, K. Nishimura-Gasparian, E. Ong, C. Olah, A. Pearce, F. Roger, J. Salle, A. Shih, M. Tong, D. Thomas, K. Rivoire, A. Jermyn, M. MacDiarmid, T. Henighan, E. Hubinger, Auditing language models for hidden objectives, 2025. arXiv: 2503.10965."},{"key":"10.1016\/j.inffus.2026.104444_bib0062","unstructured":"M. Andriushchenko, A. Souly, M. Dziemian, D. Duenas, M. Lin, J. Wang, D. Hendrycks, A. Zou, Z. Kolter, M. Fredrikson, et al., Agentharm: a benchmark for measuring harmfulness of llm agents, arXiv: 2410.09024(2024)."},{"key":"10.1016\/j.inffus.2026.104444_bib0063","doi-asserted-by":"crossref","first-page":"82895","DOI":"10.52202\/079017-2636","article-title":"Agentdojo: a dynamic environment to evaluate prompt injection attacks and defenses for llm agents","volume":"37","author":"Debenedetti","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0064","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"35680","article-title":"Is-bench: evaluating interactive safety of vlm-driven embodied agents in daily household tasks","volume":"40","author":"Lu","year":"2026"},{"key":"10.1016\/j.inffus.2026.104444_sbref0044","article-title":"Advancing embodied agent security: from safety benchmarks to input moderation","author":"Wang","year":"2025","journal-title":"Proc. Thirty-Fourth Int. Jt. Conf. Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104444_bib0066","unstructured":"I. Uchendu, J. Jabbour, K.V.d. Berghe, J. Runevic, M. Stewart, J. Ma, S. Krishnan, I. Gur, A. Huang, C. Bishop, et al., A2Perf: real-world autonomous agents benchmark, arXiv: 2503.03056(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0067","doi-asserted-by":"crossref","unstructured":"Uddin A., Salam H., Beyond rule-based context awareness: large language models as adaptive cognitive layers in cyber-physical systems 6(1) 2025, pp. 140\u2013147https:\/\/www.ijcai.org\/aaai.org.","DOI":"10.1609\/aaaiss.v6i1.36045"},{"key":"10.1016\/j.inffus.2026.104444_sbref0046","series-title":"The Twelfth International Conference on Learning Representations (ICLR)","article-title":"AgentBench: evaluating LLMs as agents","author":"Liu","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0069","unstructured":"T. Bogavelli, R. Sharma, H. Subramani, AgentArch: a comprehensive benchmark to evaluate agent architectures in enterprise, arXiv: 2509.10769(2025)."},{"key":"10.1016\/j.inffus.2026.104444_sbref0048","series-title":"The Twelfth International Conference on Learning Representations","article-title":"CivRealm: a learning and reasoning odyssey in civilization for decision-Making agents","author":"Qi","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0071","series-title":"Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 1","first-page":"2652","article-title":"Realm-bench: a benchmark for evaluating multi-agent systems on real-world, dynamic planning and scheduling tasks","author":"Geng","year":"2026"},{"key":"10.1016\/j.inffus.2026.104444_sbref0050","series-title":"Forty-second International Conference on Machine Learning","article-title":"SWE-lancer: can frontier LLMs earn $1 million from real-world freelance software engineering?","author":"Miserendino","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0073","doi-asserted-by":"crossref","first-page":"52040","DOI":"10.52202\/079017-1650","article-title":"Osworld: benchmarking multimodal agents for open-ended tasks in real computer environments","volume":"37","author":"Xie","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0074","unstructured":"H.S. Zheng, S. Mishra, H. Zhang, X. Chen, M. Chen, A. Nova, L. Hou, H.-T. Cheng, Q.V. Le, E.H. Chi, et al., Natural plan: benchmarking llms on natural language planning, arXiv: 2406.04520(2024)."},{"key":"10.1016\/j.inffus.2026.104444_bib0075","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2024","first-page":"10883","article-title":"Flowbench: revisiting and benchmarking workflow-guided planning for llm-based agents","author":"Xiao","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0076","doi-asserted-by":"crossref","first-page":"107039","DOI":"10.52202\/079017-3398","article-title":"Streambench: towards benchmarking continuous improvement of language agents","volume":"37","author":"Wu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_sbref0055","series-title":"Forty-second International Conference on Machine Learning","article-title":"Reflection-bench: evaluating epistemic agency in large language models","author":"Li","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0078","series-title":"Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 2","first-page":"5482","article-title":"DCA-bench: a benchmark for dataset curation agents","author":"Huang","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0079","unstructured":"K. Li, M. Jiang, D. Fu, Y. Wu, X. Hu, D. Wang, P. Liu, Datasetresearch: benchmarking agent systems for demand-driven dataset discovery, arXiv: 2508.06960(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0080","series-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"8580","article-title":"Multiagentbench: evaluating the collaboration and competition of llm agents","author":"Zhu","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0081","unstructured":"W. Wang, D. Zhang, T. Feng, B. Wang, J. Tang, Battleagentbench: a benchmark for evaluating cooperation and competition capabilities of language models in multi-agent systems, arXiv: 2408.15971(2024)."},{"key":"10.1016\/j.inffus.2026.104444_sbref0060","article-title":"CREW-wildfire: benchmarking agentic multi-agent collaborations at scale","author":"Hyun","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.inffus.2026.104444_bib0083","series-title":"2024IEEE International Conference on Robotics and Automation (ICRA)","first-page":"286","article-title":"Roco: dialectic multi-robot collaboration with large language models","author":"Mandi","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0084","series-title":"Findings of the Association for Computational Linguistics: ACL 2024","first-page":"16290","article-title":"Villageragent: a graph-based multi-agent framework for coordinating complex task dependencies in minecraft","author":"Dong","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0085","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"13055","article-title":"Llmarena: assessing capabilities of large language models in dynamic multi-agent environments","author":"Chen","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0086","series-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing","first-page":"4922","article-title":"Collab-overcooked: benchmarking and evaluating large language models as collaborative agents","author":"Sun","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0087","series-title":"Findings of the Association for Computational Linguistics: NAACL 2024","first-page":"3154","article-title":"Mindagent: emergent gaming interaction","author":"Gong","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0066","series-title":"The Twelfth International Conference on Learning Representations","article-title":"ToolLLM: facilitating large language models to master 16000+ real-world APIs","author":"Qin","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0067","series-title":"The Twelfth International Conference on Learning Representations","article-title":"MetaTool benchmark for large language models: deciding whether to use tools and which to use","author":"Huang","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0090","doi-asserted-by":"crossref","first-page":"126544","DOI":"10.52202\/079017-4020","article-title":"Gorilla: large language model connected with massive apis","volume":"37","author":"Patil","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_sbref0069","article-title":"Tool-planner: task planning with clusters across multiple tools","author":"Liu","year":"2024","journal-title":"ICLR"},{"key":"10.1016\/j.inffus.2026.104444_bib0092","doi-asserted-by":"crossref","first-page":"100428","DOI":"10.52202\/079017-3188","article-title":"Embodied agent interface: benchmarking llms for embodied decision making","volume":"37","author":"Li","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0093","doi-asserted-by":"crossref","first-page":"20744","DOI":"10.52202\/068431-1508","article-title":"Webshop: towards scalable real-world web interaction with grounded language agents","volume":"35","author":"Yao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_sbref0072","series-title":"The Twelfth International Conference on Learning Representations (ICLR)","article-title":"WebArena: a realistic web environment for building autonomous agents","author":"Zhou","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0095","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"16022","article-title":"AppWorld: a controllable world of apps and people for benchmarking interactive coding agents","author":"Trivedi","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0096","series-title":"Computer Vision \u2013 ECCV 2024","first-page":"161","article-title":"OmniACT: a dataset and benchmark for enabling multimodal generalist autonomous agents for desktop and web","volume":"15126","author":"Kapoor","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0097","series-title":"Proceedings of the International Conference on Automated Planning and Scheduling","first-page":"250","article-title":"Automating the generation of prompts for llm-based action choice in pddl planning","volume":"35","author":"Stein","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0098","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"26559","article-title":"Acpbench: reasoning about action, change, and planning","volume":"39","author":"Kokel","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0099","series-title":"European Conference on Computer Vision","first-page":"18","article-title":"m & m\u2019s: a benchmark to evaluate tool-use for m ulti-step m ulti-modal tasks","author":"Ma","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0100","unstructured":"Y. Nakajima, BabyAGI: an autonomous task management system, 2023, Original release date: March 28, 2023. Accessed: [Insert Current Date], https:\/\/github.com\/yoheinakajima\/babyagi."},{"issue":"6","key":"10.1016\/j.inffus.2026.104444_bib0101","doi-asserted-by":"crossref","DOI":"10.1007\/s11704-024-40231-1","article-title":"A survey on large language model based autonomous agents","volume":"18","author":"Wang","year":"2024","journal-title":"Front. Comput. Sci."},{"key":"10.1016\/j.inffus.2026.104444_bib0102","series-title":"Proceedings of the 36th Annual Acm Symposium on User Interface Software and Technology","first-page":"1","article-title":"Generative agents: interactive simulacra of human behavior","author":"Park","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_bib0103","series-title":"International Conference on Learning Representations (ICLR)","article-title":"React: synergizing reasoning and acting in language models","author":"Yao","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_sbref0082","series-title":"Advances in Neural Information Processing Systems 36 (NeurIPS 2023)","article-title":"Toolformer: language models can teach themselves to use tools","author":"Schick","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_sbref0083","series-title":"Advances in Neural Information Processing Systems 36 (NeurIPS 2023)","article-title":"Reflexion: language agents with verbal reinforcement learning","author":"Shinn","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_bib0106","first-page":"11375","article-title":"Position: Building guardrails for large language models requires systematic design","author":"Dong","year":"2024","journal-title":"Proc. 41st Int. Conf. Mach. Learn. Vienna, Austria. PMLR 235, 2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0107","doi-asserted-by":"crossref","first-page":"27730","DOI":"10.52202\/068431-2011","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_sbref0086","series-title":"The Eleventh International Conference on Learning Representations (ICLR)","article-title":"Self-consistency improves chain of thought reasoning in language models","author":"Wang","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_sbref0087","article-title":"Program of thoughts prompting: disentangling computation from reasoning for numerical reasoning tasks","author":"Chen","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.inffus.2026.104444_bib0110","article-title":"Large language model based multi-agents: a survey of progress and challenges","author":"Guo","year":"2024","journal-title":"Proc. Thirty-Third Int. Jt. Conf. Artif. Intell. Surv. Track"},{"key":"10.1016\/j.inffus.2026.104444_bib0111","article-title":"ViLBias: detecting and reasoning about bias in multimodal content","author":"Raza","year":"2026","journal-title":"Int. Assoc. Safe Ethical AI"},{"key":"10.1016\/j.inffus.2026.104444_bib0112","unstructured":"X. Qu, A. Damoah, J. Sherwood, P. Liu, C.S. Jin, L. Chen, M. Shen, N. Aleisa, Z. Hou, C. Zhang, et al., A comprehensive review of AI agents: transforming possibilities in technology and beyond, arXiv: 2508.11957(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0113","unstructured":"P. Belcak, G. Heinrich, S. Diao, Y. Fu, X. Dong, S. Muralidharan, Y.C. Lin, P. Molchanov, Small language models are the future of agentic AI, arXiv: 2506.02153(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0114","unstructured":"K. Vongthongsri, LLM agent evaluation: assessing tool use, task completion, agentic reasoning, and more, 2025. https:\/\/www.confident-ai.com\/blog\/llm-agent-evaluation-complete-guide#evaluating-tool-use."},{"key":"10.1016\/j.inffus.2026.104444_bib0115","unstructured":"Q. Xu, F. Hong, B. Li, C. Hu, Z. Chen, J. Zhang, On the tool manipulation capability of open-source large language models, arXiv: 2305.16504(2023)."},{"issue":"8","key":"10.1016\/j.inffus.2026.104444_bib0116","doi-asserted-by":"crossref","DOI":"10.1007\/s11704-024-40678-2","article-title":"Tool learning with large language models: a survey","volume":"19","author":"Qu","year":"2025","journal-title":"Front. Comput. Sci."},{"key":"10.1016\/j.inffus.2026.104444_bib0117","doi-asserted-by":"crossref","first-page":"416","DOI":"10.1109\/TLT.2025.3561332","article-title":"Eduplanner: llm-based multi-agent systems for customized and intelligent instructional design","volume":"18","author":"Zhang","year":"2025","journal-title":"IEEE Trans. Learn. Technol."},{"key":"10.1016\/j.inffus.2026.104444_sbref0096","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"Scienceagentbench: toward rigorous assessment of language agents for data-driven scientific discovery","author":"Chen","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0097","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"tau-bench: a benchmark for tool-agent-user interaction in real-world domains","author":"Yao","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0120","unstructured":"D. Deshpande, V. Gangal, H. Mehta, J. Krishnan, A. Kannappan, R. Qian, TRAIL: trace reasoning and agentic issue localization, arXiv: 2505.08638(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0121","unstructured":"A. Moteki, S. Masui, F. Yang, Y. Song, Y. Bisk, G. Neubig, I. Kusajima, Y. Watanabe, H. Ishida, J. Takahashi, et al., FieldWorkArena: agentic AI benchmark for real field work tasks, arXiv: 2505.19662(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0122","doi-asserted-by":"crossref","first-page":"180815","DOI":"10.1109\/ACCESS.2024.3509353","article-title":"Exploring bias and prediction metrics to characterise the fairness of machine learning for equity-centered public health decision-making: a narrative review","volume":"12","author":"Raza","year":"2024","journal-title":"IEEE Access."},{"key":"10.1016\/j.inffus.2026.104444_sbref0101","series-title":"The Thirteenth International Conference on Learning Representations (ICLR)","article-title":"MLE-bench: evaluating machine learning agents on machine learning engineering","author":"Chan","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0124","series-title":"Proceedings of the 41st International Conference on Machine Learning","article-title":"MLAgentBench: evaluating language agents on machine learning experimentation","author":"Huang","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0125","unstructured":"IBM, Agentic AI evaluation, 2025, (https:\/\/www.ibm.com\/docs\/en\/watsonx\/saas?topic=sdk-agentic-ai-evaluation). Accessed: 5 October 2025."},{"key":"10.1016\/j.inffus.2026.104444_bib0126","unstructured":"S. Kapoor, B. Stroebl, P. Kirgis, N. Nadgir, Z.S. Siegel, B. Wei, T. Xue, Z. Chen, F. Chen, S. Utpala, F. Ndzomga, D. Oruganty, S. Luskin, K. Liu, B. Yu, A. Arora, D. Hahm, H. Trivedi, H. Sun, J. Lee, T. Jin, Y. Mai, Y. Zhou, Y. Zhu, R. Bommasani, D. Kang, D. Song, P. Henderson, Y. Su, P. Liang, A. Narayanan, Holistic Agent leaderboard: the missing infrastructure for AI agent evaluation, 2025, (https:\/\/github.com\/princeton-pli\/hal-harness)."},{"key":"10.1016\/j.inffus.2026.104444_sbref0105","series-title":"Technical Report","article-title":"Artificial Intelligence Risk Management Framework (AI RMF 1.0)","author":"Standards","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_bib0128","unstructured":"ISO\/IEC, Information technology \u2013 artificial intelligence \u2013 management system, 2023, (ISO\/IEC 42001:2023). International Organization for Standardization, Geneva, Switzerland, 51 pages."},{"key":"10.1016\/j.inffus.2026.104444_sbref0107","series-title":"Technical Report","article-title":"The Minimum Elements for a Software Bill of Materials (SBOM)","author":"Telecommunications","year":"2021"},{"key":"10.1016\/j.inffus.2026.104444_bib0130","first-page":"74325","article-title":"Agentboard: an analytical evaluation board of multi-turn llm agents","volume":"37","author":"Chang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_sbref0109","series-title":"International Conference on Learning Representations","article-title":"Alfworld: aligning text and embodied environments for interactive learning","author":"Shridhar","year":"2021"},{"key":"10.1016\/j.inffus.2026.104444_sbref0110","series-title":"Advances in Neural Information Processing Systems 36 (NeurIPS 2023) Datasets and Benchmarks Track","first-page":"28091","article-title":"Mind2Web: towards a generalist agent for the web","author":"Deng","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_bib0133","unstructured":"I. Levy, B. Wiesel, S. Marreed, A. Oved, A. Yaeli, S. Shlomov, St-webagentbench: a benchmark for evaluating safety and trustworthiness in web agents, arXiv: 2410.06703(2024)."},{"key":"10.1016\/j.inffus.2026.104444_bib0134","first-page":"1","article-title":"The browsergym ecosystem for web agent research","author":"Chezelles","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.inffus.2026.104444_sbref0113","series-title":"Agentic Markets Workshop at ICML 2024","article-title":"WebCanvas: benchmarking web agents in online environments","author":"Pan","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0136","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"881","article-title":"Visualwebarena: evaluating multimodal agents on realistic visual web tasks","author":"Koh","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0137","series-title":"Findings of the Association for Computational Linguistics: ACL 2025","first-page":"13682","article-title":"Mmina: benchmarking multihop multimodal internet agents","author":"Tian","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0138","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"8938","article-title":"Assistantbench: can web agents solve realistic and time-consuming tasks?","author":"Yoran","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0117","series-title":"The Twelfth International Conference on Learning Representations (ICLR)","article-title":"SWE-bench: can language models resolve real-world github issues?","author":"Jimenez","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0118","article-title":"CORE-bench: fostering the credibility of published research through a computational reproducibility agent benchmark","author":"Siegel","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.inffus.2026.104444_sbref0119","series-title":"Proceedings of the 42nd International Conference on Machine Learning (ICML)","article-title":"Paperbench: evaluating AI\u2019s ability to replicate AI research","author":"Starace","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0120","series-title":"The 2023 Conference on Empirical Methods in Natural Language Processing","article-title":"API-bank: a comprehensive benchmark for tool-augmented LLMs","author":"Li","year":"2023"},{"key":"10.1016\/j.inffus.2026.104444_bib0143","series-title":"Findings of the Association for Computational Linguistics: ACL 2024","first-page":"2108","article-title":"SocialBench: sociality evaluation of role-playing conversational agents","author":"Chen","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0144","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"3894","article-title":"TimeArena: shaping efficient multitasking language agents in a time-aware simulation","author":"Zhang","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0123","series-title":"The Twelfth International Conference on Learning Representations (ICLR)","article-title":"GAIA: a benchmark for general AI assistants","author":"Mialon","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0124","series-title":"The Twelfth International Conference on Learning Representations","article-title":"MINT: evaluating LLMs in multi-turn interaction with tools and language feedback","author":"Wang","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0147","unstructured":"Y. Zhou, S. Jiang, Y. Tian, J. Weston, S. Levine, S. Sukhbaatar, X. Li, Sweet-rl: training multi-turn llm agents on collaborative reasoning tasks, arXiv: 2503.15478(2025)."},{"key":"10.1016\/j.inffus.2026.104444_sbref0126","series-title":"The Twelfth International Conference on Learning Representations","article-title":"Identifying the risks of LM agents with an LM-Emulated sandbox","author":"Ruan","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0149","doi-asserted-by":"crossref","first-page":"5338","DOI":"10.52202\/079017-0173","article-title":"Can graph learning improve planning in LLM-based agents?","volume":"37","author":"Wu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0150","series-title":"Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","first-page":"6061","article-title":"Understanding the weakness of large language model agents within a complex android environment","author":"Xing","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0129","series-title":"The Twelfth International Conference on Learning Representations","article-title":"SOTOPIA: interactive evaluation for social intelligence in language agents","author":"Zhou","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0130","series-title":"Forty-second International Conference on Machine Learning","article-title":"RE-bench: evaluating frontier AI R&D capabilities of language model agents against human experts","author":"Wijk","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0153","doi-asserted-by":"crossref","first-page":"71","DOI":"10.1016\/j.aiopen.2026.02.006","article-title":"Trism for agentic ai: a review of trust, risk, and security management in llm-based agentic multi-agent systems","volume":"7","author":"Raza","year":"2026","journal-title":"AI Open."},{"key":"10.1016\/j.inffus.2026.104444_bib0154","doi-asserted-by":"crossref","unstructured":"S. Raza, R. Qureshi, A. Zahid, S. Kamawal, F. Sadak, J. Fioresi, M. Saeed, R. Sapkota, A. Jain, A. Zafar, et al., Who is responsible? The data, models, users or regulations? a comprehensive survey on responsible generative ai for a sustainable future, arXiv: 2502.08650(2025).","DOI":"10.36227\/techrxiv.173834932.29831105\/v1"},{"key":"10.1016\/j.inffus.2026.104444_bib0155","doi-asserted-by":"crossref","first-page":"38975","DOI":"10.52202\/075280-1693","article-title":"Planbench: an extensible benchmark for evaluating large language models on planning and reasoning about change","volume":"36","author":"Valmeekam","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0156","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"12622","article-title":"Super: evaluating agents on setting up and executing tasks from research repositories","author":"Bogin","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0157","series-title":"2025 USENIX Annual Technical Conference (USENIX ATC 25)","first-page":"563","article-title":"{CLONE}: customizing {LLMs} for efficient {Latency-aware} inference at the edge","author":"Tian","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0136","series-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems","article-title":"Win fast or lose slow: balancing speed and accuracy in latency-sensitive decisions of LLMs","author":"Kang","year":"2026"},{"key":"10.1016\/j.inffus.2026.104444_bib0159","series-title":"Proceedings of the 62Nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"9510","article-title":"T-eval: evaluating the tool utilization capability of large language models step by step","author":"Chen","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_bib0160","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)","first-page":"951","article-title":"Easytool: enhancing llm-based agents with concise tool instruction","author":"Yuan","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0139","series-title":"Forty-first International Conference on Machine Learning","article-title":"GPTSwarm: language agents as optimizable graphs","author":"Zhuge","year":"2024"},{"key":"10.1016\/j.inffus.2026.104444_sbref0140","series-title":"Advances in Neural Information Processing Systems 37 (NeurIPS 2024)","article-title":"Graph edit distance with general costs using neural set divergence","author":"Jain","year":"2024"},{"issue":"1","key":"10.1016\/j.inffus.2026.104444_bib0163","doi-asserted-by":"crossref","first-page":"142","DOI":"10.1007\/s13278-024-01290-1","article-title":"FakeWatch: a framework for detecting fake news to ensure credible elections","volume":"14","author":"Raza","year":"2024","journal-title":"Soc. Netw. Anal. Min."},{"issue":"2","key":"10.1016\/j.inffus.2026.104444_bib0164","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-024-4222-0","article-title":"The rise and potential of large language model based agents: a survey","volume":"68","author":"Xi","year":"2025","journal-title":"Sci. China Inf. Sci."},{"key":"10.1016\/j.inffus.2026.104444_bib0165","unstructured":"M.S. Rashid, C. Bock, Y. Zhuang, A. Buchholz, T.B. Esler, S. Valentin, L. Franceschi, M. Wistuba, S. Prabhu Teja, W. Kim, A. Deoras, G. Zappella, L. Callot, SWE-PolyBench: a multi-language benchmark for repository level evaluation of coding agents, 2026. https:\/\/openreview.net\/forum?id=n577FC6CKk."},{"key":"10.1016\/j.inffus.2026.104444_sbref0144","series-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track","article-title":"Mind2Web 2: evaluating agentic search with agent-as-a-judge","author":"Gou","year":"2026"},{"key":"10.1016\/j.inffus.2026.104444_bib0167","series-title":"Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing (EMNLP)","first-page":"2369","article-title":"HotpotQA: a dataset for diverse, explainable multi-hop question answering","author":"Yang","year":"2018"},{"key":"10.1016\/j.inffus.2026.104444_bib0168","unstructured":"R. Peeters, A. Steiner, L. Schwarz, J.Y. Caspary, C. Bizer, WebMall\u2013a multi-shop benchmark for evaluating web agents, arXiv: 2508.13024(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0169","unstructured":"V. Nath, P. Raja, C. Yoon, S. Hendryx, Toolcomp: a multi-tool reasoning & process supervision benchmark, arXiv: 2501.01290(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0170","doi-asserted-by":"crossref","first-page":"132208","DOI":"10.52202\/079017-4202","article-title":"Chain of agents: large language models collaborating on long-context tasks","volume":"37","author":"Zhang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0171","unstructured":"Z. Ke, Y. Ming, A. Xu, R. Chin, X.-P. Nguyen, P. Jwalapuram, S. Yavuz, C. Xiong, S. Joty, MAS-Orchestra: understanding and improving multi-agent reasoning through holistic orchestration and controlled benchmarks, arXiv: 2601.14652(2026)."},{"issue":"1","key":"10.1016\/j.inffus.2026.104444_bib0172","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1007\/s41060-022-00359-4","article-title":"Dbias: detecting biases and ensuring fairness in news articles","volume":"17","author":"Raza","year":"2024","journal-title":"Int. J. Data Sci. Anal."},{"key":"10.1016\/j.inffus.2026.104444_sbref0151","series-title":"Technical Report","article-title":"Zero Trust Architecture","author":"Rose","year":"2020"},{"key":"10.1016\/j.inffus.2026.104444_bib0174","unstructured":"Open Policy Agent Project, Open policy agent: policy as code, 2024, (https:\/\/openpolicyagent.org\/docs\/). Accessed 2025-10-18."},{"key":"10.1016\/j.inffus.2026.104444_bib0175","unstructured":"SLSA Community, SLSA specification v1.0, 2023, (https:\/\/slsa.dev\/spec\/v1.0\/). Accessed 2025-10-18."},{"key":"10.1016\/j.inffus.2026.104444_sbref0154","series-title":"28th USENIX Security Symposium (USENIX Security 2019)","article-title":"in-toto: providing farm-to-table guarantees for bits and bytes","author":"Torres-Arias","year":"2019"},{"key":"10.1016\/j.inffus.2026.104444_sbref0155","series-title":"W3C Recommendation","article-title":"Verifiable Credentials Data Model 2.0","author":"Verifiable Credentials Working Group","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0178","unstructured":"European Union, Regulation (EU) 2024\/1689 of 13 June 2024 laying down harmonised rules on artificial intelligence (Artificial Intelligence Act), 2024, (Official Journal of the European Union, OJ L 2024\/1689, 12 July 2024). https:\/\/eur-lex.europa.eu\/eli\/reg\/2024\/1689\/oj\/eng."},{"key":"10.1016\/j.inffus.2026.104444_bib0179","unstructured":"E. Union, General data protection regulation, 2018. [Accessed 01-10-2024], https:\/\/gdpr-info.eu\/."},{"key":"10.1016\/j.inffus.2026.104444_bib0180","unstructured":"OWASP Foundation, OWASP top 10 for large language model applications (2025), 2025, (https:\/\/genai.owasp.org\/llm-top-10\/). Accessed 2025-10-18."},{"key":"10.1016\/j.inffus.2026.104444_bib0181","unstructured":"MITRE Corporation, ATLAS: adversarial threat landscape for artificial-intelligence systems, 2025, (https:\/\/atlas.mitre.org\/). Accessed 2025-10-18."},{"key":"10.1016\/j.inffus.2026.104444_bib0182","unstructured":"S. Thurgood, B. Beyer, D. Ferguson, Example error budget policy, 2018, (https:\/\/sre.google\/workbook\/error-budget-policy\/). In The Site Reliability Workbook, Google SRE."},{"key":"10.1016\/j.inffus.2026.104444_sbref0161","series-title":"Technical Report","article-title":"Responsible Scaling Policy, Version 2.1","author":"Anthropic","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0184","unstructured":"L.R. Lifshitz, R. Hung, BC tribunal confirms companies remain liable for information provided by AI chatbot, 2024. https:\/\/www.americanbar.org\/groups\/business_law\/resources\/business-law-today\/2024-february\/bc-tribunal-confirms-companies-remain-liable-information-provided-ai-chatbot\/."},{"key":"10.1016\/j.inffus.2026.104444_bib0185","unstructured":"Open Policy Agent Contributors, Open policy agent, 2026. Accessed: 2026-02-06, https:\/\/www.openpolicyagent.org\/."},{"key":"10.1016\/j.inffus.2026.104444_sbref0164","series-title":"NeurIPS 2025 Workshop on Bridging Language, Agent, and World Models for Reasoning and Planning","article-title":"The SWE-bench illusion: when state-of-the-art LLMs remember instead of reason","author":"Liang","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0187","unstructured":"A. Drouin, M. Gasse, M. Caccia, I.H. Laradji, M. Del Verme, T. Marty, L. Boisvert, M. Thakkar, Q. Cappart, D. Vazquez, et al., Workarena: how capable are web agents at solving common knowledge work tasks?, arXiv: 2403.07718(2024)."},{"key":"10.1016\/j.inffus.2026.104444_bib0188","doi-asserted-by":"crossref","first-page":"5996","DOI":"10.52202\/079017-0195","article-title":"Workarena++: towards compositional planning and reasoning-based common knowledge work tasks","volume":"37","author":"Boisvert","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.inffus.2026.104444_sbref0167","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"HELMET: how to evaluate long-context models effectively and thoroughly","author":"Yen","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0190","unstructured":"R. Qureshi, R. Sapkota, A. Shah, A. Muneer, A. Zafar, A. Vayani, M. Shoman, A. Eldaly, K. Zhang, F. Sadak, et al., Thinking beyond tokens: from brain-inspired intelligence to cognitive foundations for artificial general intelligence and its societal impact, arXiv: 2507.00951(2025)."},{"key":"10.1016\/j.inffus.2026.104444_bib0191","unstructured":"S. Raza, R. Qureshi, M. Lotif, A. Chadha, D. Pandya, C. Emmanouilidis, Just as humans need vaccines, so do models: model immunization to combat falsehoods, arXiv: 2505.17870(2025)."},{"issue":"3","key":"10.1016\/j.inffus.2026.104444_bib0192","doi-asserted-by":"crossref","first-page":"213","DOI":"10.1007\/s43681-021-00043-6","article-title":"Sustainable AI: AI for sustainability and the sustainability of AI","volume":"1","author":"Van Wynsberghe","year":"2021","journal-title":"AI Ethics"},{"key":"10.1016\/j.inffus.2026.104444_bib0193","series-title":"2025 IEEE Conference on Artificial Intelligence (CAI)","first-page":"370","article-title":"Optimizing large language models: metrics, energy efficiency, and case study insights","author":"Khan","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_bib0194","unstructured":"Khodaygani M., Ali A.T., Dohnke T., Groth T., Baake E., Leucker M., Russwinkel N., Cognitive modeling of agents: integrating emotions, goals, needs, and decision-making, Proceedings of the AAAI Symposium Series2025."},{"key":"10.1016\/j.inffus.2026.104444_bib0195","unstructured":"M.A. Gonz\u00e1lez-Santamarta, F.J. Rodr\u00edguez-Lera, \u00c1. M. Guerrero-Higueras, V. Matell\u00e1n-Olivera, Integration of large language models within cognitive architectures for autonomous robots (2023). arXiv: 2309.14945."},{"key":"10.1016\/j.inffus.2026.104444_bib0196","doi-asserted-by":"crossref","DOI":"10.1177\/29498732251377341","article-title":"Cognitive LLMs: toward human-like artificial intelligence by integrating cognitive architectures and large language models for manufacturing decision-Making","volume":"1","author":"Wu","year":"2025","journal-title":"Neurosymbolic Artif. Intell."},{"key":"10.1016\/j.inffus.2026.104444_bib0197","article-title":"Curiosity and affect-driven cognitive architecture for HRI","author":"Berto","year":"2025","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.inffus.2026.104444_bib0198","doi-asserted-by":"crossref","first-page":"98","DOI":"10.1080\/09515089.2013.828569","article-title":"\u201dScaffolding\u201d and \u201daffordance\u201d as integrative concepts in the cognitive sciences","volume":"27","author":"Estany","year":"2014","journal-title":"Philos. Psychol."},{"key":"10.1016\/j.inffus.2026.104444_bib0199","doi-asserted-by":"crossref","first-page":"22","DOI":"10.1109\/MIS.2017.3121556","article-title":"Interactive cognitive systems and social intelligence","volume":"32","author":"Langley","year":"2017","journal-title":"IEEE Intell. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0200","doi-asserted-by":"crossref","unstructured":"M.T. Cox, D. Dannenhauer, S. Kondrakunta, Goal operations for cognitive systems(2017). https:\/\/www.ijcai.org\/aaai.org.","DOI":"10.1609\/aaai.v31i1.11163"},{"key":"10.1016\/j.inffus.2026.104444_bib0201","doi-asserted-by":"crossref","first-page":"750","DOI":"10.1109\/TCDS.2021.3052548","article-title":"Morphological development in robotic learning: a survey","volume":"13","author":"Naya-Varela","year":"2021","journal-title":"IEEE Trans. Cogn. Dev. Syst."},{"key":"10.1016\/j.inffus.2026.104444_bib0202","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1080\/0952813X.2012.661236","article-title":"Psychologically realistic cognitive agents: taking human cognition seriously","volume":"25","author":"Sun","year":"2013","journal-title":"J. Exp. Theor. Artif. Intell."},{"issue":"3","key":"10.1016\/j.inffus.2026.104444_bib0203","doi-asserted-by":"crossref","first-page":"261","DOI":"10.1016\/j.websem.2003.11.006","article-title":"Agent-based service selection","volume":"1","author":"Sreenath","year":"2004","journal-title":"J. Web Semant."},{"key":"10.1016\/j.inffus.2026.104444_bib0204","doi-asserted-by":"crossref","first-page":"25","DOI":"10.1016\/j.cogsys.2017.05.005","article-title":"Evolution of the ICARUS cognitive architecture","volume":"48","author":"Choi","year":"2018","journal-title":"Cogn. Syst. Res."},{"key":"10.1016\/j.inffus.2026.104444_sbref00XX","article-title":"Hammerbench: Fine-grained function-calling evaluation in real mobile device scenarios","author":"Wang","year":"2025","journal-title":"Findings of the Association for Computational Linguistics"},{"key":"10.1016\/j.inffus.2026.104444_sbref0002c","series-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems Datasets and Benchmarks Track","article-title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","author":"Xu","year":"2026"},{"key":"10.1016\/j.inffus.2026.104444_bib0207","series-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)","first-page":"4805","article-title":"Mmevalpro: Calibrating multimodal benchmarks towards trustworthy and efficient evaluation","author":"Huang","year":"2025"},{"key":"10.1016\/j.inffus.2026.104444_sbref0002e","series-title":"NeurIPS 2024 Workshop on Open-World Agents","article-title":"Spa-bench: A comprehensive benchmark for smartphone agent evaluation","author":"Chen","year":"2024"}],"container-title":["Information Fusion"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526003246?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1566253526003246?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:58:13Z","timestamp":1783195093000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1566253526003246"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":208,"alternative-id":["S1566253526003246"],"URL":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104444","relation":{"has-preprint":[{"id-type":"doi","id":"10.36227\/techrxiv.176186841.18883348\/v1","asserted-by":"object"},{"id-type":"doi","id":"10.36227\/techrxiv.176186841.18883348\/v2","asserted-by":"object"},{"id-type":"doi","id":"10.36227\/techrxiv.176186841.18883348\/v3","asserted-by":"object"}]},"ISSN":["1566-2535"],"issn-type":[{"value":"1566-2535","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Evaluating and regulating agentic AI: A study of benchmarks, metrics, and regulation","name":"articletitle","label":"Article Title"},{"value":"Information Fusion","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.inffus.2026.104444","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"104444"}}