{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T18:18:19Z","timestamp":1783016299091,"version":"3.54.6"},"reference-count":65,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100022963","name":"Key Research and Development Program of Zhejiang Province","doi-asserted-by":"publisher","award":["2025C01104"],"award-info":[{"award-number":["2025C01104"]}],"id":[{"id":"10.13039\/100022963","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neucom.2026.134180","type":"journal-article","created":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T15:19:49Z","timestamp":1780586389000},"page":"134180","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Inclusion arena: A theoretically grounded framework for evaluating large foundation models via application-embedded pairwise comparisons"],"prefix":"10.1016","volume":"697","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-7575-2570","authenticated-orcid":false,"given":"Hongliang","family":"He","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kangyu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruiqi","family":"Liang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuai","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7566-7948","authenticated-orcid":false,"given":"Renjun","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenzhong","family":"Lan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianguo","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.134180_bib0005","author":"Achiam"},{"key":"10.1016\/j.neucom.2026.134180_bib0010","author":"Team"},{"key":"10.1016\/j.neucom.2026.134180_bib0015","unstructured":"Anthropic, Introducing claude 4, https:\/\/www.anthropic.com\/news\/claude-4., Anthropic News (Announcements)."},{"key":"10.1016\/j.neucom.2026.134180_bib0020","series-title":"International Conference on Learning Representations","article-title":"Measuring massive multitask language understanding","author":"Hendrycks","year":"2020"},{"key":"10.1016\/j.neucom.2026.134180_bib0025","author":"Chen"},{"key":"10.1016\/j.neucom.2026.134180_bib0030","series-title":"Forty-First International Conference on Machine Learning","article-title":"Chatbot arena: an open platform for evaluating LLMs by human preference","author":"Chiang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134180_bib0035","author":"Li"},{"key":"10.1016\/j.neucom.2026.134180_bib0040","author":"Achiam"},{"issue":"8","key":"10.1016\/j.neucom.2026.134180_bib0045","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"key":"10.1016\/j.neucom.2026.134180_bib0050","doi-asserted-by":"crossref","first-page":"46595","DOI":"10.52202\/075280-2020","article-title":"Judging LLM-as-a-judge with MT-bench and chatbot arena","volume":"36","author":"Zheng","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134180_bib0055","author":"Zheng"},{"issue":"3\/4","key":"10.1016\/j.neucom.2026.134180_bib0060","doi-asserted-by":"crossref","first-page":"324","DOI":"10.2307\/2334029","article-title":"Rank analysis of incomplete block designs: I. the method of paired comparisons","volume":"39","author":"Bradley","year":"1952","journal-title":"Biometrika"},{"key":"10.1016\/j.neucom.2026.134180_bib0065","first-page":"273","article-title":"Bayesian experimental design: a review","author":"Chaloner","year":"1995","journal-title":"Stat. Sci."},{"key":"10.1016\/j.neucom.2026.134180_bib0070","series-title":"Active learning literature survey","author":"Settles","year":"2009"},{"key":"10.1016\/j.neucom.2026.134180_bib0075","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.131606","article-title":"Evaluation of memristor performance in neural networks using an AHaH framework","volume":"657","author":"Xu","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134180_bib0080","doi-asserted-by":"crossref","first-page":"109","DOI":"10.1016\/j.neucom.2017.12.067","article-title":"Anomaly detection in smart card logs and distant evaluation with twitter: a robust framework","volume":"298","author":"Tonnelier","year":"2018","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134180_bib0085","doi-asserted-by":"crossref","first-page":"440","DOI":"10.1016\/j.neucom.2021.07.095","article-title":"Anomaly detection in predictive maintenance: a new evaluation framework for temporal unsupervised anomaly detection algorithms","volume":"462","author":"Carrasco","year":"2021","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134180_bib0090","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2023.127227","article-title":"FedAVE: adaptive data value evaluation framework for collaborative fairness in federated learning","volume":"574","author":"Wang","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134180_bib0095","author":"Chen"},{"key":"10.1016\/j.neucom.2026.134180_bib0100","author":"Cobbe"},{"key":"10.1016\/j.neucom.2026.134180_bib0105","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence, 38","first-page":"17709","article-title":"Medbench: a large-scale Chinese benchmark for evaluating medical large language models","author":"Cai","year":"2024"},{"key":"10.1016\/j.neucom.2026.134180_bib0110","doi-asserted-by":"crossref","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","article-title":"Visual genome: connecting language and vision using crowdsourced dense image annotations","volume":"123","author":"Krishna","year":"2017","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.134180_bib0115","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"2200","article-title":"Docvqa: a dataset for VQA on document images","author":"Mathew","year":"2021"},{"key":"10.1016\/j.neucom.2026.134180_bib0120","doi-asserted-by":"crossref","first-page":"6155","DOI":"10.1109\/ACCESS.2024.3514079","article-title":"MathVision: an accessible intelligent agent for visually impaired people to understand mathematical equations","volume":"13","author":"Awais","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.neucom.2026.134180_bib0125","author":"Lu"},{"key":"10.1016\/j.neucom.2026.134180_bib0130","doi-asserted-by":"crossref","first-page":"28091","DOI":"10.52202\/075280-1220","article-title":"Mind2web: towards a generalist agent for the web","volume":"36","author":"Deng","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134180_bib0135","author":"Pan"},{"key":"10.1016\/j.neucom.2026.134180_bib0140","author":"He"},{"key":"10.1016\/j.neucom.2026.134180_bib0145","author":"Hendrycks"},{"key":"10.1016\/j.neucom.2026.134180_bib0150","author":"Myrzakhan"},{"key":"10.1016\/j.neucom.2026.134180_bib0165","author":"Liang"},{"key":"10.1016\/j.neucom.2026.134180_bib0170","author":"Zhou"},{"key":"10.1016\/j.neucom.2026.134180_bib0175","series-title":"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","first-page":"4110","article-title":"Dynabench: rethinking benchmarking in NLP","author":"Kiela","year":"2021"},{"key":"10.1016\/j.neucom.2026.134180_bib0180","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.130135","article-title":"Can large language models independently complete tasks? A dynamic evaluation framework for multi-turn task planning and completion","volume":"638","author":"Gao","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134180_bib0185","doi-asserted-by":"crossref","first-page":"27","DOI":"10.1016\/j.neucom.2022.10.036","article-title":"Dialogue-adaptive language model pre-training from quality estimation","volume":"516","author":"Li","year":"2023","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134180_bib0190","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2026.132787","article-title":"MODE+: a benchmark and a probe into multimodal open-domain dialogue evaluation","volume":"675","author":"Yin","year":"2026","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134180_bib0200","author":"Analysis"},{"key":"10.1016\/j.neucom.2026.134180_bib0215","author":"Hu"},{"key":"10.1016\/j.neucom.2026.134180_bib0225","author":"Tan"},{"key":"10.1016\/j.neucom.2026.134180_bib0240","series-title":"First Conference on Language Modeling","article-title":"AutoGen: enabling next-gen LLM applications via multi-agent conversations","author":"Wu","year":"2024"},{"issue":"4","key":"10.1016\/j.neucom.2026.134180_bib0245","doi-asserted-by":"crossref","DOI":"10.1037\/h0070288","article-title":"A law of comparative judgment","volume":"34","author":"Thurstone","year":"1927","journal-title":"Psychol. Rev."},{"issue":"1","key":"10.1016\/j.neucom.2026.134180_bib0250","doi-asserted-by":"crossref","first-page":"384","DOI":"10.1214\/aos\/1079120141","article-title":"MM algorithms for generalized Bradley-terry models","volume":"32","author":"Hunter","year":"2004","journal-title":"The annals of statistics"},{"key":"10.1016\/j.neucom.2026.134180_bib0255","series-title":"The Rating of Chessplayers, Past and Present","author":"Elo","year":"1978"},{"key":"10.1016\/j.neucom.2026.134180_bib0260","series-title":"TrueSkill: A Bayesian Skill Rating System","author":"Herbrich","year":"2005"},{"key":"10.1016\/j.neucom.2026.134180_bib0265","series-title":"International Conference on Artificial Intelligence and Statistics","first-page":"2905","article-title":"On the limitations of the Elo, real-world games are transitive, not additive","author":"Bertrand","year":"2023"},{"key":"10.1016\/j.neucom.2026.134180_bib0270","series-title":"International Conference on Machine Learning","first-page":"2354","article-title":"Choicerank: identifying preferences from node traffic in networks","author":"Maystre","year":"2017"},{"key":"10.1016\/j.neucom.2026.134180_bib0275","author":"Min"},{"key":"10.1016\/j.neucom.2026.134180_bib0280","author":"Singh"},{"key":"10.1016\/j.neucom.2026.134180_bib0285","unstructured":"Ant Group and Ant Design Community, Tbox \u2014 ant design X (playground), https:\/\/x.ant.design\/docs\/playground\/agent-tbox\/"},{"key":"10.1016\/j.neucom.2026.134180_bib0290","series-title":"The Thirty-Eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track","article-title":"MMLU-pro: a more robust and challenging multi-task language understanding benchmark","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134180_bib0300","author":"Liu"},{"key":"10.1016\/j.neucom.2026.134180_bib0305","author":"Team"},{"key":"10.1016\/j.neucom.2026.134180_bib0315","author":"Kamath"},{"key":"10.1016\/j.neucom.2026.134180_bib0330","author":"Bertrand"},{"key":"10.1016\/j.neucom.2026.134180_bib0335","author":"Ouyang"},{"key":"10.1016\/j.neucom.2026.134180_bib0340","author":"Rafailov"},{"key":"10.1016\/j.neucom.2026.134180_bib0345","doi-asserted-by":"crossref","DOI":"10.1109\/TBDATA.2024.3524104","article-title":"Aligning crowd-sourced human feedback for reinforcement learning on code generation by large language models","author":"Wong","year":"2024","journal-title":"IEEE Trans. Big Data"},{"key":"10.1016\/j.neucom.2026.134180_bib0350","series-title":"The Twelfth International Conference on Learning Representations","article-title":"SWE-bench: can language models resolve real-world GitHub issues?","author":"Jimenez","year":"2023"},{"key":"10.1016\/j.neucom.2026.134180_bib0355","author":"White"},{"key":"10.1016\/j.neucom.2026.134180_bib0360","author":"Merrill"},{"issue":"4","key":"10.1016\/j.neucom.2026.134180_bib0365","doi-asserted-by":"crossref","first-page":"34","DOI":"10.1109\/MS.2025.3549628","article-title":"From code generation to software testing: AI copilot with context-based RAG","volume":"42","author":"Wang","year":"2025","journal-title":"IEEE Softw."},{"key":"10.1016\/j.neucom.2026.134180_bib0370","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence, 39","first-page":"7907","article-title":"Divide, conquer and combine: a training-free framework for high-resolution image perception in multimodal large language models","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134180_bib0375","series-title":"Findings of the Association for Computational Linguistics: ACL 2024","first-page":"8801","article-title":"Error analysis prompting enables human-like translation evaluation in large language models","author":"Lu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134180_bib0380","author":"Grattafiori"},{"key":"10.1016\/j.neucom.2026.134180_bib0390","author":"Guo"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S092523122601578X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S092523122601578X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T17:22:48Z","timestamp":1783012968000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S092523122601578X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":65,"alternative-id":["S092523122601578X"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134180","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Inclusion arena: A theoretically grounded framework for evaluating large foundation models via application-embedded pairwise comparisons","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134180","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"134180"}}