{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T06:15:50Z","timestamp":1774419350230,"version":"3.50.1"},"reference-count":35,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,4,6]]},"DOI":"10.1109\/icassp49660.2025.10890123","type":"proceedings-article","created":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T13:52:43Z","timestamp":1741787563000},"page":"1-5","source":"Crossref","is-referenced-by-count":1,"title":["IDE: A Multi-Agent-Driven Iterative Framework for Dynamic Evaluation of LLMs"],"prefix":"10.1109","author":[{"given":"Xin","family":"Tong","sequence":"first","affiliation":[{"name":"People&#x2019;s Public Security University of China,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Jin","sequence":"additional","affiliation":[{"name":"The Third Research Institute of the Ministry of Public Security of China,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingya","family":"Wang","sequence":"additional","affiliation":[{"name":"People&#x2019;s Public Security University of China,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenpeng","family":"Xing","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tian","family":"Xia","sequence":"additional","affiliation":[{"name":"Renmin University of China,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meng","family":"Han","sequence":"additional","affiliation":[{"name":"Zhejiang University,Hangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"3266","article-title":"Superglue: a stickier benchmark for general-purpose language understanding systems","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","author":"Wang"},{"key":"ref2","first-page":"62991","article-title":"C-eval: a multi-level multi-discipline chinese evaluation suite for foundation models","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Huang"},{"key":"ref3","article-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i23.34678"},{"key":"ref5","article-title":"Chatbot arena: An open platform for evaluating llms by human preference","author":"Chiang","year":"2024"},{"key":"ref6","article-title":"Benchmark self-evolving: A multi-agent framework for dynamic llm evaluation","author":"Wang","year":"2024"},{"key":"ref7","article-title":"Measuring massive multitask language understanding","author":"Hendrycks","year":"2020"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.452"},{"key":"ref9","article-title":"Laiw: A chinese legal large language models benchmark (a technical report)","author":"Dai","year":"2023"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-023-06291-2"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.bionlp-1.30"},{"key":"ref12","article-title":"Evaluating the performance of large language models on gaokao benchmark","author":"Zhang","year":"2023"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.521"},{"key":"ref14","article-title":"Financebench: A new benchmark for financial question answering","author":"Islam","year":"2023"},{"key":"ref15","article-title":"The finben: An holistic financial benchmark for large language models","author":"Xie","year":"2024"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.482"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29808"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.716"},{"key":"ref19","article-title":"Dyval 2: Dynamic evaluation of large language models by meta probing agents","author":"Zhu","year":"2024"},{"key":"ref20","article-title":"Darg: Dynamic evaluation of large language models via adaptive reasoning graph","author":"Zhang","year":"2024"},{"key":"ref21","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang","year":"2022","journal-title":"Advances in neural information processing systems"},{"key":"ref22","article-title":"Towards understanding sycophancy in language models","author":"Sharma","year":"2023"},{"key":"ref23","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.26"},{"key":"ref25","article-title":"Glm-130b: An open bilingual pre-trained model","author":"Zeng","year":"2022"},{"key":"ref26","article-title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools","author":"GLM","year":"2024"},{"key":"ref27","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref28","article-title":"Mistral 7b","author":"Jiang","year":"2023"},{"key":"ref29","article-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"ref30","volume-title":"Hello gpt-4o","year":"2024"},{"key":"ref31","article-title":"Baichuan 2: Open large-scale language models","author":"Yang","year":"2023"},{"key":"ref32","first-page":"2924","article-title":"Boolq: Exploring the surprising difficulty of natural yes\/no questions","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)","author":"Clark"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1458"},{"key":"ref34","article-title":"Fireact: Toward language agent fine-tuning","author":"Chen","year":"2023"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3748302"}],"event":{"name":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Hyderabad, India","start":{"date-parts":[[2025,4,6]]},"end":{"date-parts":[[2025,4,11]]}},"container-title":["ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10887540\/10887541\/10890123.pdf?arnumber=10890123","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T05:22:25Z","timestamp":1774416145000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10890123\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,6]]},"references-count":35,"URL":"https:\/\/doi.org\/10.1109\/icassp49660.2025.10890123","relation":{},"subject":[],"published":{"date-parts":[[2025,4,6]]}}}