{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:44:29Z","timestamp":1782798269943,"version":"3.54.5"},"reference-count":52,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"name":"State Grid Science and Technology Project \u201cResearch on Key Technologies of Power Knowledge Question Answering Based on Large-Scale Language Model\u201d","award":["5211DS24000H"],"award-info":[{"award-number":["5211DS24000H"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/access.2026.3692810","type":"journal-article","created":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T19:58:44Z","timestamp":1778788724000},"page":"96147-96161","source":"Crossref","is-referenced-by-count":0,"title":["PowerIntelBench: Benchmarking Evidence-Grounded and Explainable Reasoning in Power Knowledge Management"],"prefix":"10.1109","volume":"14","author":[{"given":"Tao","family":"Yang","sequence":"first","affiliation":[{"name":"State Grid Zhejiang Electric Power Company Ltd. Research Institute, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiajia","family":"Han","sequence":"additional","affiliation":[{"name":"State Grid Zhejiang Electric Power Company Ltd. Research Institute, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cai","family":"Zhang","sequence":"additional","affiliation":[{"name":"State Grid Zhejiang Electric Power Company Ltd. Research Institute, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2663-0848","authenticated-orcid":false,"given":"Junjie","family":"Huang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Ammus: A survey of transformer-based pretrained models in natural language processing","author":"Kalyan","year":"2021"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3605943"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3458754"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.52202\/079017-4037"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1317"},{"key":"ref6","first-page":"9","article-title":"Enhancing llm complex reasoning capability through hyperbolic geometry","volume-title":"Proc. ICML Workshop LLMs Cognition","author":"Yang"},{"key":"ref7","article-title":"Forest-of-thought: Scaling test-time compute for enhancing llm reasoning","author":"Bi","year":"2024","journal-title":"arXiv:2412.09078"},{"key":"ref8","article-title":"Multi-agent collaboration: Harnessing the power of intelligent llm agents","author":"Talebirad","year":"2023","journal-title":"arXiv:2306.03314"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1052"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2025.112310"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.52152\/4073"},{"key":"ref12","article-title":"Elecbench: A power dispatch evaluation benchmark for large language models","author":"Zhou","year":"2024","journal-title":"arXiv:2407.05365"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2024.09.178"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671470"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.647"},{"key":"ref16","article-title":"Data generation using large language models for text classification: An empirical case study","author":"Li","year":"2024","journal-title":"arXiv:2407.12813"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3641289"},{"key":"ref18","article-title":"A survey of useful llm evaluation","author":"Peng","year":"2024","journal-title":"arXiv:2406.00936"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00499"},{"key":"ref21","first-page":"95266","article-title":"MMLU-pro: A more robust and challenging multi-task language understanding benchmark","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"37","author":"Arulraj"},{"issue":"5","key":"ref22","first-page":"1","article-title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models","volume":"2023","author":"Srivastava","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref23","article-title":"Think you have solved question answering? Try arc, the AI2 reasoning challenge","author":"Clark","year":"2018","journal-title":"arXiv:1803.05457"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.183"},{"key":"ref26","article-title":"TinyBenchmarks: Evaluating llms with fewer examples","author":"Polo","year":"2024","journal-title":"arXiv:2402.14992"},{"key":"ref27","article-title":"BEIR: A heterogeneous benchmark for zero-shot evaluation of information retrieval models","author":"Thakur","year":"2021","journal-title":"arXiv:2104.08663"},{"key":"ref28","article-title":"LegalBench-RAG: A benchmark for retrieval-augmented generation in the legal domain","author":"Pipitone","year":"2024","journal-title":"arXiv:2408.10343"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.452"},{"key":"ref30","article-title":"Large language model benchmarks in medical tasks","author":"Yan","year":"2024","journal-title":"arXiv:2410.21348"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.ijcnlp-long.72"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCMA59762.2023.10374622"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2023.111165"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2020"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.138"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.824"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.407"},{"key":"ref38","article-title":"Mineru2.5: A decoupled vision-language model for efficient high-resolution document parsing","author":"Niu","year":"2025","journal-title":"arXiv:2509.22186"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"ref40","first-page":"74","article-title":"Rouge: A package for automatic evaluation of summaries","volume-title":"Proc. Text Summarization Branches Out","author":"Lin"},{"key":"ref41","article-title":"BertScore: Evaluating text generation with BERT","author":"Zhang","year":"2019","journal-title":"arXiv:1904.09675"},{"key":"ref42","article-title":"Holistic evaluation of language models","author":"Liang","year":"2022","journal-title":"arXiv:2211.09110"},{"key":"ref43","article-title":"SCOUT: Teaching pre-trained language models to enhance reasoning via flow chain-of-thought","author":"Li","year":"2025","journal-title":"arXiv:2505.24181"},{"key":"ref44","article-title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities","author":"Comanici","year":"2025","journal-title":"arXiv:2507.06261"},{"key":"ref45","volume-title":"Introducing GPT-4.1 in the API","year":"2024"},{"key":"ref46","article-title":"Qwen3 technical report","volume-title":"arXiv:2505.09388","author":"Yang","year":"2025"},{"key":"ref47","article-title":"ChatGLM: A family of large language models from GLM-130b to GLM-4 all tools","author":"Glm","year":"2024","journal-title":"arXiv:2406.12793"},{"key":"ref48","article-title":"DeepSeek LLM: Scaling open-source language models with longtermism","author":"Bi","year":"2024","journal-title":"arXiv:2401.02954"},{"key":"ref49","article-title":"DeepSeek-V2: A strong, economical, and efficient mixture-of-experts language model","author":"Liu","year":"2024","journal-title":"arXiv:2405.04434"},{"key":"ref50","article-title":"InternVL3.5: Advancing open-source multimodal models in versatility, reasoning, and efficiency","author":"Wang","year":"2025","journal-title":"arXiv:2508.18265"},{"key":"ref51","volume-title":"Qwen3-VL: Multimodal Large Language Model Repository","year":"2025"},{"key":"ref52","article-title":"DeepSeek-VL: Towards real-world vision-language understanding","author":"Lu","year":"2024","journal-title":"arXiv:2403.05525"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/11323511\/11517387.pdf?arnumber=11517387","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T05:10:41Z","timestamp":1782796241000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11517387\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":52,"URL":"https:\/\/doi.org\/10.1109\/access.2026.3692810","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}