{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T23:17:13Z","timestamp":1783207033578,"version":"3.54.6"},"reference-count":84,"publisher":"Elsevier BV","issue":"8","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100007839","name":"Yunnan University","doi-asserted-by":"publisher","award":["K207003250006"],"award-info":[{"award-number":["K207003250006"]}],"id":[{"id":"10.13039\/501100007839","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007846","name":"Yunnan Provincial Department of Education","doi-asserted-by":"publisher","award":["2023J0019"],"award-info":[{"award-number":["2023J0019"]}],"id":[{"id":"10.13039\/501100007846","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100008871","name":"Yunnan Provincial Science and Technology Department","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100008871","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Information Processing &amp; Management"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.ipm.2026.104835","type":"journal-article","created":{"date-parts":[[2026,5,28]],"date-time":"2026-05-28T19:31:52Z","timestamp":1779996712000},"page":"104835","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"title":["BJZM: A scalable framework for developing and benchmarking LLMs with an open leaderboard in Chinese literature"],"prefix":"10.1016","volume":"63","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7867-8087","authenticated-orcid":false,"given":"Gang","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingqing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Min","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.ipm.2026.104835_b1","doi-asserted-by":"crossref","DOI":"10.3389\/fdata.2025.1455442","article-title":"Impact of imbalanced features on large datasets","volume":"8","author":"Albattah","year":"2025","journal-title":"Frontiers in Big Data"},{"key":"10.1016\/j.ipm.2026.104835_b2","series-title":"Qwen technical report","author":"Bai","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b3","series-title":"BaiJia: A large scale role-playing agent corpus of Chinese historical characters","author":"Bai","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b4","series-title":"Deepseek LLM: Scaling open-source language models with longtermism","author":"Bi","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b5","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"10.1016\/j.ipm.2026.104835_b6","series-title":"InternLM2 technical report","author":"Cai","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b7","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"17709","article-title":"Medbench: A large-scale chinese benchmark for evaluating medical large language models","volume":"vol. 38","author":"Cai","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b8","doi-asserted-by":"crossref","unstructured":"Cao, J., Liu, Y., Shi, Y., Ding, K., & Jin, L. (2024). WenMind: A comprehensive benchmark for evaluating large language models in Chinese classical literature and language arts. 37, 51358\u201351410. http:\/\/dx.doi.org\/10.52202\/079017-1626.","DOI":"10.52202\/079017-1626"},{"key":"10.1016\/j.ipm.2026.104835_b9","series-title":"C3Bench: A comprehensive classical Chinese understanding benchmark for large language models","author":"Cao","year":"2024"},{"issue":"3","key":"10.1016\/j.ipm.2026.104835_b10","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3641289","article-title":"A survey on evaluation of large language models","volume":"15","author":"Chang","year":"2024","journal-title":"ACM Transactions on Intelligent Systems and Technology"},{"key":"10.1016\/j.ipm.2026.104835_b11","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1016\/j.sbspro.2012.05.073","article-title":"Conception and validation of a tool of creation and management of the evaluation of the lessons and formations of students\u2019 distance learning: CG-EVAL-EFDE","volume":"46","author":"Chemsi","year":"2012","journal-title":"Procedia - Social and Behavioral Sciences"},{"key":"10.1016\/j.ipm.2026.104835_b12","series-title":"Chronicle of modern Chinese novels (1922\u20131949)","author":"Chen","year":"2021"},{"key":"10.1016\/j.ipm.2026.104835_b13","doi-asserted-by":"crossref","unstructured":"Chen, Z. (2024). Sentence Segmentation and Sentence Punctuation based on XunziALLM. In Proceedings of the third workshop on language technologies for historical and ancient languages (LT4HALA)@ LREC-COLING-2024 (pp. 246\u2013250).","DOI":"10.63317\/2nkn58nsxh4v"},{"key":"10.1016\/j.ipm.2026.104835_b14","series-title":"Proceedings of the 2025 conference on empirical methods in natural language processing","first-page":"33007","article-title":"Benchmarking llms for translating classical chinese poetry: Evaluating adequacy, fluency, and elegance","author":"Chen","year":"2025"},{"key":"10.1016\/j.ipm.2026.104835_b15","series-title":"DISC-FinLLM: A Chinese financial large language model based on multiple experts fine-tuning","author":"Chen","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b16","series-title":"Bianque: Balancing the questioning and suggestion ability of health LLMs with multi-turn health conversations polished by ChatGPT","author":"Chen","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b17","series-title":"Findings of the association for computational linguistics","first-page":"1170","article-title":"Soulchat: Improving llms\u2019 empathy, listening, and comfort abilities through fine-tuning with multi-turn empathy conversations","author":"Chen","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b18","series-title":"Consensus attention-based neural networks for Chinese reading comprehension","author":"Cui","year":"2016"},{"key":"10.1016\/j.ipm.2026.104835_b19","series-title":"ChatLaw: A multi-agent collaborative legal assistant with knowledge graph enhanced mixture-of-experts large language model","author":"Cui","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b20","series-title":"EduChat: A large-scale language model-based chatbot system for intelligent education","author":"Dan","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b21","doi-asserted-by":"crossref","unstructured":"Derczynski, L. (2016). Complementarity, F-score, and NLP Evaluation. In Proceedings of the tenth international conference on language resources and evaluation (pp. 261\u2013266).","DOI":"10.63317\/2qtxhh68xnxz"},{"key":"10.1016\/j.ipm.2026.104835_b22","series-title":"The llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b23","series-title":"CCLUE: Classical Chinese language understanding evaluation benchmark","author":"Ethan-yt","year":"2021"},{"key":"10.1016\/j.ipm.2026.104835_b24","series-title":"Proceedings of the 2024 conference on empirical methods in natural language processing","first-page":"7933","article-title":"Lawbench: Benchmarking legal knowledge of large language models","author":"Fei","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b25","series-title":"A framework for few-shot language model evaluation","author":"Gao","year":"2021"},{"issue":"6","key":"10.1016\/j.ipm.2026.104835_b26","article-title":"Koala: A dialogue model for academic research","volume":"1","author":"Geng","year":"2023","journal-title":"Blog post"},{"key":"10.1016\/j.ipm.2026.104835_b27","series-title":"ChatGLM: A family of large language models from GLM-130b to GLM-4 all tools","author":"GLM","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b28","series-title":"European conference on information retrieval","first-page":"345","article-title":"A probabilistic interpretation of precision, recall and F-score, with implication for evaluation","author":"Goutte","year":"2005"},{"key":"10.1016\/j.ipm.2026.104835_b29","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"18099","article-title":"Xiezhi: An ever-updating benchmark for holistic domain knowledge evaluation","volume":"vol. 38","author":"Gu","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b30","series-title":"Evaluating large language models: A comprehensive survey","author":"Guo","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b31","series-title":"Proceedings of the 2025 conference of the nations of the americas chapter of the association for computational linguistics: Human language technologies","first-page":"6258","article-title":"Fineval: A chinese financial domain knowledge evaluation benchmark for large language models","author":"Guo","year":"2025"},{"key":"10.1016\/j.ipm.2026.104835_b32","series-title":"No language is an Island: Unifying Chinese and english in financial large language models, instruction data, and benchmarks","author":"Hu","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b33","series-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2021"},{"key":"10.1016\/j.ipm.2026.104835_b34","article-title":"C-eval: A multi-level multi-discipline chinese evaluation suite for foundation models","volume":"36","author":"Huang","year":"2024","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"10.1016\/j.ipm.2026.104835_b35","series-title":"Proceedings of the 63rd annual meeting of the association for computational linguistics","first-page":"33167","article-title":"Opencoder: The open cookbook for top-tier code large language models","author":"Huang","year":"2025"},{"key":"10.1016\/j.ipm.2026.104835_b36","series-title":"Proceedings of the 23rd Chinese national conference on computational linguistics","first-page":"1351","article-title":"AuditWen: An open-source large language model for audit","author":"Jiajia","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b37","series-title":"CCF international conference on natural language processing and Chinese computing","first-page":"387","article-title":"Towards better translations from classical to modern Chinese: A new dataset and a new method","author":"Jiang","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b38","series-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020"},{"key":"10.1016\/j.ipm.2026.104835_b39","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics","first-page":"13484","article-title":"How are we detecting inconsistent method names? An empirical study from code review perspective","author":"Kim","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b40","series-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014"},{"key":"10.1016\/j.ipm.2026.104835_b41","series-title":"Proceedings of the 7th joint SIGHUM workshop on computational linguistics for cultural heritage, social sciences, humanities and literature","first-page":"1","article-title":"Standard and non-standard adverbial markers: A diachronic analysis in modern Chinese literature","author":"Lee","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b42","series-title":"Are ChatGPT and GPT-4 general-purpose solvers for financial text analytics? A study on several typical tasks","author":"Li","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b43","doi-asserted-by":"crossref","unstructured":"Li, B., Chang, B., Xu, Z., Feng, M., Xu, C., Qu, W., Shen, S., & Wang, D. (2024). Overview of EvaHan2024: The First International Evaluation on Ancient Chinese Sentence Segmentation and Punctuation. In Proceedings of the third workshop on language technologies for historical and ancient languages (pp. 229\u2013236).","DOI":"10.63317\/35gtq836br2j"},{"issue":"11","key":"10.1016\/j.ipm.2026.104835_b44","doi-asserted-by":"crossref","first-page":"4906","DOI":"10.3390\/app14114906","article-title":"Text classification model based on graph attention networks and adversarial training","volume":"14","author":"Li","year":"2024","journal-title":"Applied Sciences"},{"key":"10.1016\/j.ipm.2026.104835_b45","series-title":"Findings of the association for computational linguistics: ACL 2024","first-page":"11260","article-title":"Cmmlu: Measuring massive multitask language understanding in chinese","author":"Li","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b46","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhao, J., Zheng, D., Hu, Z.-Y., Chen, Z., Su, X., Huang, Y., Huang, S., Lin, D., Lyu, M., et al. (2023). Cleva: Chinese language models evaluation platform. In Proceedings of the 2023 conference on empirical methods in natural language processing: System demonstrations (pp. 186\u2013217).","DOI":"10.18653\/v1\/2023.emnlp-demo.17"},{"key":"10.1016\/j.ipm.2026.104835_b47","series-title":"Self-supervised learning is more robust to dataset imbalance","author":"Liu","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b48","series-title":"Findings of the association for computational linguistics ACL 2024","first-page":"8801","article-title":"Error analysis prompting enables human-like translation evaluation in large language models","author":"Lu","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b49","series-title":"Capabilities and evaluation biases of large language models in classical Chinese poetry generation: A case study on tang poetry","author":"Ma","year":"2025"},{"key":"10.1016\/j.ipm.2026.104835_b50","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics","first-page":"15991","article-title":"Crosslingual generalization through multitask finetuning","author":"Muennighoff","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b51","series-title":"Proceedings of the 10th international conference on natural language generation","first-page":"11","article-title":"A survey on intelligent poetry generation: Languages, features, techniques, reutilisation and evaluation","author":"Oliveira","year":"2017"},{"key":"10.1016\/j.ipm.2026.104835_b52","series-title":"Findings of the association for computational linguistics: EMNLP 2023","first-page":"5622","article-title":"Towards making the most of ChatGPT for machine translation","author":"Peng","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b53","series-title":"Plutus: Benchmarking large language models in low-resource greek finance","author":"Peng","year":"2025"},{"issue":"8","key":"10.1016\/j.ipm.2026.104835_b54","doi-asserted-by":"crossref","DOI":"10.1073\/pnas.2422455122","article-title":"Do LLMs write like humans? Variation in grammatical and rhetorical styles","volume":"122","author":"Reinhart","year":"2025","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"10.1016\/j.ipm.2026.104835_b55","series-title":"Proceedings of the second workshop on ancient language processing","first-page":"1","article-title":"Automatic text segmentation of ancient and historic hebrew","author":"Rosensweig","year":"2025"},{"key":"10.1016\/j.ipm.2026.104835_b56","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"8968","article-title":"Ernie 2.0: A continual pre-training framework for language understanding","volume":"vol. 34","author":"Sun","year":"2020"},{"key":"10.1016\/j.ipm.2026.104835_b57","series-title":"Gemma: Open models based on gemini research and technology","author":"Team","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b58","series-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b59","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics","first-page":"13484","article-title":"Self-instruct: Aligning language models with self-generated instructions","author":"Wang","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b60","series-title":"Huatuo: Tuning llama model with Chinese medical knowledge","author":"Wang","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b61","series-title":"Emergent abilities of large language models","author":"Wei","year":"2022"},{"key":"10.1016\/j.ipm.2026.104835_b62","series-title":"Findings of the association for computational linguistics: EMNLP 2024","first-page":"1600","article-title":"Ac-eval: Evaluating ancient chinese language understanding in large language models","author":"Wei","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b63","first-page":"1","article-title":"Densing law of llms","author":"Xiao","year":"2025","journal-title":"Nature Machine Intelligence"},{"key":"10.1016\/j.ipm.2026.104835_b64","doi-asserted-by":"crossref","first-page":"95716","DOI":"10.52202\/079017-3033","article-title":"Finben: A holistic financial benchmark for large language models","volume":"37","author":"Xie","year":"2024","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"10.1016\/j.ipm.2026.104835_b65","series-title":"The wall street neophyte: A zero-shot analysis of ChatGPT over multimodal stock movement prediction challenges","author":"Xie","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b66","doi-asserted-by":"crossref","unstructured":"Xie, Q., Han, W., Zhang, X., Lai, Y., Peng, M., Lopez-Lira, A., & Huang, J. (2023). Pixiu: A comprehensive benchmark, instruction dataset and large language model for finance. 36.","DOI":"10.52202\/075280-1454"},{"key":"10.1016\/j.ipm.2026.104835_b67","series-title":"Proceedings of the 28th international conference on computational linguistics","first-page":"4762","article-title":"CLUE: A Chinese language understanding evaluation benchmark","author":"Xu","year":"2020"},{"key":"10.1016\/j.ipm.2026.104835_b68","series-title":"Superclue: A comprehensive Chinese large language model benchmark","author":"Xu","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b69","series-title":"Native Chinese reader: A dataset towards native-level Chinese machine reading comprehension","author":"Xu","year":"2021"},{"key":"10.1016\/j.ipm.2026.104835_b70","series-title":"A discourse-level named entity recognition and relation extraction dataset for Chinese literature text","author":"Xu","year":"2017"},{"key":"10.1016\/j.ipm.2026.104835_b71","series-title":"HSKBenchmark: Modeling and benchmarking Chinese second language acquisition in large language models through curriculum tuning","author":"Yang","year":"2025"},{"key":"10.1016\/j.ipm.2026.104835_b72","series-title":"Baichuan 2: Open large-scale language models","author":"Yang","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b73","series-title":"Qwen2 technical report","author":"Yang","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b74","series-title":"Yi: Open foundation models by 01. AI","author":"Young","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b75","series-title":"Disc-LawLLM: Fine-tuning large language models for intelligent legal services","author":"Yue","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b76","series-title":"Measuring massive multitask Chinese understanding","author":"Zeng","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b77","series-title":"Proceedings of the ancient language processing workshop","first-page":"80","article-title":"Can large langauge model comprehend ancient Chinese? A preliminary test on ACLUE","author":"Zhang","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b78","series-title":"Evaluating the performance of large language models on gaokao benchmark","author":"Zhang","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b79","series-title":"Proceedings of the 30th ACM SIGKDD conference on knowledge discovery and data mining","first-page":"6236","article-title":"D\u00f3lares or dollars? unraveling the bilingual prowess of financial llms between spanish and english","author":"Zhang","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b80","series-title":"F\u00f9x\u00ec: A benchmark for evaluating language models on ancient Chinese text understanding and generation","author":"Zhao","year":"2025"},{"key":"10.1016\/j.ipm.2026.104835_b81","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics","first-page":"400","article-title":"LlamaFactory: Unified efficient fine-tuning of 100+ language models","author":"Zheng","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b82","series-title":"Findings of the association for computational linguistics: NAACL 2024","first-page":"2299","article-title":"Agieval: A human-centric benchmark for evaluating foundation models","author":"Zhong","year":"2024"},{"key":"10.1016\/j.ipm.2026.104835_b83","series-title":"Can ChatGPT understand too? A comparative study on ChatGPT and fine-tuned BERT","author":"Zhong","year":"2023"},{"key":"10.1016\/j.ipm.2026.104835_b84","series-title":"LawGPT: A Chinese legal knowledge-enhanced large language model","author":"Zhou","year":"2024"}],"container-title":["Information Processing &amp; Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0306457326002268?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0306457326002268?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T22:27:22Z","timestamp":1783204042000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0306457326002268"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":84,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2026,12]]}},"alternative-id":["S0306457326002268"],"URL":"https:\/\/doi.org\/10.1016\/j.ipm.2026.104835","relation":{},"ISSN":["0306-4573"],"issn-type":[{"value":"0306-4573","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"BJZM: A scalable framework for developing and benchmarking LLMs with an open leaderboard in Chinese literature","name":"articletitle","label":"Article Title"},{"value":"Information Processing & Management","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.ipm.2026.104835","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104835"}}