{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T14:40:06Z","timestamp":1780756806659,"version":"3.54.1"},"reference-count":10,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T00:00:00Z","timestamp":1760659200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T00:00:00Z","timestamp":1760659200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front. Comput. Sci."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s11704-025-50442-9","type":"journal-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T04:11:32Z","timestamp":1760674292000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Trustworthy evaluation of large language models"],"prefix":"10.1007","volume":"20","author":[{"given":"Xin-Yi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Han-Jia","family":"Ye","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"De-Chuan","family":"Zhan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,17]]},"reference":[{"key":"50442_CR1","unstructured":"Liu Y, Yao Y, Ton J F, Zhang X, Guo R, Cheng H, Klochkov Y, Taufiq M F, Li H. Trustworthy LLMs: a survey and guideline for evaluating large language models\u2019 alignment. 2023, arXiv preprint arXiv: 2308.05374"},{"key":"50442_CR2","first-page":"20166","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Y Huang","year":"2024","unstructured":"Huang Y, Sun L, Wang H, Wu S, Zhang Q, Li Y, Gao C, Huang Y, Lyu W, Zhang Y, Li X, Sun H, Liu Z, Liu Y, Wang Y, Zhang Z, Vidgen B, Kailkhura B, Xiong C, Xiao C, Li C, Xing E P, Huang F, Liu H, Ji H, Wang H, Zhang H, Yao H, Kellis M, Zitnik M, Jiang M, Bansal M, Zou J, Pei J, Liu J, Gao J, Han J, Zhao J, Tang J, Wang J, Vanschoren J, Mitchell J, Shu K, Xu K, Chang K W, He L, Huang L, Backes M, Gong N Z, Yu P S, Chen P Y, Gu Q, Xu R, Ying R, Ji S, Jana S, Chen T, Liu T, Zhou T, Wang W Y, Li X, Zhang X, Wang X, Xie X, Chen X, Wang X, Liu Y, Ye Y, Cao Y, Chen Y, Zhao Y. TrustLLM: trustworthiness in large language models. In: Proceedings of the 41st International Conference on Machine Learning. 2024, 20166\u201320270"},{"issue":"3","key":"50442_CR3","doi-asserted-by":"publisher","first-page":"709","DOI":"10.2307\/258792","volume":"20","author":"R C Mayer","year":"1995","unstructured":"Mayer R C, Davis J H, Schoorman F D. An integrative model of organizational trust. The Academy of Management Review, 1995, 20(3): 709\u2013734","journal-title":"The Academy of Management Review"},{"key":"50442_CR4","doi-asserted-by":"publisher","first-page":"272","DOI":"10.1145\/3351095.3372834","volume-title":"Proceedings of 2020 Conference on Fairness, Accountability, and Transparency","author":"E Toreini","year":"2020","unstructured":"Toreini E, Aitken M, Coopamootoo K, Elliott K, Zelaya C G, van Moorsel A. The relationship between trust in AI and trustworthy machine learning technologies. In: Proceedings of 2020 Conference on Fairness, Accountability, and Transparency. 2020, 272\u2013283"},{"issue":"1","key":"50442_CR5","first-page":"4","volume":"14","author":"H Liu","year":"2022","unstructured":"Liu H, Wang Y, Fan W, Liu X, Li Y, Jain S, Liu Y, Jain A, Tang J. Trustworthy AI: a computational perspective. ACM Transactions on Intelligent Systems and Technology, 2022, 14(1): 4","journal-title":"ACM Transactions on Intelligent Systems and Technology"},{"issue":"9","key":"50442_CR6","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1145\/3555803","volume":"55","author":"B Li","year":"2023","unstructured":"Li B, Qi P, Liu B, Di S, Liu J, Pei J, Yi J, Zhou B. Trustworthy AI: from principles to practices. ACM Computing Surveys, 2023, 55(9): 177","journal-title":"ACM Computing Surveys"},{"key":"50442_CR7","first-page":"1800","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"J Wei","year":"2022","unstructured":"Wei J, Wang X, Schuurmans D, Bosma M, Ichter B, Xia F, Chi E H, Le Q V, Zhou D. Chain-of-thought prompting elicits reasoning in large language models. In: Proceedings of the 36th International Conference on Neural Information Processing Systems. 2022, 1800"},{"key":"50442_CR8","first-page":"15012","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"D Paul","year":"2024","unstructured":"Paul D, West R, Bosselut A, Faltings B. Making reasoning matter: measuring and improving faithfulness of chain-of-thought reasoning. In: Proceedings of Findings of the Association for Computational Linguistics. 2024, 15012\u201315032"},{"key":"50442_CR9","first-page":"15696","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"N Kandpal","year":"2023","unstructured":"Kandpal N, Deng H, Roberts A, Wallace E, Raffel C. Large language models struggle to learn long-tail knowledge. In: Proceedings of the 40th International Conference on Machine Learning. 2023, 15696\u201315707"},{"issue":"3","key":"50442_CR10","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1145\/3617680","volume":"56","author":"H Zhang","year":"2024","unstructured":"Zhang H, Song H, Li S, Zhou M, Song D. A survey of controllable text generation using transformer-based pre-trained language models. ACM Computing Surveys, 2024, 56(3): 64","journal-title":"ACM Computing Surveys"}],"container-title":["Frontiers of Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-025-50442-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11704-025-50442-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-025-50442-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T04:11:34Z","timestamp":1760674294000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11704-025-50442-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,17]]},"references-count":10,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["50442"],"URL":"https:\/\/doi.org\/10.1007\/s11704-025-50442-9","relation":{},"ISSN":["2095-2228","2095-2236"],"issn-type":[{"value":"2095-2228","type":"print"},{"value":"2095-2236","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,17]]},"assertion":[{"value":"13 April 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 June 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 October 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare that they have no competing interests or financial conflicts to disclose.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"2002324"}}