{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,28]],"date-time":"2025-05-28T04:12:08Z","timestamp":1748405528562,"version":"3.41.0"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T00:00:00Z","timestamp":1748304000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T00:00:00Z","timestamp":1748304000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["2024JKF13"],"award-info":[{"award-number":["2024JKF13"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"name":"the Natural Science Foundation of Xinjiang Uygur Autonomous Region","award":["2024D01A55"],"award-info":[{"award-number":["2024D01A55"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Discov Artif Intell"],"DOI":"10.1007\/s44163-025-00294-w","type":"journal-article","created":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T10:55:39Z","timestamp":1748343339000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["T2F: a domain-agnostic multi-agent framework for unstructured text to factuality evaluation items generation"],"prefix":"10.1007","volume":"5","author":[{"given":"Xin","family":"Tong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingya","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yasen","family":"Aizezi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hanming","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,27]]},"reference":[{"key":"294_CR1","unstructured":"Dan Y, Lei Z, Gu Y, Li Y, Yin J, Lin J, Ye L, Tie Z, Zhou Y, Wang Y, et al. Educhat: A large-scale language model-based chatbot system for intelligent education, 2023. arxiv:2308.02773."},{"key":"294_CR2","unstructured":"Zhang K, Yu J, Yan Z, Liu Y, Adhikarla E, Fu S, Chen X, Chen C, Zhou Y, Li X, et al. BiomedGPT: a unified and generalist biomedical generative pre-trained transformer for vision, language, and multimodal tasks, 2023. arxiv:2305.17100."},{"key":"294_CR3","doi-asserted-by":"crossref","unstructured":"Yang S, Zhao H, Zhu S, Zhou G, Xu H, Jia Y, Zan H. Zhongjing: Enhancing the chinese medical capabilities of large language model through expert feedback and real-world multi-turn dialogue. In: Proceedings of the AAAI conference on artificial intelligence, 2024;38:19368\u201319376.","DOI":"10.1609\/aaai.v38i17.29907"},{"key":"294_CR4","unstructured":"Cui J, Li Z, Yan Y, Chen B, Yuan L. Chatlaw: Open-source legal large language model with integrated external knowledge bases, 2023. arxiv.org:2306.16092."},{"key":"294_CR5","unstructured":"Wu S, Irsoy O, Lu S, Dabravolski V, Dredze M, Gehrmann S, Kambadur P, Rosenberg D, Mann G. Bloomberggpt: a large language model for finance, 2023. arxiv:2303.17564."},{"key":"294_CR6","doi-asserted-by":"crossref","unstructured":"Li J, Cheng X, Zhao WX, Nie J-Y, Wen J-R. Halueval: A large-scale hallucination evaluation benchmark for large language models. In: Proceedings of the 2023 conference on empirical methods in natural language processing, 2023;6449\u20136464.","DOI":"10.18653\/v1\/2023.emnlp-main.397"},{"key":"294_CR7","doi-asserted-by":"crossref","unstructured":"Vu T, Iyyer M, Wang X, Constant N, Wei J, Wei J, Tar C, Sung Y-H, Zhou D, Le Q, et al. Freshllms: refreshing large language models with search engine augmentation, 2023. arXiv:2310.03214.","DOI":"10.18653\/v1\/2024.findings-acl.813"},{"key":"294_CR8","unstructured":"Muhlgay D, Ram O, Magar I, Levine Y, Ratner N, Belinkov Y, Abend O, Leyton-Brown K, Shashua A, Shoham Y. Generating benchmarks for factuality evaluation of language models. In: Proceedings of the 18th conference of the european chapter of the association for computational linguistics (Volume 1: Long Papers), 2024;49\u201366."},{"key":"294_CR9","unstructured":"Hu X, Chen J, Li X, Guo Y, Wen L, Philip SY, Guo Z. Towards understanding factual knowledge of large language models. In: The twelfth international conference on learning representations."},{"key":"294_CR10","doi-asserted-by":"crossref","unstructured":"Luo Z, Xie Q, Ananiadou S. Factual consistency evaluation of summarisation in the era of large language models. Expert Syst Appl, 2024;124456.","DOI":"10.1016\/j.eswa.2024.124456"},{"key":"294_CR11","doi-asserted-by":"crossref","unstructured":"Tam D, Mascarenhas A, Zhang S, Kwan S, Bansal M, Raffel C. Evaluating the factual consistency of large language models through news summarization. In: Findings of the association for computational linguistics: ACL 2023, 2023;5220\u20135255.","DOI":"10.18653\/v1\/2023.findings-acl.322"},{"key":"294_CR12","unstructured":"Jin Z, Cao P, Wang C, He Z, Yuan H, Li J, Chen Y, Liu K, Zhao J. Rwku: Benchmarking real-world knowledge unlearning for large language models, 2024. arXiv:2406.10890."},{"key":"294_CR13","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J. Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the association for computational linguistics, 2002;311\u2013318.","DOI":"10.3115\/1073083.1073135"},{"key":"294_CR14","unstructured":"Lin C-Y. Rouge: A package for automatic evaluation of summaries. In: Text summarization branches out, 2004;74\u201381."},{"key":"294_CR15","unstructured":"Dong Q, Xu J, Kong L, Sui Z, Li L. Statistical knowledge assessment for large language models. Adv Neural Inf Process Syst 2024;36."},{"key":"294_CR16","unstructured":"Zhang T, Kishore V, Wu F, Weinberger KQ, Artzi Y. Bertscore: evaluating text generation with bert, 2019. arXiv:1904.09675."},{"key":"294_CR17","doi-asserted-by":"crossref","unstructured":"Sellam T, Das D, Parikh A. Bleurt: Learning robust metrics for text generation. In: Proceedings of the 58th annual meeting of the association for computational linguistics, 2020;7881\u20137892.","DOI":"10.18653\/v1\/2020.acl-main.704"},{"key":"294_CR18","first-page":"27263","volume":"34","author":"W Yuan","year":"2021","unstructured":"Yuan W, Neubig G, Liu P. Bartscore: evaluating generated text as text generation. Adv Neural Inf Process Syst. 2021;34:27263\u201377.","journal-title":"Adv Neural Inf Process Syst"},{"key":"294_CR19","doi-asserted-by":"crossref","unstructured":"Fu J, Ng SK, Jiang Z, Liu P. Gptscore: Evaluate as you desire. In: Proceedings of the 2024 conference of the North American chapter of the association for computational linguistics: human language technologies (Volume 1: Long Papers), 2024;6556\u20136576.","DOI":"10.18653\/v1\/2024.naacl-long.365"},{"key":"294_CR20","doi-asserted-by":"crossref","unstructured":"Lin S, Hilton J, Evans O. Truthfulqa: measuring how models mimic human falsehoods. In: Proceedings of the 60th annual meeting of the association for computational linguistics (Volume 1: Long Papers), 2022;3214\u20133252.","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"294_CR21","doi-asserted-by":"crossref","unstructured":"Chen L, Deng Y, Bian Y, Qin Z, Wu B, Chua T-S, Wong K-F. Beyond factuality: a comprehensive evaluation of large language models as knowledge generators. In: Proceedings of the 2023 conference on empirical methods in natural language processing, 2023;6325\u20136341.","DOI":"10.18653\/v1\/2023.emnlp-main.390"},{"key":"294_CR22","doi-asserted-by":"crossref","unstructured":"Utama PA, Bambrick J, Moosavi NS, Gurevych I. Falsesum: Generating document-level nli examples for recognizing factual inconsistency in summarization, 2022. arXiv:2205.06009.","DOI":"10.18653\/v1\/2022.naacl-main.199"},{"key":"294_CR23","doi-asserted-by":"crossref","unstructured":"Soleimani A, Monz C, Worring M. Nonfacts: Nonfactual summary generation for factuality evaluation in document summarization. In: Findings of the association for computational linguistics: ACL 2023, 2023;6405\u20136419.","DOI":"10.18653\/v1\/2023.findings-acl.400"},{"key":"294_CR24","doi-asserted-by":"crossref","unstructured":"Rykov E, Shishkina Y, Petrushina K, Titova K, Petrakov S, Panchenko A. Smurfcat at semeval-2024 task 6: Leveraging synthetic data for hallucination detection, 2024. arXiv:2404.06137.","DOI":"10.18653\/v1\/2024.semeval-1.125"},{"key":"294_CR25","unstructured":"Scir\u00e8 A, Bejgu AS, Tedeschi S, Ghonim K, Martelli F, Navigli R. Truth or mirage? towards end-to-end factuality evaluation with llm-oasis, 2024. arXiv:2411.19655."},{"key":"294_CR26","unstructured":"Chern I, Chern S, Chen S, Yuan W, Feng K, Zhou C, He J, Neubig G, Liu P, et al. Factool: Factuality detection in generative ai\u2013a tool augmented framework for multi-task and multi-domain scenarios, 2023. arXiv:2307.13528."},{"key":"294_CR27","unstructured":"Wang Y, Gangi Reddy R, Mujahid ZM, Arora A, Rubashevskii A, Geng J, Afzal OM, Pan L, Borenstein N, Pillai A, et al. Factcheck-gpt: End-to-end fine-grained document-level fact-checking and correction of llm output. arXiv e-prints, 2023;2311."},{"key":"294_CR28","unstructured":"Wang Y, Wang M, Iqbal H, Georgiev G, Geng J, Nakov P. Openfactcheck: A unified framework for factuality evaluation of llms, 2024. arXiv:2405.05583."},{"key":"294_CR29","doi-asserted-by":"crossref","unstructured":"Islam R, Moushi OM. Gpt-4o: The cutting-edge advancement in multimodal llm. Authorea Preprints 2024.","DOI":"10.36227\/techrxiv.171986596.65533294\/v1"},{"key":"294_CR30","unstructured":"Team G, Anil R, Borgeaud S, Alayrac J-B, Yu J, Soricut R, Schalkwyk J, Dai AM, Hauth A, Millican K, et al. Gemini: a family of highly capable multimodal models, 2023. arXiv:2312.11805."},{"key":"294_CR31","unstructured":"Dubey A, Jauhri A, Pandey A, Kadian A, Al-Dahle A, Letman A, Mathur A, Schelten A, Yang A, Fan A, et al. The llama 3 herd of models, 2024. arXiv:2407.21783."},{"key":"294_CR32","unstructured":"Yang A, Yang B, Zhang B, Hui B, Zheng B, Yu B, Li C, Liu D, Huang F, Wei H, et al. Qwen2. 5 technical report, 2024. arXiv:2412.15115."},{"key":"294_CR33","unstructured":"Bi X, Chen D, Chen G, Chen S, Dai D, Deng C, Ding H, Dong K, Du Q, Fu Z, et al. Deepseek llm: Scaling open-source language models with longtermism, 2024. arXiv:2401.02954."},{"key":"294_CR34","unstructured":"Weng Y, Zhu M, Xia F, Li B, He S, Liu K, Zhao J. Mastering symbolic operations: augmenting language models with compiled neural networks, 2023. arXiv:2304.01665."},{"key":"294_CR35","doi-asserted-by":"crossref","unstructured":"Xu Y, He S, Chen J, Wang Z, Song Y, Tong H, Liu G, Liu K, Zhao J. Generate-on-graph: Treat llm as both agent and kg in incomplete knowledge graph question answering, 2024. arXiv:2404.14741.","DOI":"10.18653\/v1\/2024.emnlp-main.1023"},{"key":"294_CR36","doi-asserted-by":"crossref","unstructured":"Yang L, Chen H, Li Z, Ding X, Wu X. Give us the facts: enhancing large language models with knowledge graphs for fact-aware language modeling. IEEE Trans Knowl Data Eng; 2024.","DOI":"10.1109\/TKDE.2024.3360454"},{"key":"294_CR37","unstructured":"Jiao S, Zhang G, Li G. A prompt-focused privacy evaluation and obfuscation method for large language model. Netinfo Secur, 2024;1396\u20131408"}],"container-title":["Discover Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44163-025-00294-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s44163-025-00294-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44163-025-00294-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T10:55:46Z","timestamp":1748343346000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s44163-025-00294-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,27]]},"references-count":37,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["294"],"URL":"https:\/\/doi.org\/10.1007\/s44163-025-00294-w","relation":{},"ISSN":["2731-0809"],"issn-type":[{"value":"2731-0809","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,27]]},"assertion":[{"value":"29 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 May 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The author has no relevant financial or non-financial interests to disclose.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"77"}}