{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T15:02:05Z","timestamp":1781190125350,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":36,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819596935","type":"print"},{"value":"9789819596942","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-9694-2_2","type":"book-chapter","created":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T14:23:48Z","timestamp":1781187828000},"page":"12-26","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["GeoClaim: Programmable Geoscientific Fact Verification and\u00a0Judge-Guided Evaluation for\u00a0Open-Ended Mineral Exploration QA"],"prefix":"10.1007","author":[{"given":"Yuang","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pu","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fanyu","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaheng","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinjun","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,1]]},"reference":[{"key":"2_CR1","unstructured":"Schmidtov\u00e1, P., et al.: Automatic metrics in natural language generation: a survey of current evaluation practices. arXiv preprint arXiv:2408.09169 (2024)"},{"key":"2_CR2","doi-asserted-by":"crossref","unstructured":"Samad, A. S., Sushma, R., Mohan, G. B., et\u00a0al.: Advancing abstractive summarization: evaluating GPT-2, BART, T5-Small, and Pegasus models with baseline in ROUGE and BLEU metrics. In: Proceedings of the International Conference on Innovations in Cybersecurity and Data Science (ICICDS), pp.\u00a0119\u2013131. Springer, Singapore (2024)","DOI":"10.1007\/978-981-97-5791-6_10"},{"key":"2_CR3","first-page":"229","volume":"4","author":"PR Chinthalapelly","year":"2024","unstructured":"Chinthalapelly, P.R., Selvaraj, A., Murthy, C.J.: Evaluating LLM outputs for legal contracts using BLEU, ROUGE, and BERTScore. Am. J. Data Sci. Artif. Intell. Innov. 4, 229\u2013262 (2024)","journal-title":"Am. J. Data Sci. Artif. Intell. Innov."},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Smee, B. W., Bloom, L., Arne, D., et\u00a0al.: Practical applications of quality assurance and quality control in mineral exploration, resource estimation and mining programmes: a review of recommended international practices. Geochem. Explorat. Environ. Anal. 24(2), geochem2023\u2013046 (2024)","DOI":"10.1144\/geochem2023-046"},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Fu, Y., Wang, M., Wang, C., et\u00a0al.: GeoMinLM: a large language model in geology and mineral survey in Yunnan Province. Ore Geol. Rev. 106638 (2025)","DOI":"10.1016\/j.oregeorev.2025.106638"},{"key":"2_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Z., Wang, X., Zhang, X., et\u00a0al.: GeoFactory: an LLM performance enhancement framework for geoscience factual and inferential tasks. Big Earth Data \u201333 (2025)","DOI":"10.1080\/20964471.2025.2506291"},{"issue":"3","key":"2_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3641289","volume":"15","author":"Y Chang","year":"2024","unstructured":"Chang, Y., Wang, X., Wang, J., et al.: A survey on evaluation of large language models. ACM Trans. Intell. Syst. Technol. 15(3), 1\u201345 (2024)","journal-title":"ACM Trans. Intell. Syst. Technol."},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Zhou, P., Peng, X., Song, J., et\u00a0al.: OpenING: a comprehensive benchmark for judging open-ended interleaved image-text generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp.\u00a056\u201366 (2025)","DOI":"10.1109\/CVPR52734.2025.00015"},{"key":"2_CR9","unstructured":"Liu, S., Gemp, I., Marris, L., et\u00a0al.: Re-evaluating open-ended evaluation of large language models. arXiv preprint arXiv:2502.20170 (2025)"},{"key":"2_CR10","doi-asserted-by":"crossref","unstructured":"Li, D., Jiang, B., Huang, L., et\u00a0al.: From generation to judgment: opportunities and challenges of LLM-as-a-judge. In: Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp.\u00a02757\u20132791 (2025)","DOI":"10.18653\/v1\/2025.emnlp-main.138"},{"key":"2_CR11","doi-asserted-by":"crossref","unstructured":"Cao, Z., Ma, Z., Chen, M.: An evaluation system for large language models based on open-ended questions. In: Proceedings of the 2024 IEEE 11th International Conference on Cyber Security and Cloud Computing (CSCloud), pp.\u00a065\u201372. IEEE (2024)","DOI":"10.1109\/CSCloud62866.2024.00019"},{"key":"2_CR12","doi-asserted-by":"crossref","unstructured":"Zhao, W., Liu, Y., Niu, T., et\u00a0al.: DIVKNOWQA: assessing the reasoning ability of LLMs via open-domain question answering over knowledge base and text. In: Findings of the Association for Computational Linguistics: NAACL 2024, pp.\u00a051\u201368 (2024)","DOI":"10.18653\/v1\/2024.findings-naacl.5"},{"key":"2_CR13","unstructured":"Shafayat, S., Kim, E., Oh, J., et\u00a0al.: Multi-fact: assessing factuality of multilingual LLMs using FActScore. arXiv preprint arXiv:2402.18045 (2024)"},{"key":"2_CR14","doi-asserted-by":"crossref","unstructured":"Chen, Y. S., Jin, J., Kuo, P. T., et\u00a0al.: LLMs are biased evaluators but not biased for fact-centric retrieval augmented generation. In: Findings of the Association for Computational Linguistics: ACL 2025, pp.\u00a026669\u201326684 (2025)","DOI":"10.18653\/v1\/2025.findings-acl.1369"},{"issue":"10","key":"2_CR15","doi-asserted-by":"publisher","first-page":"382","DOI":"10.3390\/geosciences15100382","volume":"15","author":"B Zhou","year":"2025","unstructured":"Zhou, B., Li, K.: Fusing geoscience large language models and lightweight RAG for enhanced geological question answering. Geosciences 15(10), 382 (2025)","journal-title":"Geosciences"},{"key":"2_CR16","unstructured":"Li, H., Dong, Q., Chen, J., et\u00a0al.: LLMs-as-judges: a comprehensive survey on LLM-based evaluation methods. arXiv preprint arXiv:2412.05579 (2024)"},{"key":"2_CR17","unstructured":"Yu, B., Shen, T., Na, H., et\u00a0al.: MineAgent: towards remote-sensing mineral exploration with multimodal large language models. arXiv preprint arXiv:2412.17339 (2024)"},{"key":"2_CR18","unstructured":"Dorner, F.E., Nastl, V.Y., Hardt, M.: Limits to scalable evaluation at the frontier: LLM as judge will not beat twice the data. arXiv preprint arXiv:2410.13341 (2024)"},{"key":"2_CR19","unstructured":"Jayakumar, E., Dash, N. S., Mukherjee, D.: Large language model agent personality and response appropriateness: evaluation by human linguistic experts, LLM-as-judge, and natural language processing models. arXiv preprint arXiv:2510.23875 (2025)"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Zhou, X., Kim, K., Zhang, T., et\u00a0al.: An LLM-as-judge metric for bridging the gap with human evaluation in SE tasks. arXiv preprint arXiv:2505.20854 (2025)","DOI":"10.1109\/ASE63991.2025.00214"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Xu, A., Bansal, S., Ming, Y., et\u00a0al.: Does context matter? ContextualJudgeBench for evaluating LLM-based judges in contextual settings. arXiv preprint arXiv:2503.15620 (2025)","DOI":"10.18653\/v1\/2025.acl-long.470"},{"key":"2_CR22","doi-asserted-by":"crossref","unstructured":"Dechtiar, M., Katz, D. M., Jaume, S., et\u00a0al.: LLM as a judge for evaluating contract graphs: multi-judge benchmarking and agentic uncertainty-aware refinement. SSRN 5937996 (2025)","DOI":"10.2139\/ssrn.5937996"},{"issue":"2","key":"2_CR23","doi-asserted-by":"publisher","first-page":"376","DOI":"10.1080\/13658816.2024.2412731","volume":"39","author":"L Hu","year":"2025","unstructured":"Hu, L., Li, W., Xu, J., et al.: GeoEntity-type constrained knowledge graph embedding for predicting natural-language spatial relations. Int. J. Geogr. Inf. Sci. 39(2), 376\u2013399 (2025)","journal-title":"Int. J. Geogr. Inf. Sci."},{"key":"2_CR24","doi-asserted-by":"crossref","unstructured":"Raz, T., Luchini, S., Beaty, R., et\u00a0al.: Automated scoring of open-ended question complexity: a large language model approach (2024)","DOI":"10.21203\/rs.3.rs-3890828\/v1"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Zhang, K., Wu, P., Yu, B., et\u00a0al.: Logical rule-constrained large language models for document-level relation extraction. In: Proceedings of the CCF International Conference on Natural Language Processing and Chinese Computing, pp.\u00a0132\u2013145. Springer, Singapore (2025)","DOI":"10.1007\/978-981-95-3343-5_11"},{"key":"2_CR26","unstructured":"Dmonte, A., Oruche, R., Zampieri, M., et\u00a0al.: Claim verification in the age of large language models: a survey. arXiv preprint arXiv:2408.14317 (2024)"},{"key":"2_CR27","doi-asserted-by":"crossref","unstructured":"Wang, H., Pan, Y., Song, X., et\u00a0al.: F2RL: factuality and faithfulness reinforcement learning framework for claim-guided evidence-supported counterspeech generation. In: Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp.\u00a04457\u20134470 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.255"},{"issue":"1","key":"2_CR28","doi-asserted-by":"publisher","first-page":"1418","DOI":"10.1038\/s41467-024-45563-x","volume":"15","author":"J Dagdelen","year":"2024","unstructured":"Dagdelen, J., Dunn, A., Lee, S., et al.: Structured information extraction from scientific text with large language models. Nat. Commun. 15(1), 1418 (2024)","journal-title":"Nat. Commun."},{"issue":"1","key":"2_CR29","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1007\/s13042-023-01885-8","volume":"15","author":"M Cui","year":"2024","unstructured":"Cui, M., Huang, R., Hu, Z., et al.: Semantic rule-based information extraction for meteorological reports. Int. J. Mach. Learn. Cybern. 15(1), 177\u2013188 (2024)","journal-title":"Int. J. Mach. Learn. Cybern."},{"issue":"11","key":"2_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3674501","volume":"56","author":"X Zhao","year":"2024","unstructured":"Zhao, X., Deng, Y., Yang, M., et al.: A comprehensive survey on relation extraction: recent advances and new frontiers. ACM Comput. Surv. 56(11), 1\u201339 (2024)","journal-title":"ACM Comput. Surv."},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Li, F., Hogg, D. C., Cohn, A. G.: Advancing spatial reasoning in large language models: an in-depth evaluation and enhancement using the StepGame benchmark. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, no. 17, pp. 18500\u201318507 (2024)","DOI":"10.1609\/aaai.v38i17.29811"},{"key":"2_CR32","unstructured":"Sykes, B., Simon, L., Rabin, J.: Unifying and extending precision\u2013recall metrics for assessing generative models. arXiv preprint arXiv:2405.01611 (2024)"},{"key":"2_CR33","doi-asserted-by":"crossref","unstructured":"Diaz, F., Ekstrand, M.D., Mitra, B.: Recall, robustness, and lexicographic evaluation. ACM Trans. Recommend. Syst. (2025)","DOI":"10.1145\/3728373"},{"issue":"12","key":"2_CR34","doi-asserted-by":"publisher","first-page":"7717","DOI":"10.1007\/s10115-024-02217-0","volume":"66","author":"E Davoodijam","year":"2024","unstructured":"Davoodijam, E., Alambardar Meybodi, M.: Evaluation metrics on text summarization: a comprehensive survey. Knowl. Inf. Syst. 66(12), 7717\u20137738 (2024)","journal-title":"Knowl. Inf. Syst."},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Choi, J.H.: Large language models are unreliable judges. SSRN 5188865 (2025)","DOI":"10.2139\/ssrn.5188865"},{"key":"2_CR36","unstructured":"Tang, Y., Feng, K., Wang, Y., et\u00a0al.: Learning an efficient multi-turn dialogue evaluator from multiple judges. arXiv preprint arXiv:2508.00454 (2025)"}],"container-title":["Lecture Notes in Computer Science","Evaluation Science and Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-9694-2_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T14:23:59Z","timestamp":1781187839000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-9694-2_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819596935","9789819596942"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-9694-2_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors declare no competing interests.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"Bench","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Benchmarking, Measuring and Optimization","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chengdu","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 December 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 December 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"bench2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.benchcouncil.org\/bench2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}