{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T21:43:46Z","timestamp":1769550226460,"version":"3.49.0"},"publisher-location":"Singapore","reference-count":23,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819783663","type":"print"},{"value":"9789819783670","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,29]],"date-time":"2024-11-29T00:00:00Z","timestamp":1732838400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,29]],"date-time":"2024-11-29T00:00:00Z","timestamp":1732838400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8367-0_27","type":"book-chapter","created":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T11:56:16Z","timestamp":1732794976000},"page":"451-462","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Mitigating the\u00a0Bias of\u00a0Large Language Model Evaluation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-0736-1652","authenticated-orcid":false,"given":"Hongli","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3544-9719","authenticated-orcid":false,"given":"Hui","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunfei","family":"Long","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3132-3059","authenticated-orcid":false,"given":"Conghui","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hailong","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5940-0266","authenticated-orcid":false,"given":"Muyun","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4659-4935","authenticated-orcid":false,"given":"Tiejun","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,29]]},"reference":[{"key":"27_CR1","unstructured":"Achiam, J., et\u00a0al.: Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"27_CR2","doi-asserted-by":"publisher","unstructured":"Bang, Y., et al.: A multitask, multilingual, multimodal evaluation of ChatGPT on reasoning, hallucination, and interactivity. In: Park, J.C., et al. (eds.) Proceedings of the 13th International Joint Conference on Natural Language Processing and the 3rd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 675\u2013718. Association for Computational Linguistics, Nusa Dua (2023). https:\/\/doi.org\/10.18653\/v1\/2023.ijcnlp-main.45","DOI":"10.18653\/v1\/2023.ijcnlp-main.45"},{"key":"27_CR3","unstructured":"Bubeck, S., et al.: Sparks of artificial general intelligence: early experiments with gpt-4 (2023)"},{"key":"27_CR4","unstructured":"Chang, Y., et\u00a0al.: A survey on evaluation of large language models. ACM Trans. Intell. Syst. Technol. (2023)"},{"key":"27_CR5","unstructured":"Chiang, W.L., et al.: Vicuna: an open-source chatbot impressing gpt-4 with 90%* chatgpt quality (2023). https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"27_CR6","doi-asserted-by":"crossref","unstructured":"Fu, J., Ng, S.K., Jiang, Z., Liu, P.: Gptscore: evaluate as you desire (2023)","DOI":"10.18653\/v1\/2024.naacl-long.365"},{"key":"27_CR7","unstructured":"Guo, B., et al.: How close is chatgpt to human experts? comparison corpus, evaluation, and detection (2023)"},{"key":"27_CR8","unstructured":"Huang, H., Qu, Y., Liu, J., Yang, M., Zhao, T.: An empirical study of LLM-as-a-judge for LLM evaluation: fine-tuned judge models are task-specific classifiers (2024)"},{"key":"27_CR9","unstructured":"Li, X., et al.: Alpacaeval: an automatic evaluator of instruction-following models (2023). https:\/\/github.com\/tatsu-lab\/alpaca_eval"},{"key":"27_CR10","unstructured":"Liang, P., et\u00a0al.: Holistic evaluation of language models. arXiv preprint arXiv:2211.09110 (2022)"},{"key":"27_CR11","unstructured":"Lin, C.Y.: Rouge: a package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381 (2004)"},{"key":"27_CR12","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"27_CR13","doi-asserted-by":"crossref","unstructured":"Qin, C., Zhang, A., Zhang, Z., Chen, J., Yasunaga, M., Yang, D.: Is chatgpt a general-purpose natural language processing task solver? arXiv preprint arXiv:2302.06476 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.85"},{"key":"27_CR14","unstructured":"Saito, K., Wachi, A., Wataoka, K., Akimoto, Y.: Verbosity bias in preference labeling by large language models. arXiv preprint arXiv:2310.10076 (2023)"},{"key":"27_CR15","doi-asserted-by":"publisher","unstructured":"Su, H., et al.: One embedder, any task: instruction-finetuned text embeddings. In: Rogers, A., Boyd-Graber, J., Okazaki, N. (eds.) Findings of the Association for Computational Linguistics: ACL 2023, pp. 1102\u20131121. Association for Computational Linguistics, Toronto (2023). https:\/\/doi.org\/10.18653\/v1\/2023.findings-acl.71","DOI":"10.18653\/v1\/2023.findings-acl.71"},{"key":"27_CR16","unstructured":"Wang, P., et al.: Large language models are not fair evaluators (2023)"},{"key":"27_CR17","unstructured":"Wang, Y., et al.: Pandalm: an automatic evaluation benchmark for LLM instruction tuning optimization (2024)"},{"key":"27_CR18","unstructured":"Yu, J., et al..: KoLA: carefully benchmarking world knowledge of large language models. In: The Twelfth International Conference on Learning Representations (2024). https:\/\/openreview.net\/forum?id=AqN23oqraW"},{"key":"27_CR19","unstructured":"Zeng, Z., Yu, J., Gao, T., Meng, Y., Goyal, T., Chen, D.: Evaluating large language models at evaluating instruction following. arXiv preprint arXiv:2310.07641 (2023)"},{"key":"27_CR20","unstructured":"Zhao, T.Z., Wallace, E., Feng, S., Klein, D., Singh, S.: Calibrate before use: improving few-shot performance of language models (2021)"},{"key":"27_CR21","unstructured":"Zheng, L., et\u00a0al.: Judging LLM-as-a-judge with MT-bench and chatbot arena. arXiv preprint arXiv:2306.05685 (2023)"},{"key":"27_CR22","doi-asserted-by":"crossref","unstructured":"Zhong, W., et al.: Agieval: a human-centric benchmark for evaluating foundation models (2023)","DOI":"10.18653\/v1\/2024.findings-naacl.149"},{"key":"27_CR23","unstructured":"Zhu, L., Wang, X., Wang, X.: Judgelm: fine-tuned large language models are scalable judges. arXiv preprint arXiv:2310.17631 (2023)"}],"container-title":["Lecture Notes in Computer Science","Chinese Computational Linguistics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8367-0_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T12:08:36Z","timestamp":1732795716000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8367-0_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,29]]},"ISBN":["9789819783663","9789819783670"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8367-0_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,29]]},"assertion":[{"value":"29 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"CCL","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China National Conference on Chinese Computational Linguistics","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taiyuan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 July 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 July 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cncl2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/cips-cl.org\/static\/CCL2024\/en\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}