{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:52:34Z","timestamp":1781931154391,"version":"3.54.5"},"reference-count":7,"publisher":"Association for Natural Language Processing","issue":"2","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Journal of Natural Language Processing"],"published-print":{"date-parts":[[2026]]},"DOI":"10.5715\/jnlp.33.1027","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T22:11:27Z","timestamp":1781475087000},"page":"1027-1032","source":"Crossref","is-referenced-by-count":0,"title":["Research Process of Likelihood-based Mitigation of Evaluation Bias in Large Language Models","\u300c\u5927\u898f\u6a21\u8a00\u8a9e\u30e2\u30c7\u30eb\u306b\u304a\u3051\u308b\u8a55\u4fa1\u30d0\u30a4\u30a2\u30b9\u306e\u5c24\u5ea6\u306b\u57fa\u3065\u304f\u7de9\u548c\u300d\u306e\u7814\u7a76\u904e\u7a0b"],"prefix":"10.5715","volume":"33","author":[{"given":"Masanari","family":"Oi","sequence":"first","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"3685","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"Huang, J., Gu, S., Hou, L., Wu, Y., Wang, X., Yu, H., and Han, J. (2023). \u201cLarge Language Models Can Self-Improve.\u201d In <i>Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing<\/i>, pp. 1051\u20131068. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2023.emnlp-main.67"},{"key":"2","doi-asserted-by":"crossref","unstructured":"Kuribayashi, T., Ito, T., Suzuki, J., and Inui, K. (2020). \u201cLanguage Models as an Alternative Evaluator of Word Order Hypotheses: A Case Study in Japanese.\u201d In <i>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics<\/i>, pp. 488\u2013504. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2020.acl-main.47"},{"key":"3","unstructured":"Ohi, M., Kaneko, M., Okazaki, N., and Inoue, N. (2026). \u201cMulti-modal, Multi-task, Multi-criteria Automatic Evaluation with Vision Language Models.\u201d <i>arXiv preprint arXiv:2412.14613<\/i>."},{"key":"4","doi-asserted-by":"crossref","unstructured":"\u5927\u4e95\u8056\u4e5f\uff0c\u91d1\u5b50\u6b63\u5f18\uff0c\u5c0f\u6c60\u9686\u6597\uff0cMengsayLoem\uff0c\u5ca1\u5d0e\u76f4\u89b3 (2025). \u5927\u898f\u6a21\u8a00\u8a9e\u30e2\u30c7\u30eb\u306b\u304a\u3051\u308b\u8a55\u4fa1\u30d0\u30a4\u30a2\u30b9\u306e\u5c24\u5ea6\u306b\u57fa\u3065\u304f\u7de9\u548c. \u81ea\u7136\u8a00\u8a9e\u51e6\u7406, <b>32<\/b> (2), pp. 480\u2013496. [M. Oi et al. (2025). Ronbunshi \u201cShizengengo Shori\u201d Daikibogengomoderu niokeru Hyoukabaiasu no Yuudo nimotoduku Kanwa. Journal of Natural Language Processing, 32(2), pp. 480\u2013496.].","DOI":"10.5715\/jnlp.32.480"},{"key":"5","doi-asserted-by":"crossref","unstructured":"Paul, D., Ismayilzada, M., Peyrard, M., Borges, B., Bosselut, A., West, R., and Faltings, B. (2024). \u201cREFINER: Reasoning Feedback on Intermediate Representations.\u201d In <i>Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics (Volume 1: Long Papers)<\/i>, pp. 1100\u20131126. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.eacl-long.67"},{"key":"6","unstructured":"Touvron, H., Martin, L., Stone, K., et al. (2023). \u201cLlama 2: Open Foundation and Fine-Tuned Chat Models.\u201d <i>arXiv preprint arXiv:2007.09288<\/i>."},{"key":"7","doi-asserted-by":"crossref","unstructured":"Zheng, L., Chiang, W.-L., Sheng, Y., Zhuang, S., Wu, Z., Zhuang, Y., Lin, Z., Li, Z., Li, D., Xing, E., Zhang, H., Gonzalez, J. E., and Stoica, I. (2023). \u201cJudging LLM-as-a-Judge with MT-Bench and Chatbot Arena.\u201d In <i>Advances in Neural Information Processing Systems<\/i>, Vol. 36, pp. 46595\u201346623. Curran Associates, Inc.","DOI":"10.52202\/075280-2020"}],"container-title":["Journal of Natural Language Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_1027\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:43:00Z","timestamp":1781930580000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_1027\/_article\/-char\/ja\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":7,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/doi.org\/10.5715\/jnlp.33.1027","relation":{},"ISSN":["1340-7619","2185-8314"],"issn-type":[{"value":"1340-7619","type":"print"},{"value":"2185-8314","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}