{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:52:33Z","timestamp":1781931153647,"version":"3.54.5"},"reference-count":42,"publisher":"Association for Natural Language Processing","issue":"2","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Journal of Natural Language Processing"],"published-print":{"date-parts":[[2026]]},"DOI":"10.5715\/jnlp.33.848","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T22:11:44Z","timestamp":1781475104000},"page":"848-879","source":"Crossref","is-referenced-by-count":0,"title":["Automatic Post-editing through Word-level Quality Estimation with Minimum Bayes Risk Decoding"],"prefix":"10.5715","volume":"33","author":[{"given":"Youyuan","family":"Lin","sequence":"first","affiliation":[{"name":"Kyoto University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masaaki","family":"Nagata","sequence":"additional","affiliation":[{"name":"NTT, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenhui","family":"Chu","sequence":"additional","affiliation":[{"name":"Kyoto University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"3685","reference":[{"key":"1","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F. L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et al. (2023). \u201cGPT-4 Technical Report.\u201d <i>arXiv preprint arXiv:2303.08774<\/i>."},{"key":"2","unstructured":"Akhbardeh, F., Arkhangorodsky, A., Biesialska, M., Bojar, O., Chatterjee, R., Chaudhary, V., Costa-jussa, M. R., Espa\u00f1a-Bonet, C., Fan, A., Federmann, C., et al. (2021). \u201cFindings of the 2021 Conference on Machine Translation (WMT21).\u201d In <i>Proceedings of the 6th Conference on Machine Translation<\/i>, pp. 1\u201388. Association for Computational Linguistics."},{"key":"3","unstructured":"Alves, D. M., Pombal, J., Guerreiro, N. M., Martins, P. H., Alves, J., Farajian, A., Peters, B., Rei, R., Fernandes, P., Agrawal, S., Colombo, P., de Souza, J. G. C., and Martins, A. F. T. (2024). \u201cTower: An Open Multilingual Large Language Model for Translation-Related Tasks.\u201d <i>arXiv preprint arXiv:2402.17733<\/i>."},{"key":"4","doi-asserted-by":"crossref","unstructured":"Bhattacharyya, P., Chatterjee, R., Freitag, M., Kanojia, D., Negri, M., and Turchi, M. (2023). \u201cFindings of the WMT 2023 Shared Task on Automatic Post-Editing.\u201d In <i>Proceedings of the 8th Conference on Machine Translation<\/i>, pp. 672\u2013681.","DOI":"10.18653\/v1\/2023.wmt-1.55"},{"key":"5","unstructured":"Chen, X., Mitsuda, K., Wakatsuki, T., and Sawada, K. (2024). \u201crinna\/llama-3-youko-8b-instruct.\u201d https:\/\/huggingface.co\/rinna\/llama-3-youko-8b-instruct."},{"key":"6","doi-asserted-by":"crossref","unstructured":"Deguchi, H., Imamura, K., Kaneko, M., Nishida, Y., Sakai, Y., Vasselli, J., Vu, H. H., and Watanabe, T. (2022). \u201cNAIST-NICT-TIT WMT22 General MT Task Submission.\u201d In <i>Proceedings of the 7th Conference on Machine Translation (WMT)<\/i>, pp. 244\u2013250.","DOI":"10.18653\/v1\/2022.wmt-1.16"},{"key":"7","unstructured":"Deguchi, H., Nagata, M., and Watanabe, T. (2024). \u201cDetector Corrector: Edit-Based Automatic Post Editing for Human Post Editing.\u201d In <i>European Association for Machine Translation<\/i>."},{"key":"8","doi-asserted-by":"crossref","unstructured":"Deoghare, S., Kanojia, D., Blain, F., Ranasinghe, T., and Bhattacharyya, P. (2023). \u201cQuality Estimation-Assisted Automatic Post-Editing.\u201d In <i>Findings of the Association for Computational Linguistics: EMNLP 2023<\/i>, pp. 1686\u20131698.","DOI":"10.18653\/v1\/2023.findings-emnlp.115"},{"key":"9","unstructured":"Dong, Q., Li, L., Dai, D., Zheng, C., Wu, Z., Chang, B., Sun, X., Xu, J., and Sui, Z. (2022). \u201cA Survey for In-context Learning.\u201d <i>arXiv preprint arXiv:2301.00234<\/i>."},{"key":"10","unstructured":"Dyer, C., Chahuneau, V., and Smith, N. A. (2013). \u201cA Simple, Fast, and Effective Reparameterization of IBM Model 2.\u201d In <i>Proceedings of the 2013 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies<\/i>, pp. 644\u2013648."},{"key":"11","doi-asserted-by":"crossref","unstructured":"Fernandes, P., Deutsch, D., Finkelstein, M., Riley, P., Martins, A. F., Neubig, G., Garg, A., Clark, J. H., Freitag, M., and Firat, O. (2023). \u201cThe Devil is in the Errors: Leveraging Large Language Models for Fine-grained Machine Translation Evaluation.\u201d <i>arXiv preprint arXiv:2308.07286<\/i>.","DOI":"10.18653\/v1\/2023.wmt-1.100"},{"key":"12","unstructured":"Fomicheva, M., Sun, S., Fonseca, E., Zerva, C., Blain, F., Chaudhary, V., Guzm\u00e1n, F., Lopatina, N., Specia, L., and Martins, A. F. (2020). \u201cMLQE-PE: A Multilingual Quality Estimation and Post-editing Dataset.\u201d <i>arXiv preprint arXiv:2010.04480<\/i>."},{"key":"13","doi-asserted-by":"crossref","unstructured":"Freitag, M., Ghorbani, B., and Fernandes, P. (2023). \u201cEpsilon Sampling Rocks: Investigating Sampling Strategies for Minimum Bayes Risk Decoding for Machine Translation.\u201d <i>arXiv preprint arXiv:2305.09860<\/i>.","DOI":"10.18653\/v1\/2023.findings-emnlp.617"},{"key":"14","doi-asserted-by":"crossref","unstructured":"Guerreiro, N. M., Rei, R., van Stigt, D., Coheur, L., Colombo, P., and Martins, A. F. (2023). \u201cxCOMET: Transparent Machine Translation Evaluation through Fine-grained Error Detection.\u201d <i>arXiv preprint arXiv:2310.10482<\/i>.","DOI":"10.1162\/tacl_a_00683"},{"key":"15","unstructured":"Hayakawa, T., and Arase, Y. (2020). \u201cFine-grained Error Analysis on English-to-Japanese Machine Translation in the Medical Domain.\u201d In <i>Proceedings of the 22nd Annual Conference of the European Association for Machine Translation<\/i>, pp. 155\u2013164."},{"key":"16","unstructured":"Hu, E. J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., and Chen, W. (2021). \u201cLora: Low-rank Adaptation of Large Language Models.\u201d <i>arXiv preprint arXiv:2106.09685<\/i>."},{"key":"17","doi-asserted-by":"crossref","unstructured":"Kalkar, S., Matsuzaki, Y., and Li, B. (2022). \u201cKYB General Machine Translation Systems for WMT22.\u201d In <i>Proceedings of the 7th Conference on Machine Translation (WMT)<\/i>, pp. 290\u2013294.","DOI":"10.18653\/v1\/2022.wmt-1.22"},{"key":"18","doi-asserted-by":"crossref","unstructured":"Ki, D., and Carpuat, M. (2024). \u201cGuiding Large Language Models to Post-Edit Machine Translation with Error Annotations.\u201d In Duh, K., Gomez, H., and Bethard, S. (Eds.), <i>Findings of the Association for Computational Linguistics: NAACL 2024<\/i>, pp. 4253\u20134273, Mexico City, Mexico. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.findings-naacl.265"},{"key":"19","doi-asserted-by":"crossref","unstructured":"Kocmi, T., and Federmann, C. (2023). \u201cGEMBA-MQM: Detecting Translation Quality Error Spans with GPT-4.\u201d <i>arXiv preprint arXiv:2310.13988<\/i>.","DOI":"10.18653\/v1\/2023.wmt-1.64"},{"key":"20","unstructured":"Kumar, S., and Byrne, W. (2004). \u201cMinimum Bayes-risk Decoding for Statistical Machine Translation.\u201d In <i>Proceedings of the Human Language Technology Conference of the North American Chapter of the Association for Computational Linguistics: HLT-NAACL 2004<\/i>, pp. 169\u2013176."},{"key":"21","unstructured":"Lightman, H., Kosaraju, V., Burda, Y., Edwards, H., Baker, B., Lee, T., Leike, J., Schulman, J., Sutskever, I., and Cobbe, K. (2023). \u201cLet\u2019s Verify Step by Step.\u201d In <i>The 12th International Conference on Learning Representations<\/i>."},{"key":"22","doi-asserted-by":"crossref","unstructured":"Liu, Z., Riley, P., Deutsch, D., Lui, A., Niu, M., Shah, A., and Freitag, M. (2024). \u201cBeyond Human-Only: Evaluating Human-Machine Collaboration for Collecting High-Quality Translation Data.\u201d <i>arXiv preprint arXiv:2410.11056<\/i>.","DOI":"10.18653\/v1\/2024.wmt-1.110"},{"key":"23","doi-asserted-by":"crossref","unstructured":"Lommel, A., Uszkoreit, H., and Burchardt, A. (2014). \u201cMultidimensional Quality Metrics (MQM): A Framework for Declaring and Describing Translation Quality Metrics.\u201d <i>Tradum\u00e0tica<\/i>, pp. 455\u2013463.","DOI":"10.5565\/rev\/tradumatica.77"},{"key":"24","unstructured":"Lu, Q., Ding, L., Zhang, K., Zhang, J., and Tao, D. (2024). \u201cMQM-APE: Toward High-Quality Error Annotation Predictors with Automatic Post-Editing in LLM Translation Evaluators.\u201d <i>arXiv preprint arXiv:2409.14335<\/i>."},{"key":"25","doi-asserted-by":"crossref","unstructured":"Morishita, M., Kudo, K., Oka, Y., Chousa, K., Kiyono, S., Takase, S., and Suzuki, J. (2022). \u201cNT5 at WMT 2022 General Translation Task.\u201d In <i>Proceedings of the 7th Conference on Machine Translation (WMT)<\/i>, pp. 318\u2013325.","DOI":"10.18653\/v1\/2022.wmt-1.25"},{"key":"26","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., and Zhu, W.-J. (2002). \u201cBleu: A Method for Automatic Evaluation of Machine Translation.\u201d In <i>Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics<\/i>, pp. 311\u2013318.","DOI":"10.3115\/1073083.1073135"},{"key":"27","doi-asserted-by":"crossref","unstructured":"Post, M. (2018). \u201cA Call for Clarity in Reporting BLEU Scores.\u201d In <i>Proceedings of the 3rd Conference on Machine Translation: Research Papers<\/i>, pp. 186\u2013191, Brussels, Belgium. Association for Computational Linguistics.","DOI":"10.18653\/v1\/W18-6319"},{"key":"28","doi-asserted-by":"crossref","unstructured":"Raunak, V., Sharaf, A., Awadallah, H. H., and Menezes, A. (2023). \u201cLeveraging GPT-4 for Automatic Translation Post-Editing.\u201d <i>arXiv preprint arXiv:2305.14878<\/i>.","DOI":"10.18653\/v1\/2023.findings-emnlp.804"},{"key":"29","doi-asserted-by":"crossref","unstructured":"Rei, R., Treviso, M., Guerreiro, N. M., Zerva, C., Farinha, A. C., Maroti, C., De Souza, J. G., Glushkova, T., Alves, D. M., Lavie, A., and Martins, A. F. T. (2022). \u201cCometKiwi: IST-unbabel 2022 Submission for the Quality Estimation Shared Task.\u201d <i>arXiv preprint arXiv:2209.06243<\/i>.","DOI":"10.18653\/v1\/2022.wmt-1.60"},{"key":"30","doi-asserted-by":"crossref","unstructured":"Sellam, T., Das, D., and Parikh, A. P. (2020). \u201cBLEURT: Learning Robust Metrics for Text Generation.\u201d <i>arXiv preprint arXiv:2004.04696<\/i>.","DOI":"10.18653\/v1\/2020.acl-main.704"},{"key":"31","doi-asserted-by":"crossref","unstructured":"Shenoy, R., Herbig, N., Kr\u00fcger, A., and van Genabith, J. (2021). \u201cInvestigating the Helpfulness of Word-level Quality Estimation for Post-editing Machine Translation Output.\u201d In <i>Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing<\/i>, pp. 10173\u201310185.","DOI":"10.18653\/v1\/2021.emnlp-main.799"},{"key":"32","doi-asserted-by":"crossref","unstructured":"Shterionov, D., Carmo, F. d., Moorkens, J., Hossari, M., Wagner, J., Paquin, E., Schmidtke, D., Groves, D., and Way, A. (2020). \u201cA Roadmap to Neural Automatic Post-editing: An Empirical Approach.\u201d <i>Machine Translation<\/i>, <b>34<\/b>, pp. 67\u201396.","DOI":"10.1007\/s10590-020-09249-7"},{"key":"33","unstructured":"Simard, M., Goutte, C., and Isabelle, P. (2007). \u201cStatistical Phrase-based Post-editing.\u201d In <i>Human Language Technologies 2007: The Conference of the North American Chapter of the Association for Computational Linguistics; Proceedings of the Main Conference<\/i>, pp. 508\u2013515."},{"key":"34","unstructured":"Snover, M., Dorr, B., Schwartz, R., Micciulla, L., and Makhoul, J. (2006). \u201cA Study of Translation Edit Rate with Targeted Human Annotation.\u201d In <i>Proceedings of the 7th Conference of the Association for Machine Translation in the Americas: Technical Papers<\/i>, pp. 223\u2013231."},{"key":"35","doi-asserted-by":"crossref","unstructured":"Specia, L., Raj, D., and Turchi, M. (2010). \u201cMachine Translation Evaluation Versus Quality Estimation.\u201d <i>Machine Translation<\/i>, <b>24<\/b>, pp. 39\u201350.","DOI":"10.1007\/s10590-010-9077-2"},{"key":"36","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., et al. (2023). \u201cLlama 2: Open Foundation and Fine-tuned Chat Models.\u201d <i>arXiv preprint arXiv:2307.09288<\/i>."},{"key":"37","doi-asserted-by":"crossref","unstructured":"Treviso, M. V., Guerreiro, N. M., Agrawal, S., Rei, R., Pombal, J., Vaz, T., Wu, H., Silva, B., Stigt, D. V., and Martins, A. (2024). \u201cxTower: A Multilingual LLM for Explaining and Correcting Translation Errors.\u201d In Al-Onaizan, Y., Bansal, M., and Chen, Y.-N. (Eds.), <i>Findings of the Association for Computational Linguistics: EMNLP 2024<\/i>, pp. 15222\u201315239, Miami, Florida, USA. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.findings-emnlp.892"},{"key":"38","unstructured":"Vidal, B., Llorens, A., and Alonso, J. (2022). \u201cAutomatic Post-editing of MT Output Using Large Language Models.\u201d In <i>Proceedings of the 15th Biennial Conference of the Association for Machine Translation in the Americas (Volume 2: Users and Providers Track and Government Track)<\/i>, pp. 84\u2013106."},{"key":"39","doi-asserted-by":"crossref","unstructured":"Wang, T., Liu, H., Liu, J., and Huang, D. (2022). \u201cDUTNLP Machine Translation System for WMT22 General MT Task.\u201d In <i>Proceedings of the 7th Conference on Machine Translation (WMT)<\/i>, pp. 397\u2013402.","DOI":"10.18653\/v1\/2022.wmt-1.35"},{"key":"40","unstructured":"Wu, I., Fernandes, P., Bertsch, A., Kim, S., Pakazad, S., and Neubig, G. (2024). \u201cBetter Instruction-following through Minimum Bayes Risk.\u201d <i>arXiv preprint arXiv:2410.02902<\/i>."},{"key":"41","doi-asserted-by":"crossref","unstructured":"Xu, W., Wang, D., Pan, L., Song, Z., Freitag, M., Wang, W. Y., and Li, L. (2023). \u201cINSTRUCTSCORE: Towards Explainable Text Generation Evaluation with Automatic Feedback.\u201d <i>arXiv preprint arXiv:2305.14282<\/i>.","DOI":"10.18653\/v1\/2023.emnlp-main.365"},{"key":"42","doi-asserted-by":"crossref","unstructured":"Zack, T., Lehman, E., Suzgun, M., Rodriguez, J. A., Celi, L. A., Gichoya, J., Jurafsky, D., Szolovits, P., Bates, D. W., Abdulnour, R.-E. E., Butte, A. J., and Alsentzer, E. (2024). \u201cAssessing the Potential of GPT-4 to Perpetuate Racial and Gender Biases in Health Care: A Model Evaluation Study.\u201d <i>The Lancet Digital Health<\/i>, <b>6<\/b> (1), pp. e12\u2013e22.","DOI":"10.1016\/S2589-7500(23)00225-X"}],"container-title":["Journal of Natural Language Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_848\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:44:10Z","timestamp":1781930650000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_848\/_article\/-char\/ja\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":42,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/doi.org\/10.5715\/jnlp.33.848","relation":{},"ISSN":["1340-7619","2185-8314"],"issn-type":[{"value":"1340-7619","type":"print"},{"value":"2185-8314","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}