{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:52:56Z","timestamp":1781931176821,"version":"3.54.5"},"reference-count":43,"publisher":"Association for Natural Language Processing","issue":"2","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Journal of Natural Language Processing"],"published-print":{"date-parts":[[2026]]},"DOI":"10.5715\/jnlp.33.452","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T22:11:46Z","timestamp":1781475106000},"page":"452-472","source":"Crossref","is-referenced-by-count":0,"title":["Emotion-aware Speech Translation Correction with Large Language Models"],"prefix":"10.5715","volume":"33","author":[{"given":"Zhengdong","family":"Yang","sequence":"first","affiliation":[{"name":"Graduate School of Informatics, Kyoto University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sheng","family":"Li","sequence":"additional","affiliation":[{"name":"School of Engineering, Institute of Science Tokyo"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenhui","family":"Chu","sequence":"additional","affiliation":[{"name":"Graduate School of Informatics, Kyoto University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"3685","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"Ackley, D. H., Hinton, G. E., and Sejnowski, T. J. (1985). \u201cA Learning Algorithm for Boltzmann Machines.\u201d <i>Cognitive Science<\/i>, <b>9<\/b> (1), pp. 147\u2013169.","DOI":"10.1016\/S0364-0213(85)80012-4"},{"key":"2","unstructured":"Barrault, L., Chung, Y.-A., Meglioli, M. C., Dale, D., Dong, N., Duquenne, P.-A., Elsahar, H., Gong, H., Heffernan, K., Hoffman, J., et al. (2023). \u201cSeamlessM4T: Massively Multilingual &amp; Multimodal Machine Translation.\u201d <i>arXiv preprint arXiv:2308.11596<\/i>."},{"key":"3","doi-asserted-by":"crossref","unstructured":"Chen, G., Chai, S., Wang, G.-B., Du, J., Zhang, W.-Q., Weng, C., Su, D., Povey, D., Trmal, J., Zhang, J., Jin, M., Khudanpur, S., Watanabe, S., Zhao, S., Zou, W., Li, X., Yao, X., Wang, Y., You, Z., and Yan, Z. (2021). \u201cGigaSpeech: An Evolving, Multi-Domain ASR Corpus with 10,000 Hours of Transcribed Audio.\u201d In <i>Proceedings of Interspeech 2021<\/i>.","DOI":"10.21437\/Interspeech.2021-1965"},{"key":"4","doi-asserted-by":"crossref","unstructured":"Chen, S., Yahata, S., Shimizu, S., Yang, Z., Li, Y., Chu, C., and Kurohashi, S. (2024). \u201cMELD-ST: An Emotion-aware Speech Translation Dataset.\u201d In Ku, L.-W., Martins, A., and Srikumar, V. (Eds.), <i>Findings of the Association for Computational Linguistics, ACL 2024, Bangkok, Thailand and Virtual Meeting, August 11\u201316, 2024<\/i>, pp. 10118\u201310126. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.findings-acl.601"},{"key":"5","unstructured":"Di Gangi, M. A., Cattoni, R., Bentivogli, L., Negri, M., and Turchi, M. (2019). \u201cMuST-C: a Multilingual Speech Translation Corpus.\u201d In <i>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)<\/i>, pp. 2012\u20132017."},{"key":"6","doi-asserted-by":"crossref","unstructured":"Fan, A., Lewis, M., and Dauphin, Y. N. (2018). \u201cHierarchical Neural Story Generation.\u201d In Gurevych, I., and Miyao, Y. (Eds.), <i>Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics, ACL 2018, Melbourne, Australia, July 15\u201320, 2018, Volume 1: Long Papers<\/i>, pp. 889\u2013898. Association for Computational Linguistics.","DOI":"10.18653\/v1\/P18-1082"},{"key":"7","doi-asserted-by":"crossref","unstructured":"Freitag, M., and Al-Onaizan, Y. (2017). \u201cBeam Search Strategies for Neural Machine Translation.\u201d In Luong, T., Birch, A., Neubig, G., and Finch, A. M. (Eds.), <i>Proceedings of the 1st Workshop on Neural Machine Translation, NMT@ACL 2017, Vancouver, Canada, August 4, 2017<\/i>, pp. 56\u201360. Association for Computational Linguistics.","DOI":"10.18653\/v1\/W17-3207"},{"key":"8","doi-asserted-by":"crossref","unstructured":"Fu, Y., Yuan, S., Zhang, C., and Cao, J. (2023). \u201cEmotion Recognition in Conversations: A Survey Focusing on Context, Speaker Dependencies, and Fusion Methods.\u201d <i>Electronics<\/i>, <b>12<\/b> (22), 4717.","DOI":"10.3390\/electronics12224714"},{"key":"9","unstructured":"Guo, D., Yang, D., Zhang, H., Song, J., Zhang, R., Xu, R., Zhu, Q., Ma, S., Wang, P., Bi, X., et al. (2025). \u201cDeepseek-r1: Incentivizing Reasoning Capability in Llms via Reinforcement Learning.\u201d <i>arXiv preprint arXiv:2501.12948<\/i>."},{"key":"10","doi-asserted-by":"crossref","unstructured":"Hewitt, J., Manning, C. D., and Liang, P. (2022). \u201cTruncation Sampling as Language Model Desmoothing.\u201d In Goldberg, Y., Kozareva, Z., and Zhang, Y. (Eds.), <i>Findings of the Association for Computational Linguistics: EMNLP 2022, Abu Dhabi, United Arab Emirates, December 7\u201311, 2022<\/i>, Vol. EMNLP 2022 of <i>Findings of ACL<\/i>, pp. 3414\u20133427. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2022.findings-emnlp.249"},{"key":"11","unstructured":"Holtzman, A., Buys, J., Du, L., Forbes, M., and Choi, Y. (2020). \u201cThe Curious Case of Neural Text Degeneration.\u201d In <i>8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26\u201330, 2020<\/i>. OpenReview.net."},{"key":"12","unstructured":"Hu, E. J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., and Chen, W. (2022). \u201cLora: Low-rank Adaptation of Large Language Models.\u201d <i>International Conference on Learning Representations<\/i>, <b>1<\/b> (2), p. 3."},{"key":"13","doi-asserted-by":"crossref","unstructured":"Hu, Y., Chen, C., Yang, C.-H. H., Li, R., Zhang, D., Chen, Z., and Chng, E. (2024). \u201cGenTranslate: Large Language Models are Generative Multilingual Speech and Machine Translators.\u201d In Ku, L.-W., Martins, A., and Srikumar, V. (Eds.), <i>Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), ACL 2024, Bangkok, Thailand, August 11\u201316, 2024<\/i>, pp. 74\u201390. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.acl-long.5"},{"key":"14","unstructured":"Jaech, A., Kalai, A., Lerer, A., Richardson, A., El-Kishky, A., Low, A., Helyar, A., Madry, A., Beutel, A., Carney, A., et al. (2024). \u201cOpenai o1 System Card.\u201d <i>arXiv preprint arXiv:2412.16720<\/i>."},{"key":"15","doi-asserted-by":"crossref","unstructured":"Jia, Y., Tadmor Ramanovich, M., Wang, Q., and Zen, H. (2022). \u201cCVSS Corpus and Massively Multilingual Speech-to-Speech Translation.\u201d In <i>Proceedings of Language Resources and Evaluation Conference (LREC)<\/i>, pp. 6691\u20136703.","DOI":"10.63317\/38skf3jhvobn"},{"key":"16","doi-asserted-by":"crossref","unstructured":"Jinnai, Y., Honda, U., Morimura, T., and Zhang, P. (2024). \u201cGenerating Diverse and High-Quality Texts by Minimum Bayes Risk Decoding.\u201d In Ku, L.-W., Martins, A., and Srikumar, V. (Eds.), <i>Findings of the Association for Computational Linguistics, ACL 2024, Bangkok, Thailand and Virtual Meeting, August 11\u201316, 2024<\/i>, Vol. ACL 2024 of <i>Findings of ACL<\/i>, pp. 8494\u20138525. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.findings-acl.503"},{"key":"17","unstructured":"Koehn, P. (2004). \u201cStatistical Significance Tests for Machine Translation Evaluation.\u201d In <i>Proceedings of the 2004 Conference on Empirical Methods in Natural Language Processing<\/i>, pp. 388\u2013395, Barcelona, Spain. Association for Computational Linguistics."},{"key":"18","unstructured":"Laine, S., and Aila, T. (2017). \u201cTemporal Ensembling for Semi-Supervised Learning.\u201d In <i>5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24\u201326, 2017, Conference Track Proceedings<\/i>. OpenReview.net."},{"key":"19","unstructured":"Li, J., and Jurafsky, D. (2016). \u201cMutual Information and Diverse Decoding Improve Neural Machine Translation.\u201d <i>arXiv preprint arXiv:1601.00372<\/i>."},{"key":"20","unstructured":"Li, J., Monroe, W., and Jurafsky, D. (2016). \u201cA Simple, Fast Diverse Decoding Algorithm for Neural Generation.\u201d <i>arXiv preprint arXiv:1611.08562<\/i>."},{"key":"21","doi-asserted-by":"crossref","unstructured":"Liang, Y., Meng, F., Chen, Y., Xu, J., and Zhou, J. (2021). \u201cModeling Bilingual Conversational Characteristics for Neural Chat Translation.\u201d In Zong, C., Xia, F., Li, W., and Navigli, R. (Eds.), <i>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)<\/i>, pp. 5711\u20135724, Online. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2021.acl-long.444"},{"key":"22","doi-asserted-by":"crossref","unstructured":"Lin, Y.-T., Chen, Z., \u017belasko, P., Wan, Z., Yang, X., Chen, Z.-C., Puvvada, K. C., Hu, K., Fu, S.-W., Chiu, J. W., Balam, J., Ginsburg, B., Wang, Y.-C. F., and Yang, C.-H. H. (2025). \u201cNeKo: Cross-Modality Post-Recognition Error Correction with Tasks-Guided Mixture-of-Experts Language Model.\u201d In <i>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 6: Industry Track)<\/i>, pp. 222\u2013236.","DOI":"10.18653\/v1\/2025.acl-industry.17"},{"key":"23","unstructured":"Loshchilov, I., and Hutter, F. (2019). \u201cDecoupled Weight Decay Regularization.\u201d In <i>7th International Conference on Learning Representations, ICLR 2019, New Orleans, LA, USA, May 6\u20139, 2019<\/i>. OpenReview.net."},{"key":"24","doi-asserted-by":"crossref","unstructured":"Meister, C., Forster, M., and Cotterell, R. (2021). \u201cDeterminantal Beam Search.\u201d In Zong, C., Xia, F., Li, W., and Navigli, R. (Eds.), <i>Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, ACL\/IJCNLP 2021, (Volume 1: Long Papers), Virtual Event, August 1\u20136, 2021<\/i>, pp. 6551\u20136562. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2021.acl-long.512"},{"key":"25","doi-asserted-by":"crossref","unstructured":"Poria, S., Hazarika, D., Majumder, N., Naik, G., Cambria, E., and Mihalcea, R. (2019). \u201cMELD: A Multimodal Multi-Party Dataset for Emotion Recognition in Conversations.\u201d In <i>Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics<\/i>, Vol. 12, pp. 527\u2013536.","DOI":"10.18653\/v1\/P19-1050"},{"key":"26","doi-asserted-by":"crossref","unstructured":"Post, M. (2018). \u201cA Call for Clarity in Reporting BLEU Scores.\u201d In Bojar, O., Chatterjee, R., Federmann, C., Fishel, M., Graham, Y., Haddow, B., Huck, M., Jimeno-Yepes, A., Koehn, P., Monz, C., Negri, M., N\u00e9v\u00e9ol, A., Neves, M. L., Post, M., Specia, L., Turchi, M., and Verspoor, K. (Eds.), <i>Proceedings of the 3rd Conference on Machine Translation: Research Papers, WMT 2018, Belgium, Brussels, October 31 \u2013 November 1, 2018<\/i>, pp. 186\u2013191. Association for Computational Linguistics.","DOI":"10.18653\/v1\/W18-6319"},{"key":"27","unstructured":"Radford, A., Kim, J. W., Xu, T., Brockman, G., McLeavey, C., and Sutskever, I. (2023). \u201cRobust Speech Recognition via Large-Scale Weak Supervision.\u201d In Krause, A., Brunskill, E., Cho, K., Engelhardt, B., Sabato, S., and Scarlett, J. (Eds.), <i>International Conference on Machine Learning, ICML 2023, 23\u201329 July 2023, Honolulu, Hawaii, USA<\/i>, Vol. 202 of <i>Proceedings of Machine Learning Research<\/i>, pp. 28492\u201328518. PMLR."},{"key":"28","doi-asserted-by":"crossref","unstructured":"Sellam, T., Das, D., and Parikh, A. (2020). \u201cBLEURT: Learning Robust Metrics for Text Generation.\u201d In <i>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics<\/i>, pp. 7881\u20137892.","DOI":"10.18653\/v1\/2020.acl-main.704"},{"key":"29","doi-asserted-by":"crossref","unstructured":"Sennrich, R., Haddow, B., and Birch-Mayne, A. (2016). \u201cControlling Politeness in Neural Machine Translation via Side Constraints.\u201d In <i>15th Annual Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies<\/i>, pp. 35\u201340. Association for Computational Linguistics.","DOI":"10.18653\/v1\/N16-1005"},{"key":"30","unstructured":"Sutskever, I., Vinyals, O., and Le, Q. V. (2014). \u201cSequence to Sequence Learning with Neural Networks.\u201d <i>Advances in Neural Information Processing Systems<\/i>, <b>27<\/b>."},{"key":"31","doi-asserted-by":"crossref","unstructured":"Troiano, E., Klinger, R., and Pad\u00f3, S. (2020a). \u201cLost in Back-Translation: Emotion Preservation in Neural Machine Translation.\u201d In <i>Proceedings of the 28th International Conference on Computational Linguistics<\/i>, pp. 4340\u20134354.","DOI":"10.18653\/v1\/2020.coling-main.384"},{"key":"32","doi-asserted-by":"crossref","unstructured":"Troiano, E., Klinger, R., and Pad\u00f3, S. (2020b). \u201cLost in Back-Translation: Emotion Preservation in Neural Machine Translation.\u201d In Scott, D., Bel, N., and Zong, C. (Eds.), <i>Proceedings of the 28th International Conference on Computational Linguistics, COLING 2020, Barcelona, Spain (Online), December 8\u201313, 2020<\/i>, pp. 4340\u20134354. International Committee on Computational Linguistics.","DOI":"10.18653\/v1\/2020.coling-main.384"},{"key":"33","doi-asserted-by":"crossref","unstructured":"Vijayakumar, A., Cogswell, M., Selvaraju, R., Sun, Q., Lee, S., Crandall, D., and Batra, D. (2018). \u201cDiverse Beam Search for Improved Description of Complex Scenes.\u201d <i>Proceedings of the AAAI Conference on Artificial Intelligence<\/i>, <b>32<\/b> (1), pp. 7371\u20137379.","DOI":"10.1609\/aaai.v32i1.12340"},{"key":"34","doi-asserted-by":"crossref","unstructured":"Wang, C., Wu, A., Gu, J., and Pino, J. (2021). \u201cCoVoST 2 and Massively Multilingual Speech Translation.\u201d In <i>Proceedings of Interspeech 2021<\/i>.","DOI":"10.21437\/Interspeech.2021-2027"},{"key":"35","doi-asserted-by":"crossref","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q. V., and Zhou, D. (2022). \u201cChain-of-thought Prompting Elicits Reasoning in Large Language Models.\u201d <i>Advances in Neural Information Processing Systems<\/i>, <b>35<\/b>, pp. 24824\u201324837.","DOI":"10.52202\/068431-1800"},{"key":"36","doi-asserted-by":"crossref","unstructured":"Wu, J., Gaur, Y., Chen, Z., Zhou, L., Zhu, Y., Wang, T., Li, J., Liu, S., Ren, B., Liu, L., et al. (2023). \u201cOn Decoder-only Architecture for Speech-to-text and Large Language Model Integration.\u201d In <i>2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)<\/i>, pp. 1\u20138. IEEE.","DOI":"10.1109\/ASRU57964.2023.10389705"},{"key":"37","unstructured":"Yang, A., Yang, B., Zhang, B., et al. (2024). \u201cQwen2.5 Technical Report.\u201d <i>arXiv preprint arXiv:2412.15115<\/i>. Alibaba Qwen Team."},{"key":"38","doi-asserted-by":"crossref","unstructured":"Yang, C.-H. H., Gu, Y., Liu, Y.-C., Ghosh, S., Bulyko, I., and Stolcke, A. (2023). \u201cGenerative Speech Recognition Error Correction with Large Language Models and Task-activating Prompting.\u201d In <i>2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)<\/i>, pp. 1\u20138. IEEE.","DOI":"10.1109\/ASRU57964.2023.10389673"},{"key":"39","doi-asserted-by":"crossref","unstructured":"Yang, Z., Li, S., and Chu, C. (2025a). \u201cGenerative Error Correction for Emotion-aware Speech-to-text Translation.\u201d In <i>Findings of the Association for Computational Linguistics: ACL 2025<\/i>, pp. 20413\u201320421.","DOI":"10.18653\/v1\/2025.findings-acl.1047"},{"key":"40","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wan, Z., Li, S., Yang, C.-H. H., and Chu, C. (2025b). \u201cCoVoGER: A Multilingual Multitask Benchmark for Speech-to-text Generative Error Correction with Large Language Models.\u201d In <i>Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing<\/i>, pp. 6313\u20136325.","DOI":"10.18653\/v1\/2025.emnlp-main.320"},{"key":"41","doi-asserted-by":"crossref","unstructured":"Ye, R., Zhao, C., Ko, T., Meng, C., Wang, T., Wang, M., and Cao, J. (2023). \u201cGigaST: A 10,000-hour Pseudo Speech Translation Corpus.\u201d In <i>Proceeding of Interspeech 2023<\/i>.","DOI":"10.21437\/Interspeech.2023-1233"},{"key":"42","unstructured":"Zhang, R., Han, J., Liu, C., Zhou, A., Lu, P., Qiao, Y., Li, H., and Gao, P. (2024). \u201cLLaMA-Adapter: Efficient Fine-tuning of Large Language Models with Zero-initialized Attention.\u201d In <i>The 12th International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7\u201311, 2024<\/i>. OpenReview.net."},{"key":"43","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Rudra, K., and Anand, A. (2021). \u201cExplain and Predict, and Then Predict Again.\u201d In <i>Proceedings of the 14th ACM International Conference on Web Search and Data Mining<\/i>, pp. 418\u2013426.","DOI":"10.1145\/3437963.3441758"}],"container-title":["Journal of Natural Language Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_452\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:45:12Z","timestamp":1781930712000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_452\/_article\/-char\/ja\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":43,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/doi.org\/10.5715\/jnlp.33.452","relation":{},"ISSN":["1340-7619","2185-8314"],"issn-type":[{"value":"1340-7619","type":"print"},{"value":"2185-8314","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}