{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T15:39:08Z","timestamp":1781019548927,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,27]],"date-time":"2023-12-27T00:00:00Z","timestamp":1703635200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Social Science Major Foundation of China","award":["22&ZD035"],"award-info":[{"award-number":["22&ZD035"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,12,27]]},"DOI":"10.1145\/3639479.3639496","type":"proceedings-article","created":{"date-parts":[[2024,2,28]],"date-time":"2024-02-28T07:55:51Z","timestamp":1709106951000},"page":"84-90","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Reinforcement Learning in Natural Language Processing: A Survey"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-5384-316X","authenticated-orcid":false,"given":"Yingli","family":"Shen","sequence":"first","affiliation":[{"name":"School of Chinese Ethnic Minority Languages and Literatures, Minzu University of China, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1217-8650","authenticated-orcid":false,"given":"Xiaobing","family":"Zhao","sequence":"additional","affiliation":[{"name":"School of Information Engineering, Minzu University of China, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,2,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220122"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00496"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.5555\/1873738.1873746"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.83"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1064"},{"key":"e_1_3_2_1_7_1","volume-title":"Handbook of Markov decision processes: methods and applications. Vol.\u00a040","author":"Feinberg A","unstructured":"Eugene\u00a0A Feinberg and Adam Shwartz. 2012. Handbook of Markov decision processes: methods and applications. Vol.\u00a040. Springer Science & Business Media."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.450"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1153"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.216"},{"key":"e_1_3_2_1_11_1","volume-title":"10th International Conference on Learning Representations, ICLR 2022. International Conference on Learning Representations, ICLR.","author":"Jang Youngsoo","year":"2022","unstructured":"Youngsoo Jang, Jongmin Lee, and Kee-Eung Kim. 2022. GPT-critic: Offline reinforcement learning for end-to-end task-oriented dialogue systems. In 10th International Conference on Learning Representations, ICLR 2022. International Conference on Learning Representations, ICLR."},{"key":"e_1_3_2_1_12_1","volume-title":"Learned prioritization for trading off accuracy and speed. Advances in Neural Information Processing Systems 25","author":"Jiang Jiarong","year":"2012","unstructured":"Jiarong Jiang, Adam Teichert, Jason Eisner, and Hal Daume. 2012. Learned prioritization for trading off accuracy and speed. Advances in Neural Information Processing Systems 25 (2012)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.133"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-emnlp.315"},{"key":"e_1_3_2_1_15_1","volume-title":"Extending Multilingual Machine Translation through Imitation Learning. arXiv preprint arXiv:2311.08538","author":"Lai Wen","year":"2023","unstructured":"Wen Lai, Viktor Hangya, and Alexander Fraser. 2023. Extending Multilingual Machine Translation through Imitation Learning. arXiv preprint arXiv:2311.08538 (2023)."},{"key":"e_1_3_2_1_16_1","unstructured":"Wen Lai Jind\u0159ich Libovick\u00fd and Alexander Fraser. 2022. Improving Both Domain Robustness and Domain Adaptability in Machine Translation. In Proceedings of the 29th International Conference on Computational Linguistics Nicoletta Calzolari Chu-Ren Huang Hansaem Kim James Pustejovsky Leo Wanner Key-Sun Choi Pum-Mo Ryu Hsin-Hsi Chen Lucia Donatelli Heng Ji Sadao Kurohashi Patrizia Paggio Nianwen Xue Seokhwan Kim Younggyun Hahm Zhong He Tony\u00a0Kyungil Lee Enrico Santus Francis Bond and Seung-Hoon Na (Eds.). International Committee on Computational Linguistics Gyeongju Republic of Korea 5191\u20135204. https:\/\/aclanthology.org\/2022.coling-1.461"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/E17-1064"},{"key":"e_1_3_2_1_18_1","unstructured":"Ngan Le Vidhiwar\u00a0Singh Rathour Kashu Yamazaki Khoa Luu and Marios Savvides. [n.d.]. Deep reinforcement learning in computer vision: a comprehensive survey. Artificial Intelligence Review ([n. d.]) 1\u201387."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2108"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11946"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1042"},{"key":"e_1_3_2_1_22_1","volume-title":"Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602","author":"Mnih Volodymyr","year":"2013","unstructured":"Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Alex Graves, Ioannis Antonoglou, Daan Wierstra, and Martin Riedmiller. 2013. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)."},{"key":"e_1_3_2_1_23_1","volume-title":"Human-level control through deep reinforcement learning. nature 518, 7540","author":"Mnih Volodymyr","year":"2015","unstructured":"Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Andrei\u00a0A Rusu, Joel Veness, Marc\u00a0G Bellemare, Alex Graves, Martin Riedmiller, Andreas\u00a0K Fidjeland, Georg Ostrovski, 2015. Human-level control through deep reinforcement learning. nature 518, 7540 (2015), 529\u2013533."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1158"},{"key":"e_1_3_2_1_25_1","volume-title":"Reinforcement learning for bandit neural machine translation with simulated human feedback. arXiv preprint arXiv:1707.07402","author":"Nguyen Khanh","year":"2017","unstructured":"Khanh Nguyen, Hal Daum\u00e9\u00a0III, and Jordan Boyd-Graber. 2017. Reinforcement learning for bandit neural machine translation with simulated human feedback. arXiv preprint arXiv:1707.07402 (2017)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498515"},{"key":"e_1_3_2_1_27_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Ramamurthy Rajkumar","year":"2022","unstructured":"Rajkumar Ramamurthy, Prithviraj Ammanabrolu, Kiant\u00e9 Brantley, Jack Hessel, Rafet Sifa, Christian Bauckhage, Hannaneh Hajishirzi, and Yejin Choi. 2022. Is Reinforcement Learning (Not) for Natural Language Processing: Benchmarks, Baselines, and Building Blocks for Natural Language Policy Optimization. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/606"},{"key":"e_1_3_2_1_29_1","volume-title":"Julian Schrittwieser, Ioannis Antonoglou","author":"Silver David","year":"2016","unstructured":"David Silver, Aja Huang, Chris\u00a0J Maddison, Arthur Guez, Laurent Sifre, George Van Den\u00a0Driessche, Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, 2016. Mastering the game of Go with deep neural networks and tree search. nature 529, 7587 (2016), 484\u2013489."},{"key":"e_1_3_2_1_30_1","volume-title":"Mastering the game of go without human knowledge. nature 550, 7676","author":"Silver David","year":"2017","unstructured":"David Silver, Julian Schrittwieser, Karen Simonyan, Ioannis Antonoglou, Aja Huang, Arthur Guez, Thomas Hubert, Lucas Baker, Matthew Lai, Adrian Bolton, 2017. Mastering the game of go without human knowledge. nature 550, 7676 (2017), 354\u2013359."},{"key":"e_1_3_2_1_31_1","volume-title":"Reinforcement learning for spoken dialogue systems. Advances in neural information processing systems 12","author":"Singh Satinder","year":"1999","unstructured":"Satinder Singh, Michael Kearns, Diane Litman, and Marilyn Walker. 1999. Reinforcement learning for spoken dialogue systems. Advances in neural information processing systems 12 (1999)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.5555\/1858681.1858764"},{"key":"e_1_3_2_1_33_1","volume-title":"Neural Coreference Resolution based on Reinforcement Learning. arXiv preprint arXiv:2212.09028","author":"Wang Yu","year":"2022","unstructured":"Yu Wang and Hongxia Jin. 2022. Neural Coreference Resolution based on Reinforcement Learning. arXiv preprint arXiv:2212.09028 (2022)."},{"key":"e_1_3_2_1_34_1","volume-title":"Machine learning 8","author":"Watkins JCH","year":"1992","unstructured":"Christopher\u00a0JCH Watkins and Peter Dayan. 1992. Q-learning. Machine learning 8 (1992), 279\u2013292."},{"key":"e_1_3_2_1_35_1","volume-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning 8","author":"Williams J","year":"1992","unstructured":"Ronald\u00a0J Williams. 1992. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning 8 (1992), 229\u2013256."},{"key":"e_1_3_2_1_36_1","volume-title":"A study of reinforcement learning for neural machine translation. arXiv preprint arXiv:1808.08866","author":"Wu Lijun","year":"2018","unstructured":"Lijun Wu, Fei Tian, Tao Qin, Jianhuang Lai, and Tie-Yan Liu. 2018. A study of reinforcement learning for neural machine translation. arXiv preprint arXiv:1808.08866 (2018)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6471"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2018.01.020"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.272"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11950"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i5.20538"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12047"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3001684"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/DTPI52967.2021.9540098"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3336191.3371801"}],"event":{"name":"MLNLP 2023: 2023 6th International Conference on Machine Learning and Natural Language Processing","location":"Sanya China","acronym":"MLNLP 2023"},"container-title":["Proceedings of the 2023 6th International Conference on Machine Learning and Natural Language Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3639479.3639496","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3639479.3639496","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T23:53:04Z","timestamp":1755906784000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3639479.3639496"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,27]]},"references-count":45,"alternative-id":["10.1145\/3639479.3639496","10.1145\/3639479"],"URL":"https:\/\/doi.org\/10.1145\/3639479.3639496","relation":{},"subject":[],"published":{"date-parts":[[2023,12,27]]},"assertion":[{"value":"2024-02-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}