{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T15:44:08Z","timestamp":1778255048151,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,3,3]],"date-time":"2023-03-03T00:00:00Z","timestamp":1677801600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,3,3]]},"DOI":"10.1145\/3594409.3594417","type":"proceedings-article","created":{"date-parts":[[2023,7,26]],"date-time":"2023-07-26T17:31:06Z","timestamp":1690392666000},"page":"190-195","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Double Actors and Uncertainty-Weighted Critics for Offline Reinforcement Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8866-8024","authenticated-orcid":false,"given":"Bohao","family":"Xie","sequence":"first","affiliation":[{"name":"Department of Electronic Engineering and Information Science, University of Science and Technology of China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2332-3959","authenticated-orcid":false,"given":"Bin","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering and Information Science, University of Science and Technology of China, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,7,26]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Machine Learning for Healthcare Conference (pp. 2-35)","author":"Tang S.","year":"2021","unstructured":"Tang , S. , & Wiens , J. ( 2021 , October). Model selection for offline reinforcement learning: Practical considerations for healthcare settings . In Machine Learning for Healthcare Conference (pp. 2-35) . PMLR. Tang, S., & Wiens, J. (2021, October). Model selection for offline reinforcement learning: Practical considerations for healthcare settings. In Machine Learning for Healthcare Conference (pp. 2-35). PMLR."},{"key":"e_1_3_2_1_2_1","volume-title":"Conference on Robot Learning (pp. 651-673)","author":"Kalashnikov D.","year":"2018","unstructured":"Kalashnikov , D. , Irpan , A. , Pastor , P. , Ibarz , J. , Herzog , A. , Jang , E. , ... & Levine , S. ( 2018 , October). Scalable deep reinforcement learning for vision-based robotic manipulation . In Conference on Robot Learning (pp. 651-673) . PMLR. Kalashnikov, D., Irpan, A., Pastor, P., Ibarz, J., Herzog, A., Jang, E., ... & Levine, S. (2018, October). Scalable deep reinforcement learning for vision-based robotic manipulation. In Conference on Robot Learning (pp. 651-673). PMLR."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/1772690.1772758"},{"key":"e_1_3_2_1_4_1","volume-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643","author":"Levine S.","year":"2020","unstructured":"Levine , S. , Kumar , A. , Tucker , G. , & Fu , J. ( 2020 ). Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643 . Levine, S., Kumar, A., Tucker, G., & Fu, J. (2020). Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643."},{"key":"e_1_3_2_1_5_1","volume-title":"Behavioral cloning from observation. arXiv preprint arXiv:1805.01954","author":"Torabi F.","year":"2018","unstructured":"Torabi , F. , Warnell , G. , & Stone , P. ( 2018 ). Behavioral cloning from observation. arXiv preprint arXiv:1805.01954 . Torabi, F., Warnell, G., & Stone, P. (2018). Behavioral cloning from observation. arXiv preprint arXiv:1805.01954."},{"key":"e_1_3_2_1_6_1","volume-title":"Behavior regularized offline reinforcement learning. arXiv preprint arXiv:1911.11361","author":"Wu Y.","year":"2019","unstructured":"Wu , Y. , Tucker , G. , & Nachum , O. ( 2019 ). Behavior regularized offline reinforcement learning. arXiv preprint arXiv:1911.11361 . Wu, Y., Tucker, G., & Nachum, O. (2019). Behavior regularized offline reinforcement learning. arXiv preprint arXiv:1911.11361."},{"key":"e_1_3_2_1_7_1","volume-title":"A minimalist approach to offline reinforcement learning. Advances in neural information processing systems, 34","author":"Fujimoto S.","year":"2021","unstructured":"Fujimoto , S. , & Gu , S. S. ( 2021 ). A minimalist approach to offline reinforcement learning. Advances in neural information processing systems, 34 , 20132-20145. Fujimoto, S., & Gu, S. S. (2021). A minimalist approach to offline reinforcement learning. Advances in neural information processing systems, 34, 20132-20145."},{"key":"e_1_3_2_1_8_1","volume-title":"Uncertainty weighted actor-critic for offline reinforcement learning. arXiv preprint arXiv:2105.08140","author":"Wu Y.","year":"2021","unstructured":"Wu , Y. , Zhai , S. , Srivastava , N. , Susskind , J. , Zhang , J. , Salakhutdinov , R. , & Goh , H. ( 2021 ). Uncertainty weighted actor-critic for offline reinforcement learning. arXiv preprint arXiv:2105.08140 . Wu, Y., Zhai, S., Srivastava, N., Susskind, J., Zhang, J., Salakhutdinov, R., & Goh, H. (2021). Uncertainty weighted actor-critic for offline reinforcement learning. arXiv preprint arXiv:2105.08140."},{"key":"e_1_3_2_1_9_1","volume-title":"International Conference on Machine Learning (pp. 5774-5783)","author":"Kostrikov I.","year":"2021","unstructured":"Kostrikov , I. , Fergus , R. , Tompson , J. , & Nachum , O. ( 2021 , July). Offline reinforcement learning with fisher divergence critic regularization . In International Conference on Machine Learning (pp. 5774-5783) . PMLR. Kostrikov, I., Fergus, R., Tompson, J., & Nachum, O. (2021, July). Offline reinforcement learning with fisher divergence critic regularization. In International Conference on Machine Learning (pp. 5774-5783). PMLR."},{"key":"e_1_3_2_1_10_1","first-page":"1179","article-title":"Conservative q-learning for offline reinforcement learning","volume":"33","author":"Kumar A.","year":"2020","unstructured":"Kumar , A. , Zhou , A. , Tucker , G. , & Levine , S. ( 2020 ). Conservative q-learning for offline reinforcement learning . Advances in Neural Information Processing Systems , 33 , 1179 - 1191 . Kumar, A., Zhou, A., Tucker, G., & Levine, S. (2020). Conservative q-learning for offline reinforcement learning. Advances in Neural Information Processing Systems, 33, 1179-1191.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_11_1","first-page":"18353","article-title":"BAIL: Best-action imitation learning for batch deep reinforcement learning","volume":"33","author":"Chen X.","year":"2020","unstructured":"Chen , X. , Zhou , Z. , Wang , Z. , Wang , C. , Wu , Y. , & Ross , K. ( 2020 ). BAIL: Best-action imitation learning for batch deep reinforcement learning . Advances in Neural Information Processing Systems , 33 , 18353 - 18363 . Chen, X., Zhou, Z., Wang, Z., Wang, C., Wu, Y., & Ross, K. (2020). BAIL: Best-action imitation learning for batch deep reinforcement learning. Advances in Neural Information Processing Systems, 33, 18353-18363.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_12_1","volume-title":"Awac: Accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359","author":"Nair A.","year":"2020","unstructured":"Nair , A. , Gupta , A. , Dalal , M. , & Levine , S. ( 2020 ). Awac: Accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359 . Nair, A., Gupta, A., Dalal, M., & Levine, S. (2020). Awac: Accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359."},{"key":"e_1_3_2_1_13_1","volume-title":"Offline reinforcement learning with implicit q-learning. arXiv preprint arXiv:2110.06169","author":"Kostrikov I.","year":"2021","unstructured":"Kostrikov , I. , Nair , A. , & Levine , S. ( 2021 ). Offline reinforcement learning with implicit q-learning. arXiv preprint arXiv:2110.06169 . Kostrikov, I., Nair, A., & Levine, S. (2021). Offline reinforcement learning with implicit q-learning. arXiv preprint arXiv:2110.06169."},{"key":"e_1_3_2_1_14_1","first-page":"4933","article-title":"Offline rl without off-policy evaluation","volume":"34","author":"Brandfonbrener D.","year":"2021","unstructured":"Brandfonbrener , D. , Whitney , W. , Ranganath , R. , & Bruna , J. ( 2021 ). Offline rl without off-policy evaluation . Advances in Neural Information Processing Systems , 34 , 4933 - 4946 . Brandfonbrener, D., Whitney, W., Ranganath, R., & Bruna, J. (2021). Offline rl without off-policy evaluation. Advances in Neural Information Processing Systems, 34, 4933-4946.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_15_1","volume-title":"International Conference on Machine Learning (pp. 104-114)","author":"Agarwal R.","year":"2020","unstructured":"Agarwal , R. , Schuurmans , D. , & Norouzi , M. ( 2020 , November). An optimistic perspective on offline reinforcement learning . In International Conference on Machine Learning (pp. 104-114) . PMLR. Agarwal, R., Schuurmans, D., & Norouzi, M. (2020, November). An optimistic perspective on offline reinforcement learning. In International Conference on Machine Learning (pp. 104-114). PMLR."},{"key":"e_1_3_2_1_16_1","first-page":"7655","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 36","author":"Lyu J.","year":"2022","unstructured":"Lyu , J. , Ma , X. , Yan , J. , & Li , X. ( 2022 , June). Efficient continuous control with double actors and regularized critics . In Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 36 , No. 7, pp. 7655 - 7663 ). Lyu, J., Ma, X., Yan, J., & Li, X. (2022, June). Efficient continuous control with double actors and regularized critics. In Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 36, No. 7, pp. 7655-7663)."},{"key":"e_1_3_2_1_17_1","volume-title":"D4rl: Datasets for deep data-driven reinforcement learning. arXiv preprint arXiv:2004.07219","author":"Fu J.","year":"2020","unstructured":"Fu , J. , Kumar , A. , Nachum , O. , Tucker , G. , & Levine , S. ( 2020 ). D4rl: Datasets for deep data-driven reinforcement learning. arXiv preprint arXiv:2004.07219 . Fu, J., Kumar, A., Nachum, O., Tucker, G., & Levine, S. (2020). D4rl: Datasets for deep data-driven reinforcement learning. arXiv preprint arXiv:2004.07219."},{"key":"e_1_3_2_1_18_1","volume-title":"Combo: Conservative offline model-based policy optimization. Advances in neural information processing systems, 34, 28954-28967","author":"Yu T.","year":"2021","unstructured":"Yu , T. , Kumar , A. , Rafailov , R. , Rajeswaran , A. , Levine , S. , & Finn , C. ( 2021 ). Combo: Conservative offline model-based policy optimization. Advances in neural information processing systems, 34, 28954-28967 . Yu, T., Kumar, A., Rafailov, R., Rajeswaran, A., Levine, S., & Finn, C. (2021). Combo: Conservative offline model-based policy optimization. Advances in neural information processing systems, 34, 28954-28967."},{"key":"e_1_3_2_1_19_1","volume-title":"International conference on machine learning (pp. 1928-1937)","author":"Mnih V.","year":"2016","unstructured":"Mnih , V. , Badia , A. P. , Mirza , M. , Graves , A. , Lillicrap , T. , Harley , T. , ... & Kavukcuoglu , K. ( 2016 , June). Asynchronous methods for deep reinforcement learning . In International conference on machine learning (pp. 1928-1937) . PMLR. Mnih, V., Badia, A. P., Mirza, M., Graves, A., Lillicrap, T., Harley, T., ... & Kavukcuoglu, K. (2016, June). Asynchronous methods for deep reinforcement learning. In International conference on machine learning (pp. 1928-1937). PMLR."},{"key":"e_1_3_2_1_20_1","volume-title":"international conference on machine learning (pp. 1050-1059)","author":"Gal Y.","year":"2016","unstructured":"Gal , Y. , & Ghahramani , Z. ( 2016 , June). Dropout as a bayesian approximation: Representing model uncertainty in deep learning . In international conference on machine learning (pp. 1050-1059) . PMLR. Gal, Y., & Ghahramani, Z. (2016, June). Dropout as a bayesian approximation: Representing model uncertainty in deep learning. In international conference on machine learning (pp. 1050-1059). PMLR."},{"key":"e_1_3_2_1_21_1","first-page":"14129","article-title":"Mopo: Model-based offline policy optimization","volume":"33","author":"Yu T.","year":"2020","unstructured":"Yu , T. , Thomas , G. , Yu , L. , Ermon , S. , Zou , J. Y. , Levine , S. , ... & Ma , T. ( 2020 ). Mopo: Model-based offline policy optimization . Advances in Neural Information Processing Systems , 33 , 14129 - 14142 . Yu, T., Thomas, G., Yu, L., Ermon, S., Zou, J. Y., Levine, S., ... & Ma, T. (2020). Mopo: Model-based offline policy optimization. Advances in Neural Information Processing Systems, 33, 14129-14142.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_22_1","volume-title":"Morel: Model-based offline reinforcement learning. Advances in neural information processing systems, 33, 21810-21823","author":"Kidambi R.","year":"2020","unstructured":"Kidambi , R. , Rajeswaran , A. , Netrapalli , P. , & Joachims , T. ( 2020 ). Morel: Model-based offline reinforcement learning. Advances in neural information processing systems, 33, 21810-21823 . Kidambi, R., Rajeswaran, A., Netrapalli, P., & Joachims, T. (2020). Morel: Model-based offline reinforcement learning. Advances in neural information processing systems, 33, 21810-21823."}],"event":{"name":"ICIAI 2023: 2023 the 7th International Conference on Innovation in Artificial Intelligence","location":"Harbin China","acronym":"ICIAI 2023"},"container-title":["Proceedings of the 2023 7th International Conference on Innovation in Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3594409.3594417","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3594409.3594417","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:51:17Z","timestamp":1750182677000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3594409.3594417"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,3]]},"references-count":22,"alternative-id":["10.1145\/3594409.3594417","10.1145\/3594409"],"URL":"https:\/\/doi.org\/10.1145\/3594409.3594417","relation":{},"subject":[],"published":{"date-parts":[[2023,3,3]]},"assertion":[{"value":"2023-07-26","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}