{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T22:18:55Z","timestamp":1767824335244,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","funder":[{"name":"Postdoctoral Innovation Program of Shandong Province","award":["SDCX-ZG-202501019"],"award-info":[{"award-number":["SDCX-ZG-202501019"]}]},{"name":"China Postdoctoral Science Foundation under Grant","award":["No.2025M771502"],"award-info":[{"award-number":["No.2025M771502"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,21]]},"DOI":"10.1145\/3772429.3772436","type":"proceedings-article","created":{"date-parts":[[2025,12,23]],"date-time":"2025-12-23T13:59:08Z","timestamp":1766498348000},"page":"58-66","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SAC-B:Soft Actor-Critic with Bias for Suppressing Q-valueOverestimation in Off-Policy Reinforcement Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-3869-3941","authenticated-orcid":false,"given":"Han","family":"Wang","sequence":"first","affiliation":[{"name":"School of Software, Shandong University, Jinan, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5356-2427","authenticated-orcid":false,"given":"Wei","family":"Du","sequence":"additional","affiliation":[{"name":"School of Software, Shandong University, Jinan, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8926-7833","authenticated-orcid":false,"given":"Yanyu","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Software, Shandong University, Jinan,China, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4977-7577","authenticated-orcid":false,"given":"Lizhen","family":"Cui","sequence":"additional","affiliation":[{"name":"School of Software, Shandong University, Jinan, Shandong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,23]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Dario Amodei et\u00a0al. 2016. Concrete problems in AI safety. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1606.06565 (2016)."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.5555\/3305381.3305428"},{"key":"e_1_3_3_1_4_2","unstructured":"Xi Chen Chen Wang Zhe Zhou and Keith Ross. 2021. Randomized ensembled double Q-learning: Learning fast without a model. International Conference on Learning Representations (2021)."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"J. Duan et\u00a0al. 2025. Distributional soft actor-critic with three refinements. IEEE Transactions on Pattern Analysis and Machine Intelligence 47 5 (2025) 3935\u20133946.","DOI":"10.1109\/TPAMI.2025.3537087"},{"key":"e_1_3_3_1_6_2","first-page":"1667","volume-title":"Proc. 35th International Conference on Machine Learning (ICML)","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke van Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In Proc. 35th International Conference on Machine Learning (ICML). Stockholm, Sweden, 1667\u20131676."},{"key":"e_1_3_3_1_7_2","first-page":"1587","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Fujimoto Scott","year":"2018","unstructured":"Scott Fujimoto, Herke van Hoof, and David Meger. 2018. Addressing function approximation error in actor-critic methods. In Proc. Int. Conf. Mach. Learn. (ICML). Stockholm, Sweden, 1587\u20131596."},{"key":"e_1_3_3_1_8_2","first-page":"1352","volume-title":"Proc. International Conference on Machine Learning (ICML)","author":"Haarnoja Tuomas","year":"2017","unstructured":"Tuomas Haarnoja et\u00a0al. 2017. Reinforcement learning with deep energy-based policies. In Proc. International Conference on Machine Learning (ICML). Sydney, Australia, 1352\u20131361."},{"key":"e_1_3_3_1_9_2","first-page":"123","volume-title":"Proc. International Conference on Learning Representations (ICLR)","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja et\u00a0al. 2018. Distributed distributional deterministic policy gradients. In Proc. International Conference on Learning Representations (ICLR). Vancouver, Canada, 123\u2013132."},{"key":"e_1_3_3_1_10_2","first-page":"1861","volume-title":"Proceedings of the International Conference on Machine Learning (ICML)","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In Proceedings of the International Conference on Machine Learning (ICML). Stockholm, Sweden, 1861\u20131870."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"crossref","unstructured":"H. Lee and D. Lee. 2024. Suppressing overestimation in Q-learning through adversarial behaviors. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.06286 (2024).","DOI":"10.1109\/Allerton63246.2024.10735317"},{"key":"e_1_3_3_1_12_2","first-page":"1407","volume-title":"Proc. International Conference on Machine Learning (ICML)","author":"Levine Sergey","year":"2016","unstructured":"Sergey Levine et\u00a0al. 2016. Maximum a posteriori policy optimisation. In Proc. International Conference on Machine Learning (ICML). New York, NY, USA, 1407\u20131416."},{"key":"e_1_3_3_1_13_2","unstructured":"Sergey Levine Chelsea Finn Trevor Darrell and Pieter Abbeel. 2016. End-to-end training of deep visuomotor policies. Journal of Machine Learning Research 17 39 (2016) 1\u201328."},{"key":"e_1_3_3_1_14_2","first-page":"1874","volume-title":"Neural Information Processing Systems (NeurIPS)","author":"Lillicrap Timothy\u00a0P.","year":"2016","unstructured":"Timothy\u00a0P. Lillicrap et\u00a0al. 2016. Continuous control with deep reinforcement learning. In Neural Information Processing Systems (NeurIPS). Barcelona, Spain, 1874\u20131882."},{"key":"e_1_3_3_1_15_2","first-page":"231","volume-title":"Proc. Asia-Pacific Network Operations and Management Symposium (APNOMS)","author":"Liu J.","year":"2023","unstructured":"J. Liu, W. Fan, and M. Chen. 2023. Healthcare RL intrusion detection system with hybrid reinforcement learning. In Proc. Asia-Pacific Network Operations and Management Symposium (APNOMS). 231\u2013234."},{"key":"e_1_3_3_1_16_2","unstructured":"A. Ly R. Dazeley P. Vamplew F. Cruz and S. Aryal. 2022. Elastic step DQN: A novel multi-step algorithm to alleviate overestimation in Deep Q-Networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.03325 (2022)."},{"key":"e_1_3_3_1_17_2","first-page":"10781","volume-title":"Neural Information Processing Systems (NeurIPS)","author":"Nachum Ofir","year":"2018","unstructured":"Ofir Nachum et\u00a0al. 2018. Latent space policies for hierarchical reinforcement learning. In Neural Information Processing Systems (NeurIPS). Montreal, Canada, 10781\u201310792."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"David Silver et\u00a0al. 2018. A general reinforcement learning algorithm that masters chess shogi and go through self-play. Science 362 6419 (2018) 1140\u20131144.","DOI":"10.1126\/science.aar6404"},{"key":"e_1_3_3_1_19_2","first-page":"317","volume-title":"Proc. 10th International Conference on Machine Learning","author":"Thrun Sebastian","year":"1993","unstructured":"Sebastian Thrun and Anton Schwartz. 1993. Issues in using function approximation for reinforcement learning. In Proc. 10th International Conference on Machine Learning. Amherst, MA, USA, 317\u2013324."},{"key":"e_1_3_3_1_20_2","first-page":"2613","volume-title":"Advances in Neural Information Processing Systems","author":"Hasselt Hado van","year":"2010","unstructured":"Hado van Hasselt. 2010. Double Q-learning. In Advances in Neural Information Processing Systems. Vancouver, BC, Canada, 2613\u20132621."},{"key":"e_1_3_3_1_21_2","unstructured":"H. Wang S. Lin and J. Zhang. 2023. Adaptive ensemble Q-learning: Minimizing estimation bias via error feedback. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.11918 (2023)."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"F. Ye and W. Zhang. 2019. H2 input load disturbance rejection controller design for synchronised output regulation of time-delayed multi-agent systems with frequency domain method. Internat. J. Control 92 2 (2019) 356\u2013367.","DOI":"10.1080\/00207179.2017.1357838"},{"key":"e_1_3_3_1_23_2","first-page":"1587","volume-title":"Proc. IEEE International Conference on Intelligent Transportation Systems (ITSC)","author":"Zhang Y.","year":"2023","unstructured":"Y. Zhang et\u00a0al. 2023. 3DQN: Double dueling deep Q-network for adaptive traffic signal control. In Proc. IEEE International Conference on Intelligent Transportation Systems (ITSC). 1587\u20131596."}],"event":{"name":"DAI '25: The Seventh International Conference on Distributed Artificial Intelligence","location":"London United Kingdom","acronym":"DAI '25"},"container-title":["Proceedings of the 2025 The Seventh International Conference on Distributed Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3772429.3772436","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T19:42:40Z","timestamp":1767814960000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3772429.3772436"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,21]]},"references-count":22,"alternative-id":["10.1145\/3772429.3772436","10.1145\/3772429"],"URL":"https:\/\/doi.org\/10.1145\/3772429.3772436","relation":{},"subject":[],"published":{"date-parts":[[2025,11,21]]},"assertion":[{"value":"2025-12-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}