{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T20:32:45Z","timestamp":1767817965579,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62293541"],"award-info":[{"award-number":["62293541"]}]},{"name":"National Natural Science Foundation of China","award":["62136008"],"award-info":[{"award-number":["62136008"]}]},{"name":"National Natural Science Foundation of China","award":["62206281"],"award-info":[{"award-number":["62206281"]}]},{"name":"Beijing Natural Science Foundation","award":["4232056"],"award-info":[{"award-number":["4232056"]}]},{"name":"Beijing Nova Program","award":["20240484514"],"award-info":[{"award-number":["20240484514"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,21]]},"DOI":"10.1145\/3772429.3772496","type":"proceedings-article","created":{"date-parts":[[2025,12,23]],"date-time":"2025-12-23T13:59:08Z","timestamp":1766498348000},"page":"138-149","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["ARAC: Adaptive Regularized Multi-Agent Soft Actor-Critic in Graph-Structured Adversarial Games"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-7939-2445","authenticated-orcid":false,"given":"Ruochuan","family":"Shi","sequence":"first","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China and School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8084-3168","authenticated-orcid":false,"given":"Runyu","family":"Lu","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China and State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5384-423X","authenticated-orcid":false,"given":"Yuanheng","family":"Zhu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China and School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8218-9633","authenticated-orcid":false,"given":"Dongbin","family":"Zhao","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China and School of Artificial Intelligence, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,23]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Rishabh Agarwal Max Schwarzer Pablo\u00a0Samuel Castro Aaron\u00a0C Courville and Marc Bellemare. 2022. Reincarnating reinforcement learning: Reusing prior computation to accelerate progress. Advances in neural information processing systems 35 (2022) 28955\u201328971."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.5555\/3635637.3662876"},{"key":"e_1_3_3_1_4_2","unstructured":"Scott Fujimoto and Shixiang\u00a0Shane Gu. 2021. A minimalist approach to offline reinforcement learning. Advances in neural information processing systems 34 (2021) 20132\u201320145."},{"key":"e_1_3_3_1_5_2","first-page":"2052","volume-title":"International conference on machine learning","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-policy deep reinforcement learning without exploration. In International conference on machine learning. PMLR, 2052\u20132062."},{"key":"e_1_3_3_1_6_2","unstructured":"Chen-Xiao Gao Chenyang Wu Mingjun Cao Chenjun Xiao Yang Yu and Zongzhang Zhang. 2025. Behavior-regularized diffusion policy optimization for offline reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.04778 (2025)."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Eloy Garcia David\u00a0W Casbeer Alexander Von\u00a0Moll and Meir Pachter. 2020. Multiple pursuer multiple evader differential games. IEEE Trans. Automat. Control 66 5 (2020) 2345\u20132350.","DOI":"10.1109\/TAC.2020.3003840"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.5555\/3398761.3398819"},{"key":"e_1_3_3_1_9_2","first-page":"1861","volume-title":"International conference on machine learning","author":"Haarnoja Tuomas","year":"2018","unstructured":"Tuomas Haarnoja, Aurick Zhou, Pieter Abbeel, and Sergey Levine. 2018. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In International conference on machine learning. Pmlr, 1861\u20131870."},{"key":"e_1_3_3_1_10_2","unstructured":"Tuomas Haarnoja Aurick Zhou Kristian Hartikainen George Tucker Sehoon Ha Jie Tan Vikash Kumar Henry Zhu Abhishek Gupta Pieter Abbeel et\u00a0al. 2018. Soft actor-critic algorithms and applications. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1812.05905 (2018)."},{"key":"e_1_3_3_1_11_2","unstructured":"Joshua Hare. 2019. Dealing with sparse rewards in reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1910.09281 (2019)."},{"key":"e_1_3_3_1_12_2","unstructured":"Yifan Hu Junjie Fu and Guanghui Wen. 2023. Graph soft actor\u2013critic reinforcement learning for large-scale distributed multirobot coordination. IEEE transactions on neural networks and learning systems (2023)."},{"key":"e_1_3_3_1_13_2","first-page":"2961","volume-title":"International conference on machine learning","author":"Iqbal Shariq","year":"2019","unstructured":"Shariq Iqbal and Fei Sha. 2019. Actor-attention-critic for multi-agent reinforcement learning. In International conference on machine learning. PMLR, 2961\u20132970."},{"key":"e_1_3_3_1_14_2","volume-title":"8th International Conference on Learning Representations","author":"Jiang Jiechuan","year":"2020","unstructured":"Jiechuan Jiang, Chen Dun, Tiejun Huang, and Zongqing Lu. 2020. Graph Convolutional Reinforcement Learning. In 8th International Conference on Learning Representations."},{"key":"e_1_3_3_1_15_2","volume-title":"5th International Conference on Learning Representations","author":"Kipf Thomas\u00a0N.","year":"2017","unstructured":"Thomas\u00a0N. Kipf and Max Welling. 2017. Semi-Supervised Classification with Graph Convolutional Networks. In 5th International Conference on Learning Representations."},{"key":"e_1_3_3_1_16_2","unstructured":"Aviral Kumar Justin Fu Matthew Soh George Tucker and Sergey Levine. 2019. Stabilizing off-policy q-learning via bootstrapping error reduction. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.5555\/3635637.3662971"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Li Liang Fang Deng Zhihong Peng Xinxing Li and Wenzhong Zha. 2019. A differential game for cooperative target defense. Automatica 102 (2019) 58\u201371.","DOI":"10.1016\/j.automatica.2018.12.034"},{"key":"e_1_3_3_1_19_2","volume-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems","author":"Lu Runyu","year":"2025","unstructured":"Runyu Lu, Peng Zhang, Ruochuan Shi, Yuanheng Zhu, Dongbin Zhao, Yang Liu, Dong Wang, and Cesare Alippi. 2025. Equilibrium Policy Generalization: A Reinforcement Learning Framework for Cross-Graph Zero-Shot Generalization in Pursuit-Evasion Games. In The Thirty-ninth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Antonio Marino Esteban Restrepo Claudio Pacchierotti and Paolo\u00a0Robuffo Giordano. 2025. Decentralized Reinforcement Learning for Multi-Agent Multi-Resource Allocation via Dynamic Cluster Agreements. IEEE Robotics and Automation Letters (2025).","DOI":"10.1109\/LRA.2025.3581126"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Joshua McClellan Naveed Haghani John Winder Furong Huang and Pratap Tokekar. 2024. Boosting sample efficiency and generalization in multi-agent reinforcement learning via equivariance. Advances in Neural Information Processing Systems 37 (2024) 41132\u201341156.","DOI":"10.52202\/079017-1301"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Sai Munikoti Deepesh Agarwal Laya Das Mahantesh Halappanavar and Balasubramaniam Natarajan. 2023. Challenges and opportunities in deep reinforcement learning with graph neural networks: A comprehensive review of algorithms and applications. IEEE transactions on neural networks and learning systems 35 11 (2023) 15051\u201315071.","DOI":"10.1109\/TNNLS.2023.3283523"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"Dave\u00a0W Oyler Pierre\u00a0T Kabamba and Anouck\u00a0R Girard. 2016. Pursuit\u2013evasion games in the presence of obstacles. Automatica 65 (2016) 1\u201311.","DOI":"10.1016\/j.automatica.2015.11.018"},{"key":"e_1_3_3_1_24_2","first-page":"188","volume-title":"Conference on robot learning","author":"Pertsch Karl","year":"2021","unstructured":"Karl Pertsch, Youngwoon Lee, and Joseph Lim. 2021. Accelerating reinforcement learning with learned skill priors. In Conference on robot learning. PMLR, 188\u2013204."},{"key":"e_1_3_3_1_25_2","unstructured":"Yuan Pu Shaochen Wang Rui Yang Xin Yao and Bin Li. 2021. Decomposed soft actor-critic method for cooperative multi-agent reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2104.06655 (2021)."},{"key":"e_1_3_3_1_26_2","unstructured":"Tabish Rashid Mikayel Samvelyan Christian\u00a0Schroeder De\u00a0Witt Gregory Farquhar Jakob Foerster and Shimon Whiteson. 2020. Monotonic value function factorisation for deep multi-agent reinforcement learning. Journal of Machine Learning Research 21 178 (2020) 1\u201351."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.5555\/3237383.3238080"},{"key":"e_1_3_3_1_28_2","volume-title":"6th International Conference on Learning Representations","author":"Velickovic Petar","year":"2018","unstructured":"Petar Velickovic, Guillem Cucurull, Arantxa Casanova, Adriana Romero, Pietro Li\u00f2, and Yoshua Bengio. 2018. Graph Attention Networks. In 6th International Conference on Learning Representations."},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"crossref","unstructured":"Oriol Vinyals Igor Babuschkin Wojciech\u00a0M Czarnecki Micha\u00ebl Mathieu Andrew Dudzik Junyoung Chung David\u00a0H Choi Richard Powell Timo Ewalds Petko Georgiev et\u00a0al. 2019. Grandmaster level in StarCraft II using multi-agent reinforcement learning. nature 575 7782 (2019) 350\u2013354.","DOI":"10.1038\/s41586-019-1724-z"},{"key":"e_1_3_3_1_30_2","first-page":"2692","volume-title":"Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, December 7-12, 2015, Montreal, Quebec, Canada","author":"Vinyals Oriol","year":"2015","unstructured":"Oriol Vinyals, Meire Fortunato, and Navdeep Jaitly. 2015. Pointer Networks. In Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems 2015, December 7-12, 2015, Montreal, Quebec, Canada. 2692\u20132700."},{"key":"e_1_3_3_1_31_2","first-page":"1151","volume-title":"Machine Learning: Proceedings of the Seventeenth International Conference (ICML\u20192000)","author":"Wiering Marco\u00a0A","year":"2000","unstructured":"Marco\u00a0A Wiering et\u00a0al. 2000. Multi-agent reinforcement learning for traffic light control. In Machine Learning: Proceedings of the Seventeenth International Conference (ICML\u20192000). 1151\u20131158."},{"key":"e_1_3_3_1_32_2","unstructured":"Yifan Wu George Tucker and Ofir Nachum. 2019. Behavior regularized offline reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1911.11361 (2019)."},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"crossref","unstructured":"Maryam Zare Parham\u00a0M Kebria Abbas Khosravi and Saeid Nahavandi. 2024. A survey of imitation learning: Algorithms recent developments and challenges. IEEE Transactions on Cybernetics (2024).","DOI":"10.1109\/TCYB.2024.3395626"}],"event":{"name":"DAI '25: The Seventh International Conference on Distributed Artificial Intelligence","location":"London United Kingdom","acronym":"DAI '25"},"container-title":["Proceedings of the 2025 The Seventh International Conference on Distributed Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3772429.3772496","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,7]],"date-time":"2026-01-07T19:42:44Z","timestamp":1767814964000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3772429.3772496"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,21]]},"references-count":32,"alternative-id":["10.1145\/3772429.3772496","10.1145\/3772429"],"URL":"https:\/\/doi.org\/10.1145\/3772429.3772496","relation":{},"subject":[],"published":{"date-parts":[[2025,11,21]]},"assertion":[{"value":"2025-12-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}