{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T22:17:48Z","timestamp":1783981068405,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,24]],"date-time":"2024-08-24T00:00:00Z","timestamp":1724457600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022ZD0116402"],"award-info":[{"award-number":["2022ZD0116402"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U22B2057, U21B2036"],"award-info":[{"award-number":["U22B2057, U21B2036"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,25]]},"DOI":"10.1145\/3637528.3672052","type":"proceedings-article","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T04:54:55Z","timestamp":1724561695000},"page":"3128-3139","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["DyPS: Dynamic Parameter Sharing in Multi-Agent Reinforcement Learning for Spatio-Temporal Resource Allocation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-9175-6255","authenticated-orcid":false,"given":"Jingwei","family":"Wang","sequence":"first","affiliation":[{"name":"Department of EE, BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7109-3588","authenticated-orcid":false,"given":"Qianyue","family":"Hao","sequence":"additional","affiliation":[{"name":"Department of EE, BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0454-7516","authenticated-orcid":false,"given":"Wenzhen","family":"Huang","sequence":"additional","affiliation":[{"name":"Department of EE, BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8945-3046","authenticated-orcid":false,"given":"Xiaochen","family":"Fan","sequence":"additional","affiliation":[{"name":"Department of EE, BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3472-1460","authenticated-orcid":false,"given":"Zhentao","family":"Tang","sequence":"additional","affiliation":[{"name":"Huawei Noah's Ark Lab, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0267-3749","authenticated-orcid":false,"given":"Bin","family":"Wang","sequence":"additional","affiliation":[{"name":"Huawei Noah's Ark Lab, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0422-8235","authenticated-orcid":false,"given":"Jianye","family":"Hao","sequence":"additional","affiliation":[{"name":"Tianjin University &amp; Huawei Noah's Ark Lab, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5617-1659","authenticated-orcid":false,"given":"Yong","family":"Li","sequence":"additional","affiliation":[{"name":"Department of EE, BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,8,24]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"A game-theoretic model and best-response learning method for ad hoc coordination in multiagent systems. arXiv preprint arXiv:1506.01170","author":"Albrecht Stefano V","year":"2015","unstructured":"Stefano V Albrecht and Subramanian Ramamoorthy. 2015. A game-theoretic model and best-response learning method for ad hoc coordination in multiagent systems. arXiv preprint arXiv:1506.01170 (2015)."},{"key":"e_1_3_2_1_2_1","volume-title":"Reasoning about hypothetical agent behaviours and their parameters. arXiv preprint arXiv:1906.11064","author":"Albrecht Stefano V","year":"2019","unstructured":"Stefano V Albrecht and Peter Stone. 2019. Reasoning about hypothetical agent behaviours and their parameters. arXiv preprint arXiv:1906.11064 (2019)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3161602"},{"key":"e_1_3_2_1_4_1","volume-title":"Reinforcement learning through asynchronous advantage actor-critic on a gpu. arXiv preprint arXiv:1611.06256","author":"Babaeizadeh Mohammad","year":"2016","unstructured":"Mohammad Babaeizadeh, Iuri Frosio, Stephen Tyree, Jason Clemons, and Jan Kautz. 2016. Reinforcement learning through asynchronous advantage actor-critic on a gpu. arXiv preprint arXiv:1611.06256 (2016)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41562-022-01429-0"},{"key":"e_1_3_2_1_6_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Christianos Filippos","year":"2021","unstructured":"Filippos Christianos, Georgios Papoudakis, Muhammad A Rahman, and Stefano V Albrecht. 2021. Scaling multi-agent reinforcement learning with selective parameter sharing. In International Conference on Machine Learning. PMLR, 1989--1998."},{"key":"e_1_3_2_1_7_1","volume-title":"Parameter sharing deep deterministic policy gradient for cooperative multi-agent reinforcement learning. arXiv preprint arXiv:1710.00336","author":"Chu Xiangxiang","year":"2017","unstructured":"Xiangxiang Chu and Hangjun Ye. 2017. Parameter sharing deep deterministic policy gradient for cooperative multi-agent reinforcement learning. arXiv preprint arXiv:1710.00336 (2017)."},{"key":"e_1_3_2_1_8_1","volume-title":"A recurrent latent variable model for sequential data. arXiv preprint arXiv:1506.02216","author":"Chung Junyoung","year":"2015","unstructured":"Junyoung Chung, Kyle Kastner, Laurent Dinh, Kratarth Goel, Aaron Courville, and Yoshua Bengio. 2015. A recurrent latent variable model for sequential data. arXiv preprint arXiv:1506.02216 (2015)."},{"key":"e_1_3_2_1_9_1","volume-title":"Mingfei Sun, and Shimon Whiteson.","author":"de Witt Christian Schroeder","year":"2020","unstructured":"Christian Schroeder de Witt, Tarun Gupta, Denys Makoviichuk, Viktor Makoviychuk, Philip HS Torr, Mingfei Sun, and Shimon Whiteson. 2020. Is independent learning all you need in the starcraft multi-agent challenge? arXiv preprint arXiv:2011.09533 (2020)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"e_1_3_2_1_11_1","volume-title":"Contextual Spatio-Temporal Graph Representation Learning for Reinforced Human Mobility Mining. Information Sciences","author":"Gao Qiang","year":"2022","unstructured":"Qiang Gao, Fan Zhou, Ting Zhong, Goce Trajcevski, Xin Yang, and Tianrui Li. 2022. Contextual Spatio-Temporal Graph Representation Learning for Reinforced Human Mobility Mining. Information Sciences (2022)."},{"key":"e_1_3_2_1_12_1","volume-title":"International conference on machine learning. PMLR","author":"Grover Aditya","year":"2018","unstructured":"Aditya Grover, Maruan Al-Shedivat, Jayesh Gupta, Yuri Burda, and Harrison Edwards. 2018. Learning policy representations in multiagent systems. In International conference on machine learning. PMLR, 1802--1811."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-71682-4_5"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-09994-y"},{"key":"e_1_3_2_1_15_1","volume-title":"Variational recurrent models for solving partially observable control tasks. arXiv preprint arXiv:1912.10703","author":"Han Dongqi","year":"2019","unstructured":"Dongqi Han, Kenji Doya, and Jun Tani. 2019. Variational recurrent models for solving partially observable control tasks. arXiv preprint arXiv:1912.10703 (2019)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599359"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3542679"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467181"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2023.3302014"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2020.106302"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3357978"},{"key":"e_1_3_2_1_22_1","unstructured":"Meha Kaushik K Madhava Krishna et al. 2018. Parameter sharing reinforcement learning architecture for multi agent driving behaviors. arXiv preprint arXiv:1811.07214 (2018)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1561\/9781680836233"},{"key":"e_1_3_2_1_24_1","unstructured":"Vijay R Konda and John N Tsitsiklis. 2000. Actor-critic algorithms. In Advances in neural information processing systems. 1008--1014."},{"key":"e_1_3_2_1_25_1","volume-title":"Parameter Sharing Reinforcement Learning for Modeling Multi-Agent Driving Behavior in Roundabout Scenarios. In 2021 IEEE International Intelligent Transportation Systems Conference (ITSC). IEEE","author":"Konstantinidis Fabian","year":"2021","unstructured":"Fabian Konstantinidis, Ulrich Hofmann, Moritz Sackmann, J\u00f6rn Thielecke, Oliver De Candido, and Wolfgang Utschick. 2021. Parameter Sharing Reinforcement Learning for Modeling Multi-Agent Driving Behavior in Roundabout Scenarios. In 2021 IEEE International Intelligent Transportation Systems Conference (ITSC). IEEE, 1974--1981."},{"key":"e_1_3_2_1_26_1","first-page":"3991","article-title":"Celebrating diversity in shared multi-agent reinforcement learning","volume":"34","author":"Li Chenghao","year":"2021","unstructured":"Chenghao Li, Tonghan Wang, Chengjie Wu, Qianchuan Zhao, Jun Yang, and Chongjie Zhang. 2021. Celebrating diversity in shared multi-agent reinforcement learning. Advances in Neural Information Processing Systems, Vol. 34 (2021), 3991--4002.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSYST.2014.2364252"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Minne Li Zhiwei Qin Yan Jiao Yaodong Yang Jun Wang Chenxi Wang Guobin Wu and Jieping Ye. 2019. Efficient ridesharing order dispatching with mean field multi-agent reinforcement learning. In The world wide web conference. 983--994.","DOI":"10.1145\/3308558.3313433"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220110"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3411871"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.5555\/3091574.3091594"},{"key":"e_1_3_2_1_32_1","volume-title":"Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602","author":"Mnih Volodymyr","year":"2013","unstructured":"Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Alex Graves, Ioannis Antonoglou, Daan Wierstra, and Martin Riedmiller. 2013. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2012.2229978"},{"key":"e_1_3_2_1_34_1","first-page":"10784","article-title":"Learning to simulate self-driven particles system with coordinated policy optimization","volume":"34","author":"Peng Zhenghao","year":"2021","unstructured":"Zhenghao Peng, Quanyi Li, Ka Ming Hui, Chunxiao Liu, and Bolei Zhou. 2021. Learning to simulate self-driven particles system with coordinated policy optimization. Advances in Neural Information Processing Systems, Vol. 34 (2021), 10784--10797.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","volume-title":"International conference on machine learning. PMLR, 4295--4304","author":"Rashid Tabish","year":"2018","unstructured":"Tabish Rashid, Mikayel Samvelyan, Christian Schroeder, Gregory Farquhar, Jakob Foerster, and Shimon Whiteson. 2018. Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning. In International conference on machine learning. PMLR, 4295--4304."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3380986"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.3390\/ijgi4042306"},{"key":"e_1_3_2_1_38_1","volume-title":"Learning structured output representation using deep conditional generative models. Advances in neural information processing systems","author":"Sohn Kihyuk","year":"2015","unstructured":"Kihyuk Sohn, Honglak Lee, and Xinchen Yan. 2015. Learning structured output representation using deep conditional generative models. Advances in neural information processing systems, Vol. 28 (2015)."},{"key":"e_1_3_2_1_39_1","volume-title":"NIPs","volume":"99","author":"Sutton Richard S","year":"1999","unstructured":"Richard S Sutton, David A McAllester, Satinder P Singh, Yishay Mansour, et al. 1999. Policy gradient methods for reinforcement learning with function approximation.. In NIPs, Vol. 99. Citeseer, 1057--1063."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467096"},{"key":"e_1_3_2_1_41_1","volume-title":"Rode: Learning roles to decompose multi-agent tasks. arXiv preprint arXiv:2010.01523","author":"Wang Tonghan","year":"2020","unstructured":"Tonghan Wang, Tarun Gupta, Anuj Mahajan, Bei Peng, Shimon Whiteson, and Chongjie Zhang. 2020. Rode: Learning roles to decompose multi-agent tasks. arXiv preprint arXiv:2010.01523 (2020)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jclepro.2020.122537"},{"key":"e_1_3_2_1_43_1","volume-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning","author":"Williams Ronald J","year":"1992","unstructured":"Ronald J Williams. 1992. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning, Vol. 8, 3--4 (1992), 229--256."},{"key":"e_1_3_2_1_44_1","volume-title":"Coordinating hundreds of cooperative, autonomous vehicles in warehouses. AI magazine","author":"Wurman Peter R","year":"2008","unstructured":"Peter R Wurman, Raffaello D'Andrea, and Mick Mountz. 2008. Coordinating hundreds of cooperative, autonomous vehicles in warehouses. AI magazine, Vol. 29, 1 (2008), 9--9."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/JCC56315.2022.00013"},{"key":"e_1_3_2_1_46_1","volume-title":"International conference on machine learning. PMLR, 5571--5580","author":"Yang Yaodong","year":"2018","unstructured":"Yaodong Yang, Rui Luo, Minne Li, Ming Zhou, Weinan Zhang, and Jun Wang. 2018. Mean field multi-agent reinforcement learning. In International conference on machine learning. PMLR, 5571--5580."},{"key":"e_1_3_2_1_47_1","first-page":"24611","article-title":"The surprising effectiveness of ppo in cooperative multi-agent games","volume":"35","author":"Yu Chao","year":"2022","unstructured":"Chao Yu, Akash Velu, Eugene Vinitsky, Jiaxuan Gao, Yu Wang, Alexandre Bayen, and Yi Wu. 2022. The surprising effectiveness of ppo in cooperative multi-agent games. Advances in Neural Information Processing Systems, Vol. 35 (2022), 24611--24624.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_48_1","volume-title":"The surprising effectiveness of ppo in cooperative, multi-agent games. arXiv preprint arXiv:2103.01955","author":"Yu Chao","year":"2021","unstructured":"Chao Yu, Akash Velu, Eugene Vinitsky, Yu Wang, Alexandre Bayen, and Yi Wu. 2021. The surprising effectiveness of ppo in cooperative, multi-agent games. arXiv preprint arXiv:2103.01955 (2021)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2016.09.002"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3467979","article-title":"Supply-demand-aware deep reinforcement learning for dynamic fleet management","volume":"13","author":"Zheng Bolong","year":"2022","unstructured":"Bolong Zheng, Lingfeng Ming, Qi Hu, Zhipeng L\u00fc, Guanfeng Liu, and Xiaofang Zhou. 2022. Supply-demand-aware deep reinforcement learning for dynamic fleet management. ACM Transactions on Intelligent Systems and Technology (TIST), Vol. 13, 3 (2022), 1--19.","journal-title":"ACM Transactions on Intelligent Systems and Technology (TIST)"},{"key":"e_1_3_2_1_51_1","volume-title":"CIRED 2009--20th International Conference and Exhibition on Electricity Distribution-Part 1. IET, 1--4.","author":"Zhou Limei","year":"2009","unstructured":"Limei Zhou, Mingtian Fan, and Zuping Zhang. 2009. A study on the optimal allocation of emergency power supplies in urban electric network. In CIRED 2009--20th International Conference and Exhibition on Electricity Distribution-Part 1. IET, 1--4."}],"event":{"name":"KDD '24: The 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Barcelona Spain","acronym":"KDD '24","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3672052","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3637528.3672052","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:04:23Z","timestamp":1750291463000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3672052"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,24]]},"references-count":51,"alternative-id":["10.1145\/3637528.3672052","10.1145\/3637528"],"URL":"https:\/\/doi.org\/10.1145\/3637528.3672052","relation":{},"subject":[],"published":{"date-parts":[[2024,8,24]]},"assertion":[{"value":"2024-08-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}