{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,16]],"date-time":"2026-05-16T16:20:56Z","timestamp":1778948456709,"version":"3.51.4"},"reference-count":68,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T00:00:00Z","timestamp":1740787200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2024YFE0200600"],"award-info":[{"award-number":["2024YFE0200600"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62071425"],"award-info":[{"award-number":["62071425"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang Key Research and Development Plan","award":["2022C01093"],"award-info":[{"award-number":["2022C01093"]}]},{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LR23F010005"],"award-info":[{"award-number":["LR23F010005"]}]},{"name":"Zhejiang Lab","award":["3400-31299"],"award-info":[{"award-number":["3400-31299"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Mobile Comput."],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1109\/tmc.2024.3492272","type":"journal-article","created":{"date-parts":[[2024,11,7]],"date-time":"2024-11-07T20:02:25Z","timestamp":1731009745000},"page":"2301-2314","source":"Crossref","is-referenced-by-count":1,"title":["Noise Distribution Decomposition Based Multi-Agent Distributional Reinforcement Learning"],"prefix":"10.1109","volume":"24","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8898-4176","authenticated-orcid":false,"given":"Wei","family":"Geng","sequence":"first","affiliation":[{"name":"Zhejiang Lab, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baidi","family":"Xiao","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4297-5060","authenticated-orcid":false,"given":"Rongpeng","family":"Li","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6045-8994","authenticated-orcid":false,"given":"Ning","family":"Wei","sequence":"additional","affiliation":[{"name":"Zhejiang Lab, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1257-8905","authenticated-orcid":false,"given":"Dong","family":"Wang","sequence":"additional","affiliation":[{"name":"Research Institute of China Telecom, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5479-7890","authenticated-orcid":false,"given":"Zhifeng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang Lab, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Introduction to Reinforcement Learning","volume":"135","author":"Sutton","year":"1998"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.10827"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-09997-9"},{"key":"ref4","article-title":"Guiding pretraining in reinforcement learning with large language models","author":"Du","year":"2023"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2831228"},{"key":"ref8","article-title":"Guided deep reinforcement learning for swarm systems","author":"H\u00fcttenrauch","year":"2017"},{"key":"ref9","article-title":"Value-decomposition networks for cooperative multi-agent learning","author":"Sunehag","year":"2017"},{"issue":"1","key":"ref10","first-page":"7234","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"Proc. Conf. Mach. Learn."},{"key":"ref11","first-page":"10199","article-title":"Weighted QMIX: Expanding monotonic value function factorisation for deep multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rashid"},{"key":"ref12","first-page":"5887","article-title":"QTRAN: Learning to factorize with transformation for cooperative multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Son"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1029\/2019RS006959"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6086"},{"key":"ref15","article-title":"Multi-agent deep reinforcement learning with extremely noisy observations","author":"Kilinc","year":"2018"},{"key":"ref16","article-title":"A distributional perspective on reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bellemare"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3565287.3610255"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1515\/9781400831050"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3072435"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2019.2963462"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"ref22","first-page":"1104","article-title":"Implicit quantile networks for distributional reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dabney"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/s10479-005-5732-z"},{"key":"ref24","first-page":"9945","article-title":"DFAC framework: Factorizing the value function via quantile mixture for multi-agent distributional Q-learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sun"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2021.07.011"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-73003-5_196"},{"key":"ref27","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ho"},{"key":"ref28","first-page":"8780","article-title":"Diffusion models beat GANs on image synthesis","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dhariwal"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3528233.3530757"},{"key":"ref30","article-title":"GLIDE: Towards photorealistic image generation and editing with text-guided diffusion models","author":"Nichol","year":"2021"},{"key":"ref31","article-title":"Hierarchical text-conditional image generation with clip latents","author":"Ramesh","year":"2022"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-14435-6_7"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2006.281729"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2010.2043124"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2014.2332306"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2008.929381"},{"key":"ref38","first-page":"2137","article-title":"Learning to communicate with deep multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Foerster"},{"key":"ref39","first-page":"2244","article-title":"Learning multiagent communication with backpropagation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Sukhbaatar"},{"key":"ref40","article-title":"Playing atari with deep reinforcement learning","author":"Mnih","year":"2013"},{"key":"ref41","article-title":"Deep reinforcement learning variants of multi-agent learning algorithms","author":"Castaneda","year":"2016"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICMLA.2017.0-184"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511754098"},{"key":"ref44","first-page":"12619","article-title":"Distributional reward estimation for effective multi-agent deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hu"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TMRB.2019.2952148"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3016498"},{"key":"ref47","article-title":"Diffusion policies as an expressive policy class for offline reinforcement learning","author":"Wang","year":"2023"},{"key":"ref48","article-title":"Diffusion model is an effective planner and data synthesizer for multi-task reinforcement learning","author":"He","year":"2023"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-44213-1_30"},{"key":"ref50","article-title":"Beyond conservatism: Diffusion policies in offline multi-agent reinforcement learning","author":"Li","year":"2023"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.2307\/1968102"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.2140\/pjm.2009.241.117"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO.2018.8553410"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29680"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/BF00122574"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.42.12.1676"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.2307\/253675"},{"key":"ref59","first-page":"3509","article-title":"Algorithms for CVaR optimization in MDPs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chow"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11492"},{"key":"ref61","first-page":"6379","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lowe"},{"key":"ref62","article-title":"The starcraft multi-agent challenge","author":"Samvelyan","year":"2019"},{"key":"ref63","article-title":"Is independent learning all you need in the starcraft multi-agent challenge?","author":"de Witt","year":"2020"},{"key":"ref64","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref65","first-page":"24611","article-title":"The surprising effectiveness of PPO in cooperative multi-agent games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yu"},{"key":"ref66","first-page":"1582","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref67","first-page":"2256","article-title":"Deep unsupervised learning using nonequilibrium thermodynamics","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sohl-Dickstein"},{"issue":"3","key":"ref68","first-page":"285","article-title":"Teaching mathematical induction II","volume":"8","author":"Dubinsky","year":"1989","journal-title":"J. Math. Behav."}],"container-title":["IEEE Transactions on Mobile Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7755\/10874847\/10746312.pdf?arnumber=10746312","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,6]],"date-time":"2025-02-06T18:54:06Z","timestamp":1738868046000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10746312\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3]]},"references-count":68,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tmc.2024.3492272","relation":{},"ISSN":["1536-1233","1558-0660","2161-9875"],"issn-type":[{"value":"1536-1233","type":"print"},{"value":"1558-0660","type":"electronic"},{"value":"2161-9875","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3]]}}}