{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T16:16:45Z","timestamp":1730305005597,"version":"3.28.0"},"reference-count":18,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,11,2]],"date-time":"2023-11-02T00:00:00Z","timestamp":1698883200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,11,2]],"date-time":"2023-11-02T00:00:00Z","timestamp":1698883200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,11,2]]},"DOI":"10.1109\/wcsp58612.2023.10404257","type":"proceedings-article","created":{"date-parts":[[2024,2,2]],"date-time":"2024-02-02T18:25:28Z","timestamp":1706898328000},"page":"342-347","source":"Crossref","is-referenced-by-count":0,"title":["Decomposition-based Multi-Agent Distributional Reinforcement Learning for Task-Oriented UAV Collaboration with Noisy Rewards"],"prefix":"10.1109","author":[{"given":"Wei","family":"Geng","sequence":"first","affiliation":[{"name":"UCAS,Hangzhou Institute for Advanced Study"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baidi","family":"Xiao","sequence":"additional","affiliation":[{"name":"Zhejiang University,College of Information Science and Electronic Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rongpeng","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University,College of Information Science and Electronic Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ning","family":"Wei","sequence":"additional","affiliation":[{"name":"Zhejiang Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhifeng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Honggang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Zhejiang Lab"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref1","DOI":"10.1029\/2019RS006959"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1609\/aaai.v34i04.6086"},{"year":"2018","author":"Kilinc","article-title":"Multi-agent deep reinforcement learning with extremely noisy observations","key":"ref3"},{"year":"2017","author":"H\u00fcttenrauch","article-title":"Guided deep reinforcement learning for swarm systems","key":"ref4"},{"year":"2017","author":"Sunehag","article-title":"Value-decomposition networks for cooperative multi-agent learning","key":"ref5"},{"issue":"1","key":"ref6","first-page":"7234","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"The Journal of Machine Learning Research"},{"key":"ref7","article-title":"Weighted qmix: Expanding monotonic value function factorisation for deep multi-agent reinforcement learning","author":"Rashid","year":"2020","journal-title":"Advances in neural information processing systems"},{"volume-title":"International conference on machine learning","author":"Son","article-title":"Qtran: Learning to factorize with transformation for cooperative multi-agent reinforcement learning","key":"ref8"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.1007\/978-0-387-73003-5_196"},{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1609\/aaai.v32i1.11791"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.1007\/978-3-319-28929-8"},{"volume-title":"International conference on machine learning","author":"Bellemare","article-title":"A distributional perspective on reinforcement learning","key":"ref12"},{"volume-title":"International conference on machine learning","author":"Dabney","article-title":"Implicit quantile networks for distributional reinforcement learning","key":"ref13"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.1609\/aaai.v32i1.11492"},{"volume-title":"Advances in neural information processing systems","author":"Lowe","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","key":"ref15"},{"year":"2020","author":"de Witt","article-title":"Is independent learning all you need in the starcraft multi-agent challenge?","key":"ref16"},{"year":"2017","author":"Schulman","article-title":"Proximal policy optimization algorithms","key":"ref17"},{"volume-title":"Advances in Neural Information Processing Systems","author":"Yu","article-title":"The surprising effectiveness of ppo in cooperative multi-agent games","key":"ref18"}],"event":{"name":"2023 International Conference on Wireless Communications and Signal Processing (WCSP)","start":{"date-parts":[[2023,11,2]]},"location":"Hangzhou, China","end":{"date-parts":[[2023,11,4]]}},"container-title":["2023 International Conference on Wireless Communications and Signal Processing (WCSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10404087\/10404080\/10404257.pdf?arnumber=10404257","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,6]],"date-time":"2024-02-06T20:46:04Z","timestamp":1707252364000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10404257\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,2]]},"references-count":18,"URL":"https:\/\/doi.org\/10.1109\/wcsp58612.2023.10404257","relation":{},"subject":[],"published":{"date-parts":[[2023,11,2]]}}}