{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,6]],"date-time":"2025-08-06T13:37:05Z","timestamp":1754487425193,"version":"3.28.0"},"reference-count":42,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,10,1]],"date-time":"2023-10-01T00:00:00Z","timestamp":1696118400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,10,1]],"date-time":"2023-10-01T00:00:00Z","timestamp":1696118400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,10,1]]},"DOI":"10.1109\/iros55552.2023.10341563","type":"proceedings-article","created":{"date-parts":[[2023,12,13]],"date-time":"2023-12-13T19:17:55Z","timestamp":1702495075000},"page":"8854-8861","source":"Crossref","is-referenced-by-count":4,"title":["Decentralized Multi-Agent Reinforcement Learning with Global State Prediction"],"prefix":"10.1109","author":[{"given":"Joshua","family":"Bloom","sequence":"first","affiliation":[{"name":"Worcester Polytechnic Institute,Robotics Engineering,Worcester,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pranjal","family":"Paliwal","sequence":"additional","affiliation":[{"name":"Worcester Polytechnic Institute,Robotics Engineering,Worcester,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Apratim","family":"Mukherjee","sequence":"additional","affiliation":[{"name":"Worcester Polytechnic Institute,Robotics Engineering,Worcester,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Carlo","family":"Pinciroli","sequence":"additional","affiliation":[{"name":"Worcester Polytechnic Institute,Robotics Engineering,Worcester,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-09996-w"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.11418"},{"key":"ref3","article-title":"Learning to Communicate with Deep Multi-Agent Reinforcement Learning","volume":"29","author":"Foerster","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561359"},{"volume-title":"Dealing with Non-Stationarity in Multi-Agent Deep Reinforcement Learning","year":"2019","author":"Papoudakis","key":"ref5"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196790"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1023\/b:auro.0000033972.50769.1c"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MED.2007.4433724"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2018.00059"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3025287"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"volume-title":"Continuous control with deep reinforcement learning","year":"2015","author":"Lillicrap","key":"ref13"},{"key":"ref14","first-page":"1587","article-title":"Addressing function approxi-mation error in actor-critic methods","volume-title":"International conference on machine learning. PMLR","author":"Fujimoto","year":"2018"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/s11370-021-00398-z"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-019-09421-1"},{"volume-title":"Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments","year":"2020","author":"Lowe","key":"ref17"},{"volume-title":"Asynchronous Methods for Deep Reinforcement Learning","year":"2016","author":"Mnih","key":"ref18"},{"volume-title":"Reducing Overestimation Bias in Multi-Agent Domains Using Double Centralized Critics","year":"2019","author":"Ackermann","key":"ref19"},{"volume-title":"The Surprising Effectiveness of PPO in Cooperative, Multi-Agent Games","year":"2022","author":"Yu","key":"ref20"},{"volume-title":"Value-Decomposition Networks For Cooperative Multi-Agent Learning","year":"2017","author":"Sunehag","key":"ref21"},{"volume-title":"QMIX: Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning","year":"2018","author":"Rashid","key":"ref22"},{"volume-title":"Scalable Multi-Agent Reinforcement Learning through Intelligent Information Aggregation","year":"2023","author":"Nayak","key":"ref23"},{"volume-title":"Machine Theory of Mind","year":"2018","author":"Rabinowitz","key":"ref24"},{"volume-title":"Modeling Others using Oneself in Multi-Agent Reinforcement Learning","year":"2018","author":"Raileanu","key":"ref25"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2006.1641893"},{"volume-title":"Social Influence as Intrinsic Motivation for Multi-Agent Deep Reinforcement Learning","year":"2019","author":"Jaques","key":"ref27"},{"volume-title":"Emergent Social Learning via Multi-agent Reinforcement Learning","year":"2021","author":"Ndousse","key":"ref28"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1017\/S0140525X00076512"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.tics.2009.08.002"},{"volume-title":"Reinforcement Learning with Unsupervised Auxiliary Tasks","year":"2016","author":"Jaderberg","key":"ref31"},{"volume-title":"Multi-task Deep Reinforcement Learning with PopArt","year":"2018","author":"Hessel","key":"ref32"},{"volume-title":"Loss is its own Reward: Self-Supervision for Reinforcement Learning","year":"2017","author":"Shelhamer","key":"ref33"},{"volume-title":"Learning to Navigate in Complex Environments","year":"2017","author":"Mirowski","key":"ref34"},{"key":"ref35","article-title":"Learning Complementary Representations of the Past using Auxiliary Tasks in Partially Observable Reinforcement Learning","author":"Baisero","year":"2020","journal-title":"New Zealand"},{"key":"ref36","doi-asserted-by":"crossref","DOI":"10.1109\/CVPRW.2017.70","volume-title":"Curiosity-driven Exploration by Self-supervised Prediction","author":"Pathak","year":"2017"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-31978-6_8"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/s11721-012-0072-5"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2016.7759558"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2010.5649153"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553380"},{"key":"ref42","doi-asserted-by":"crossref","DOI":"10.1609\/aiide.v15i1.5221","volume-title":"Agent Modeling as Auxiliary Task for Deep Reinforcement Learning","author":"Hernandez-Leal","year":"2019"}],"event":{"name":"2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2023,10,1]]},"location":"Detroit, MI, USA","end":{"date-parts":[[2023,10,5]]}},"container-title":["2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10341341\/10341342\/10341563.pdf?arnumber=10341563","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,20]],"date-time":"2023-12-20T00:14:22Z","timestamp":1703031262000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10341563\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,1]]},"references-count":42,"URL":"https:\/\/doi.org\/10.1109\/iros55552.2023.10341563","relation":{},"subject":[],"published":{"date-parts":[[2023,10,1]]}}}