{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T12:27:43Z","timestamp":1730204863965,"version":"3.28.0"},"reference-count":34,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,12,6]]},"DOI":"10.1109\/cdc51059.2022.9993035","type":"proceedings-article","created":{"date-parts":[[2023,1,10]],"date-time":"2023-01-10T19:26:56Z","timestamp":1673378816000},"page":"7241-7247","source":"Crossref","is-referenced-by-count":0,"title":["Newton-based Policy Search for Networked Multi-agent Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Giorgio","family":"Manganini","sequence":"first","affiliation":[{"name":"Gran Sasso Science Institute (GSSI),Department of Computer Science,L&#x2019;Aquila,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Simone","family":"Fioravanti","sequence":"additional","affiliation":[{"name":"Gran Sasso Science Institute (GSSI),Department of Computer Science,L&#x2019;Aquila,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giorgia","family":"Ramponi","sequence":"additional","affiliation":[{"name":"ETH AI Center,Z&#x00FC;rich,Switzerland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCC.2007.913919"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-60990-0_12"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.32657\/10356\/90191"},{"article-title":"Safe, multiagent, reinforcement learning for autonomous driving","year":"2016","author":"Shalev-Shwartz","key":"ref5"},{"key":"ref6","first-page":"5872","article-title":"Fully decentralized multi-agent reinforcement learning with networked agents","volume-title":"International Conference on Machine Learning","author":"Zhang"},{"volume-title":"Numerical optimization","year":"2006","author":"Nocedal","key":"ref7"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MCS.2019.2900783"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1137\/18M1231298"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2015.2449811"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-060117-105131"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2008.02.003"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1613\/jair.806"},{"key":"ref15","first-page":"1008","article-title":"Actor-critic algorithms","author":"Konda","year":"2000","journal-title":"Advances in neural information processing systems"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-03194-1_4"},{"key":"ref17","article-title":"Approximate newton methods for policy search in markov decision processes","volume":"17","author":"Furmston","year":"2016","journal-title":"Journal of Machine Learning Research"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1002\/0471478210"},{"volume-title":"Distributed algorithms","year":"1996","author":"Lynch","key":"ref19"},{"key":"ref20","first-page":"29","article-title":"Improving the convergence of back-propagation learning with second order methods","volume-title":"Proceedings of the 1988 connectionist models summer school","author":"Becker"},{"key":"ref21","doi-asserted-by":"crossref","DOI":"10.1137\/1.9781611973433","volume-title":"Lectures on stochastic programming: modeling and theory","author":"Shapiro","year":"2014"},{"article-title":"A review of cooperative multiagent deep reinforcement learning","year":"2019","author":"OroojlooyJadid","key":"ref22"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.1900661"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2020.2976000"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2018.8619581"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CDC40024.2019.9029969"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/tac.2023.3288025"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2020.12.2021"},{"article-title":"Multi-agent reinforcement learning via double averaging primal-dual optimization","year":"2018","author":"Wai","key":"ref29"},{"key":"ref30","volume-title":"Stochastic approximation: a dynamical systems viewpoint","volume":"48","author":"Borkar","year":"2009"},{"article-title":"Multiagent actor-critic for mixed cooperative-competitive environments","volume-title":"Neural Information Processing Systems (NIPS)","author":"Lowe","key":"ref31"},{"article-title":"Policy gradient methods for reinforcement learning with function approximation and action-dependent baselines","year":"2017","author":"Thomas","key":"ref32"},{"issue":"3","key":"ref33","volume-title":"Nonlinear systems","author":"Khalil","year":"2002"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4614-5981-1"}],"event":{"name":"2022 IEEE 61st Conference on Decision and Control (CDC)","start":{"date-parts":[[2022,12,6]]},"location":"Cancun, Mexico","end":{"date-parts":[[2022,12,9]]}},"container-title":["2022 IEEE 61st Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9992315\/9992317\/09993035.pdf?arnumber=9993035","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,14]],"date-time":"2024-03-14T02:21:35Z","timestamp":1710382895000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9993035\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,6]]},"references-count":34,"URL":"https:\/\/doi.org\/10.1109\/cdc51059.2022.9993035","relation":{},"subject":[],"published":{"date-parts":[[2022,12,6]]}}}