{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T01:04:20Z","timestamp":1759971860821,"version":"build-2065373602"},"reference-count":63,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62172142","61976243"],"award-info":[{"award-number":["62172142","61976243"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Science and Technology Development Plan of Henan Province","award":["231111222600","231100220600"],"award-info":[{"award-number":["231111222600","231100220600"]}]},{"name":"Scientific and Technological Innovation Teams of Colleges and Universities in Henan Province","award":["24IRTSTHN022"],"award-info":[{"award-number":["24IRTSTHN022"]}]},{"DOI":"10.13039\/501100013066","name":"Basic Research Projects in the University of Henan Province","doi-asserted-by":"publisher","award":["23ZX003","25ZX009"],"award-info":[{"award-number":["23ZX003","25ZX009"]}],"id":[{"id":"10.13039\/501100013066","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1109\/tnnls.2025.3573801","type":"journal-article","created":{"date-parts":[[2025,6,10]],"date-time":"2025-06-10T13:52:39Z","timestamp":1749563559000},"page":"19423-19436","source":"Crossref","is-referenced-by-count":0,"title":["A Decentralized Actor\u2013Critic Algorithm With Entropy Regularization and Its Finite-Time Analysis"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-5500-7192","authenticated-orcid":false,"given":"Tao","family":"Mao","sequence":"first","affiliation":[{"name":"School of lnformation Engineering, Henan University of Science and Technology, Luoyang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6411-7035","authenticated-orcid":false,"given":"Junlong","family":"Zhu","sequence":"additional","affiliation":[{"name":"School of lnformation Engineering, Henan University of Science and Technology, Luoyang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2523-1089","authenticated-orcid":false,"given":"Mingchuan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Longmen Laboratory, Luoyang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0044-6059","authenticated-orcid":false,"given":"Quanbo","family":"Ge","sequence":"additional","affiliation":[{"name":"School of Automation, Nanjing University of Information Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0932-8788","authenticated-orcid":false,"given":"Ruijuan","family":"Zheng","sequence":"additional","affiliation":[{"name":"School of lnformation Engineering, Henan University of Science and Technology, Luoyang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1572-5293","authenticated-orcid":false,"given":"Qingtao","family":"Wu","sequence":"additional","affiliation":[{"name":"School of lnformation Engineering, Henan University of Science and Technology, Luoyang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.5772\/57313"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2019.2893683"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TCCN.2021.3063170"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3308558.3313433"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref7","first-page":"1008","article-title":"Actor-critic algorithms","volume-title":"Proc. NIPS","author":"Konda"},{"key":"ref8","article-title":"Decentralized multi-agent actor-critic with generative inference","author":"Corder","year":"2019","journal-title":"arXiv:1910.03058"},{"key":"ref9","first-page":"3794","article-title":"Sample and communication-efficient decentralized actor-critic algorithms with finite-time analysis","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","volume":"162","author":"Chen"},{"key":"ref10","article-title":"Finite-time analysis of decentralized single-timescale actor-critic","volume":"2023","author":"Luo","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-023-08845-x"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i7.26056"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CDC49753.2023.10383350"},{"article-title":"Finite-time convergence and sample complexity of multi-agent actor-critic reinforcement learning with average reward","volume-title":"Proc. 10th Int. Conf. Learn. Represent.","author":"Liu","key":"ref14"},{"key":"ref15","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih"},{"key":"ref16","first-page":"1856","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8"},{"key":"ref18","first-page":"1352","article-title":"Reinforcement learning with deep energy-based policies","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2016.2585302"},{"key":"ref20","first-page":"1","article-title":"An asynchronous multi-agent actor-critic algorithm for distributed reinforcement learning","volume-title":"Proc. IEEE Int. Conf. Commun. (ICC)","author":"Lin"},{"key":"ref21","first-page":"13762","article-title":"Decentralized TD tracking with linear function approximation and its finite-time analysis","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Wang"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2023.3287952"},{"key":"ref23","article-title":"Finite-sample analysis of decentralized temporal-difference learning with linear function approximation","author":"Sun","year":"2019","journal-title":"arXiv:1911.00934"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ACC.2016.7524910"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2020.2995814"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/IEEECONF53345.2021.9723200"},{"key":"ref27","first-page":"5729","article-title":"Hessian aided policy gradient","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","volume":"97","author":"Shen"},{"key":"ref28","first-page":"1944","article-title":"Non-asymptotic analysis of biased stochastic approximation scheme","volume-title":"Proc. 32nd Conf. Learn. Theory","volume":"99","author":"Karimi"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17252"},{"key":"ref30","first-page":"4422","article-title":"Momentum-based policy gradient methods","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Huang"},{"key":"ref31","article-title":"Joint optimization of multi-objective reinforcement learning with policy gradient based algorithm","author":"Bai","year":"2021","journal-title":"arXiv:2105.14125"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i9.21169"},{"article-title":"Actor-critic algorithms","year":"2002","author":"Konda","key":"ref33"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysconle.2010.08.013"},{"article-title":"Neural policy gradient methods: Global optimality and rates of convergence","volume-title":"Proc. 8th Int. Conf. Learn. Represent.","author":"Wang","key":"ref35"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-023-06303-2"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/JSAIT.2021.3078754"},{"key":"ref38","article-title":"Non-asymptotic convergence analysis of two time-scale (natural) actor-critic algorithms","author":"Xu","year":"2020","journal-title":"arXiv:2005.03557"},{"article-title":"A finite-time analysis of two time-scale actor-critic methods","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wu","key":"ref39"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1137\/23m1587683"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CDC40024.2019.9029257"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1016\/j.ifacol.2020.12.2021"},{"key":"ref43","first-page":"853","article-title":"Modeling the interaction between agents in cooperative multi-agent reinforcement learning","volume-title":"Proc. 20th Int. Conf. Auto. Agents MultiAgent Syst.","author":"Ma"},{"key":"ref44","first-page":"6379","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc. NIPS","author":"Lowe"},{"key":"ref45","first-page":"919","article-title":"Actor-critic fictitious play in simultaneous move multistage games","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"P\u00e9rolat"},{"key":"ref46","article-title":"Delay-aware multi-agent reinforcement learning","author":"Chen","year":"2020","journal-title":"arXiv:2005.05441"},{"key":"ref47","article-title":"Shaping advice in deep multi-agent reinforcement learning","author":"Xiao","year":"2021","journal-title":"arXiv:2103.15941"},{"key":"ref48","article-title":"Natural actor-critic converges globally for hierarchical linear quadratic regulator","author":"Luo","year":"2019","journal-title":"arXiv:1912.06875"},{"key":"ref49","first-page":"278","article-title":"Learning to coordinate in multi-agent systems: A coordinated actor-critic algorithm and finite-time guarantees","volume-title":"Proc. Learn. Dyn. Control Conf.","volume":"168","author":"Zeng"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3139138"},{"key":"ref51","first-page":"2755","article-title":"Privacy-preserving decentralized actor-critic for cooperative multi-agent reinforcement learning","volume-title":"Proc. 27th Int. Conf. Artif. Intell. Statist.","volume":"238","author":"Ahmed"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2023.3268475"},{"article-title":"Improving sample complexity bounds for (Natural) actor-critic algorithms","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xu","key":"ref53"},{"key":"ref54","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume-title":"Proc. NIPS","author":"Sutton"},{"key":"ref55","first-page":"5867","article-title":"Fully decentralized multi-agent reinforcement learning with networked agents","volume-title":"Proc. 35th Int. Conf. Mach. Learn. (ICML)","volume":"80","author":"Zhang"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2020.2024"},{"key":"ref57","first-page":"8665","article-title":"Finite-sample analysis for SARSA with linear function approximation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Zou"},{"article-title":"Reanalysis of variance reduced temporal difference learning","volume-title":"Proc. 8th Int. Conf. Learn. Represent.","author":"Xu","key":"ref58"},{"key":"ref59","first-page":"64","article-title":"Optimality and approximation with policy gradient methods in Markov decision processes","volume-title":"Proc. Conf. Learn. Theory","author":"Agarwal"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413527"},{"key":"ref61","first-page":"18825","article-title":"Taming communication and sample complexities in decentralized policy evaluation for cooperative multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Zhang"},{"key":"ref62","first-page":"23049","article-title":"RMIX: Learning risk-sensitive policies for cooperative reinforcement learning agents","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Qiu"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2023.3281899"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5962385\/11195929\/11029594.pdf?arnumber=11029594","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T17:38:43Z","timestamp":1759945123000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11029594\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10]]},"references-count":63,"journal-issue":{"issue":"10"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2025.3573801","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"type":"print","value":"2162-237X"},{"type":"electronic","value":"2162-2388"}],"subject":[],"published":{"date-parts":[[2025,10]]}}}