{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T14:07:25Z","timestamp":1775570845584,"version":"3.50.1"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T00:00:00Z","timestamp":1775520000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T00:00:00Z","timestamp":1775520000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72342006, 72371253, 72461160315"],"award-info":[{"award-number":["72342006, 72371253, 72461160315"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Discrete Event Dyn Syst"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1007\/s10626-026-00439-8","type":"journal-article","created":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T13:19:15Z","timestamp":1775567955000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Independent policy gradient-based reinforcement learning for economic and reliable energy management of multi-microgrid systems"],"prefix":"10.1007","volume":"36","author":[{"given":"Junkai","family":"Hu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Xia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,7]]},"reference":[{"issue":"3","key":"439_CR1","doi-asserted-by":"publisher","first-page":"1457","DOI":"10.1109\/TSTE.2023.3236634","volume":"14","author":"AA Abdalla","year":"2023","unstructured":"Abdalla AA, El Moursi MS, El-Fouly TH, Al Hosani KH (2023) A novel adaptive power smoothing approach for PV power plant with hybrid energy storage system. IEEE Trans Sustain Energy 14(3):1457\u20131473","journal-title":"IEEE Trans Sustain Energy"},{"issue":"98","key":"439_CR2","first-page":"1","volume":"22","author":"A Agarwal","year":"2021","unstructured":"Agarwal A, Kakade SM, Lee JD, Mahajan G (2021) On the theory of policy gradient methods: Optimality, approximation, and distribution shift. J Mach Learn Res 22(98):1\u201376","journal-title":"J Mach Learn Res"},{"issue":"3","key":"439_CR3","doi-asserted-by":"publisher","first-page":"1238","DOI":"10.1109\/TII.2018.2881540","volume":"15","author":"MN Alam","year":"2018","unstructured":"Alam MN, Chakrabarti S, Ghosh A (2018) Networked microgrids: State-of-the-art and future perspectives. IEEE Trans Ind Inform 15(3):1238\u20131250","journal-title":"IEEE Trans Ind Inform"},{"key":"439_CR4","doi-asserted-by":"publisher","first-page":"366","DOI":"10.1016\/j.renene.2023.01.059","volume":"205","author":"P Ar\u00e9valo","year":"2023","unstructured":"Ar\u00e9valo P, Benavides D, Tostado-V\u00e9liz M, Aguado JA, Jurado F (2023) Smart monitoring method for photovoltaic systems and failure control based on power smoothing techniques. Renew Energy 205:366\u2013383","journal-title":"Renew Energy"},{"issue":"1","key":"439_CR5","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1007\/s00186-024-00857-0","volume":"99","author":"N B\u00e4uerle","year":"2024","unstructured":"B\u00e4uerle N, Ja\u015bkiewicz A (2024) Markov decision processes with risk-sensitive criteria: An overview. Math Methods Oper Res 99(1):141\u2013178","journal-title":"Math Methods Oper Res"},{"key":"439_CR6","unstructured":"Berner C, Brockman G, Chan B, Cheung V, D\u0119biak P, Dennison C, Farhi D, Fischer Q, Hashme S, Hesse C et al (2019) Dota 2 with large scale deep reinforcement learning. arXiv preprint arXiv:1912.06680"},{"key":"439_CR7","volume-title":"Neuro-Dynamic Programming","author":"D Bertsekas","year":"1996","unstructured":"Bertsekas D, Tsitsiklis JN (1996) Neuro-Dynamic Programming. Athena Scientific"},{"key":"439_CR8","doi-asserted-by":"crossref","unstructured":"Bisi L, Sabbioni L, Vittori E, Papini M, Restelli M (2021) Risk-averse trust region optimization for reward-volatility reduction. In: Proceedings of the twenty-ninth international conference on international joint conferences on artificial intelligence, pp 4583\u20134589","DOI":"10.24963\/ijcai.2020\/632"},{"issue":"4","key":"439_CR9","doi-asserted-by":"publisher","first-page":"659","DOI":"10.1007\/s10626-024-00405-2","volume":"34","author":"R Blancas-Rivera","year":"2024","unstructured":"Blancas-Rivera R, Jasso-Fuentes H (2024) Discrete-time hybrid control with risk-sensitive discounted costs. Discrete Event Dyn Syst 34(4):659\u2013687","journal-title":"Discrete Event Dyn Syst"},{"key":"439_CR10","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441","volume-title":"Convex Optimization","author":"S Boyd","year":"2004","unstructured":"Boyd S, Vandenberghe L (2004) Convex Optimization. Cambridge University Press"},{"issue":"3\u20134","key":"439_CR11","doi-asserted-by":"publisher","first-page":"231","DOI":"10.1561\/2200000050","volume":"8","author":"S Bubeck","year":"2015","unstructured":"Bubeck S (2015) Convex optimization: Algorithms and complexity. Found Trends Mach Learn 8(3\u20134):231\u2013357","journal-title":"Found Trends Mach Learn"},{"key":"439_CR12","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1016\/j.arcontrol.2018.09.005","volume":"46","author":"L Bu\u015foniu","year":"2018","unstructured":"Bu\u015foniu L, De Bruin T, Toli\u0107 D, Kober J, Palunko I (2018) Reinforcement learning for control: Performance, stability, and deep approximators. Annu Rev Control 46:8\u201328","journal-title":"Annu Rev Control"},{"key":"439_CR13","doi-asserted-by":"publisher","first-page":"627","DOI":"10.1016\/j.apenergy.2019.01.102","volume":"238","author":"S Chapaloglou","year":"2019","unstructured":"Chapaloglou S, Nesiadis A, Iliadis P, Atsonios K, Nikolopoulos N, Grammelis P, Yiakopoulos C, Antoniadis I, Kakaras E (2019) Smart energy management algorithm for load smoothing and peak shaving based on load forecasting of an island\u2019s power system. Appl Energy 238:627\u2013642","journal-title":"Appl Energy"},{"key":"439_CR14","doi-asserted-by":"publisher","first-page":"119106","DOI":"10.1016\/j.apenergy.2022.119106","volume":"316","author":"W Chen","year":"2022","unstructured":"Chen W, Wang J, Yu G, Chen J, Hu Y (2022) Research on day-ahead transactions between multi-microgrid based on cooperative game model. Appl Energy 316:119106","journal-title":"Appl Energy"},{"key":"439_CR15","doi-asserted-by":"crossref","unstructured":"Chen Y, Cassandras CG (2025) Scalable adaptive traffic light control over a traffic network including turns, transit delays, and blocking. Discrete Event Dyn Syst 1\u201330","DOI":"10.1007\/s10626-025-00416-7"},{"key":"439_CR16","unstructured":"Cheng M, Zhou R, Kumar P, Tian C (2024) Provable policy gradient methods for average-reward Markov potential games. In: International conference on artificial intelligence and statistics, pp 4699\u20134707"},{"issue":"8","key":"439_CR17","doi-asserted-by":"publisher","first-page":"2327","DOI":"10.1109\/TAC.2018.2797217","volume":"63","author":"SR Etesami","year":"2018","unstructured":"Etesami SR, Saad W, Mandayam NB, Poor HV (2018) Stochastic games for the smart grid energy management with prospect prosumers. IEEE Trans Autom Control 63(8):2327\u20132342","journal-title":"IEEE Trans Autom Control"},{"issue":"1","key":"439_CR18","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1287\/moor.14.1.147","volume":"14","author":"JA Filar","year":"1989","unstructured":"Filar JA, Lee HM (1989) Variance-penalized markov decision processes. Math Oper Res 14(1):147\u2013161","journal-title":"Math Oper Res"},{"issue":"1","key":"439_CR19","first-page":"1437","volume":"16","author":"J Garc\u0131a","year":"2015","unstructured":"Garc\u0131a J, Fern\u00e1ndez F (2015) A comprehensive survey on safe reinforcement learning. J Mach Learn Res 16(1):1437\u20131480","journal-title":"J Mach Learn Res"},{"key":"439_CR20","doi-asserted-by":"publisher","first-page":"121052","DOI":"10.1016\/j.apenergy.2023.121052","volume":"340","author":"M Ghafoori","year":"2023","unstructured":"Ghafoori M, Abdallah M, Kim S (2023) Electricity peak shaving for commercial buildings using machine learning and vehicle to building (V2B) system. Appl Energy 340:121052","journal-title":"Appl Energy"},{"issue":"6","key":"439_CR21","doi-asserted-by":"publisher","first-page":"5235","DOI":"10.1109\/TPWRS.2021.3069781","volume":"36","author":"Y Guo","year":"2021","unstructured":"Guo Y, Zhang Q, Wang Z (2021) Cooperative peak shaving and voltage regulation in unbalanced distribution feeders. IEEE Trans Power Syst 36(6):5235\u20135244","journal-title":"IEEE Trans Power Syst"},{"key":"439_CR22","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1109\/TSG.2024.3452064","volume":"16","author":"J Han","year":"2025","unstructured":"Han J, Fang Y, Li Y, Du E, Zhang N (2025) Optimal planning of multi-microgrid system with shared energy storage based on capacity leasing and energy sharing. IEEE Trans Smart Grid 16:16\u201331","journal-title":"IEEE Trans Smart Grid"},{"key":"439_CR23","doi-asserted-by":"publisher","first-page":"8659","DOI":"10.1109\/TASE.2024.3487292","volume":"22","author":"J Hu","year":"2025","unstructured":"Hu J, Xia L, Hu J, Wu H (2025) Economical and reliable energy management for networked microgrids in a multi-agent collaborative manner. IEEE Trans Autom Sci Eng 22:8659\u20138669","journal-title":"IEEE Trans Autom Sci Eng"},{"key":"439_CR24","doi-asserted-by":"publisher","first-page":"539","DOI":"10.1007\/s10626-018-0273-1","volume":"28","author":"Y Huang","year":"2018","unstructured":"Huang Y (2018) Finite horizon continuous-time Markov decision processes with mean and variance criteria. Discrete Event Dyn Syst 28:539\u2013564","journal-title":"Discrete Event Dyn Syst"},{"key":"439_CR25","unstructured":"IESO (2023) Ontario and Market Demand, https:\/\/ieso.ca\/Power-Data\/Data-Directory"},{"key":"439_CR26","unstructured":"Leonardos S, Overman W, Panageas I, Piliouras G (2022) Global convergence of multi-agent policy gradient in Markov potential games. In: International conference on learning representations"},{"key":"439_CR27","doi-asserted-by":"publisher","first-page":"100208","DOI":"10.1016\/j.egyai.2022.100208","volume":"11","author":"J Li","year":"2023","unstructured":"Li J, Herdem MS, Nathwani J, Wen JZ (2023) Methods and applications for artificial intelligence, big data, internet of things, and blockchain in smart energy management. Energy AI 11:100208","journal-title":"Energy AI"},{"issue":"3","key":"439_CR28","doi-asserted-by":"publisher","first-page":"1057","DOI":"10.1016\/j.ejor.2023.06.022","volume":"311","author":"S Ma","year":"2023","unstructured":"Ma S, Ma X, Xia L (2023) A unified algorithm framework for mean-variance optimization in discounted Markov decision processes. Eur J Oper Res 311(3):1057\u20131067","journal-title":"Eur J Oper Res"},{"key":"439_CR29","doi-asserted-by":"crossref","unstructured":"Ma X, Tang X, Xia L, Yang J, Zhao Q (2021) Average-reward reinforcement learning with trust region methods. In: International joint conference on artificial intelligence, pp 2797\u20132083","DOI":"10.24963\/ijcai.2021\/385"},{"issue":"3","key":"439_CR30","doi-asserted-by":"publisher","first-page":"2177","DOI":"10.1109\/TPWRS.2022.3187069","volume":"38","author":"R Manojkumar","year":"2022","unstructured":"Manojkumar R, Kumar C, Ganguly S, Gooi HB, Mekhilef S, Catal\u00e3o JP (2022) Rule-based peak shaving using master-slave level optimization in a diesel generator supplied microgrid. IEEE Trans Power Syst 38(3):2177\u20132188","journal-title":"IEEE Trans Power Syst"},{"key":"439_CR31","doi-asserted-by":"publisher","first-page":"119596","DOI":"10.1016\/j.apenergy.2022.119596","volume":"323","author":"A Nawaz","year":"2022","unstructured":"Nawaz A, Zhou M, Wu J, Long C (2022) A comprehensive review on energy management, demand response, and coordination schemes utilization in multi-microgrids network. Appl Energy 323:119596","journal-title":"Appl Energy"},{"key":"439_CR32","unstructured":"NREL (2025) National wind technology center. (online). https:\/\/midcdmz.nrel.gov\/apps\/sitehome.pl?site=NWTC. Accessed 1 Jan 2025"},{"key":"439_CR33","doi-asserted-by":"publisher","first-page":"100323","DOI":"10.1016\/j.egyai.2023.100323","volume":"16","author":"T Peirelinck","year":"2024","unstructured":"Peirelinck T, Hermans C, Spiessens F, Deconinck G (2024) Combined peak reduction and self-consumption using proximal policy optimisation. Energy AI 16:100323","journal-title":"Energy AI"},{"issue":"5","key":"439_CR34","doi-asserted-by":"publisher","first-page":"537","DOI":"10.1561\/2200000091","volume":"15","author":"L Prashanth","year":"2022","unstructured":"Prashanth L, Fu MC (2022) Risk-sensitive reinforcement learning via policy gradient search. Found Trends Mach Learn 15(5):537\u2013693","journal-title":"Found Trends Mach Learn"},{"key":"439_CR35","first-page":"23049","volume":"34","author":"W Qiu","year":"2021","unstructured":"Qiu W, Wang X, Yu R, Wang R, He X, An B, Obraztsova S, Rabinovich Z (2021) RMIX: Learning risk-sensitive policies for cooperative reinforcement learning agents. Adv Neural Inf Process Syst 34:23049\u201323062","journal-title":"Adv Neural Inf Process Syst"},{"issue":"1","key":"439_CR36","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1007\/s10626-024-00393-3","volume":"34","author":"M Rosa","year":"2024","unstructured":"Rosa M, Cury JE, Baldissera FL (2024) A modular synthesis approach for the coordination of multi-agent systems: The multiple team case. Discrete Event Dyn Syst 34(1):163\u2013198","journal-title":"Discrete Event Dyn Syst"},{"key":"439_CR37","doi-asserted-by":"crossref","unstructured":"Rostmnezhad Z, Dessaint L (2023) Power management in smart buildings using reinforcement learning. In: 2023 IEEE Power & Energy Society Innovative Smart Grid Technologies Conference (ISGT), pp 1\u20135","DOI":"10.1109\/ISGT51731.2023.10066398"},{"key":"439_CR38","unstructured":"Schulman J, Moritz P, Levine S, Jordan M, Abbeel P (2016) High-dimensional continuous control using generalized advantage estimation. In: International conference on learning representations"},{"key":"439_CR39","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347"},{"key":"439_CR40","first-page":"34791","volume":"36","author":"S Shen","year":"2023","unstructured":"Shen S, Ma C, Li C, Liu W, Fu Y, Mei S, Liu X, Wang C (2023) RiskQ: Risk-sensitive multi-agent reinforcement learning value factorization. Adv Neural Inf Process Syst 36:34791\u201334825","journal-title":"Adv Neural Inf Process Syst"},{"issue":"7676","key":"439_CR41","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver D, Schrittwieser J, Simonyan K, Antonoglou I, Huang A, Guez A, Hubert T, Baker L, Lai M, Bolton A et al (2017) Mastering the game of Go without human knowledge. Nature 550(7676):354\u2013359","journal-title":"Nature"},{"key":"439_CR42","unstructured":"Sutton RS, Barto AG (2018) Reinforcement learning: An introduction. MIT press"},{"key":"439_CR43","first-page":"1057","volume":"12","author":"RS Sutton","year":"1999","unstructured":"Sutton RS, McAllester D, Singh S, Mansour Y (1999) Policy gradient methods for reinforcement learning with function approximation. Adv Neural Inf Process Syst 12:1057\u20131063","journal-title":"Adv Neural Inf Process Syst"},{"issue":"3","key":"439_CR44","doi-asserted-by":"publisher","first-page":"439","DOI":"10.1007\/s10626-020-00335-9","volume":"31","author":"H Tang","year":"2021","unstructured":"Tang H, Liu C, Cao Y, Lv K, Zhang Q (2021) Hierarchical scheduling learning optimisation of two-area active distribution system considering peak shaving demand of power grid. Discrete Event Dyn Syst 31(3):439\u2013468","journal-title":"Discrete Event Dyn Syst"},{"issue":"1","key":"439_CR45","doi-asserted-by":"publisher","first-page":"368","DOI":"10.1109\/TSTE.2023.3287871","volume":"15","author":"X Wang","year":"2023","unstructured":"Wang X, Zhou J, Qin B, Guo L (2023) Coordinated power smoothing control strategy of multi-wind turbines and energy storage systems in wind farm based on MADRL. IEEE Trans Sustain Energy 15(1):368\u2013380","journal-title":"IEEE Trans Sustain Energy"},{"issue":"2","key":"439_CR46","doi-asserted-by":"publisher","first-page":"582","DOI":"10.1016\/j.ejor.2017.06.052","volume":"264","author":"T Weitzel","year":"2018","unstructured":"Weitzel T, Glock CH (2018) Energy management for stationary electric energy storage systems: A systematic literature review. Eur J Oper Res 264(2):582\u2013606","journal-title":"Eur J Oper Res"},{"key":"439_CR47","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1007\/s10626-017-0258-5","volume":"28","author":"L Xia","year":"2018","unstructured":"Xia L (2018) Variance minimization of parameterized Markov decision processes. Discrete Event Dyn Syst 28:63\u201381","journal-title":"Discrete Event Dyn Syst"},{"issue":"12","key":"439_CR48","doi-asserted-by":"publisher","first-page":"2808","DOI":"10.1111\/poms.13252","volume":"29","author":"L Xia","year":"2020","unstructured":"Xia L (2020) Risk-sensitive Markov decision processes with combined metrics of mean and variance. Prod Oper Manag 29(12):2808\u20132827","journal-title":"Prod Oper Manag"},{"issue":"3","key":"439_CR49","first-page":"1195","volume":"17","author":"Z Yang","year":"2020","unstructured":"Yang Z, Xia L, Guan X (2020) Fluctuation reduction of wind power and sizing of battery energy storage systems in microgrids. IEEE Trans Autom Sci Eng 17(3):1195\u20131207","journal-title":"IEEE Trans Autom Sci Eng"},{"key":"439_CR50","first-page":"24611","volume":"35","author":"C Yu","year":"2022","unstructured":"Yu C, Velu A, Vinitsky E, Gao J, Wang Y, Bayen A, Wu Y (2022) The surprising effectiveness of ppo in cooperative multi-agent games. Adv Neural Inf Process Syst 35:24611\u201324624","journal-title":"Adv Neural Inf Process Syst"},{"issue":"10","key":"439_CR51","doi-asserted-by":"publisher","first-page":"6499","DOI":"10.1109\/TAC.2024.3387208","volume":"69","author":"R Zhang","year":"2024","unstructured":"Zhang R, Ren Z, Li N (2024) Gradient play in stochastic games: Stationary points, convergence, and sample complexity. IEEE Trans Autom Control 69(10):6499\u20136514","journal-title":"IEEE Trans Autom Control"},{"issue":"32","key":"439_CR52","first-page":"1","volume":"25","author":"Y Zhong","year":"2024","unstructured":"Zhong Y, Kuba JG, Feng X, Hu S, Ji J, Yang Y (2024) Heterogeneous-agent reinforcement learning. J Mach Learn Res 25(32):1\u201367","journal-title":"J Mach Learn Res"}],"container-title":["Discrete Event Dynamic Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10626-026-00439-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10626-026-00439-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10626-026-00439-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T13:19:21Z","timestamp":1775567961000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10626-026-00439-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,7]]},"references-count":52,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,12]]}},"alternative-id":["439"],"URL":"https:\/\/doi.org\/10.1007\/s10626-026-00439-8","relation":{},"ISSN":["0924-6703","1573-7594"],"issn-type":[{"value":"0924-6703","type":"print"},{"value":"1573-7594","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,7]]},"assertion":[{"value":"13 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 March 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 April 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"14"}}