{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,11]],"date-time":"2026-05-11T20:13:08Z","timestamp":1778530388988,"version":"3.51.4"},"reference-count":40,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T00:00:00Z","timestamp":1778803200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T00:00:00Z","timestamp":1778803200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T00:00:00Z","timestamp":1778803200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Coal-Major Project","award":["2025ZD1700800"],"award-info":[{"award-number":["2025ZD1700800"]}]},{"name":"Australian Research Council through Discovery Project","award":["DP220103881"],"award-info":[{"award-number":["DP220103881"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Internet Things J."],"published-print":{"date-parts":[[2026,5,15]]},"DOI":"10.1109\/jiot.2026.3668848","type":"journal-article","created":{"date-parts":[[2026,2,27]],"date-time":"2026-02-27T20:50:32Z","timestamp":1772225432000},"page":"22300-22313","source":"Crossref","is-referenced-by-count":0,"title":["Causally Aligned Multiagent Reinforcement Learning for Coordinated Control of Heterogeneous Home Energy Devices"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0175-7630","authenticated-orcid":false,"given":"Xiao","family":"Du","sequence":"first","affiliation":[{"name":"School of Bigdata and Software Engineering, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4041-6062","authenticated-orcid":false,"given":"Fengji","family":"Luo","sequence":"additional","affiliation":[{"name":"School of Civil Engineering, University of Sydney, Sydney, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7652-7050","authenticated-orcid":false,"given":"Juntao","family":"Hu","sequence":"additional","affiliation":[{"name":"School of Bigdata and Software Engineering, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0839-8773","authenticated-orcid":false,"given":"Wei","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Bigdata and Software Engineering, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6561-560X","authenticated-orcid":false,"given":"Junhao","family":"Wen","sequence":"additional","affiliation":[{"name":"School of Bigdata and Software Engineering, Chongqing University, Chongqing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.buildenv.2023.110435"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2022.3198401"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2021.3078462"},{"issue":"178","key":"ref4","first-page":"1","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref5","first-page":"5887","article-title":"QTRAN: Learning to factorize with transformation for cooperative multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Son"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-008-9046-9"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3403001"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.rser.2024.114648"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2020.2971427"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.scs.2023.104528"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2025.3566390"},{"key":"ref13","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/JSYST.2023.3247592"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.jobe.2023.106774"},{"key":"ref16","first-page":"1","article-title":"Integrating deep reinforcement learning into home energy management system based on soft actor-critic framework","volume-title":"Proc. ECITech; Int. Conf. Electr., Control Inf. Technol.","author":"Tan"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.apenergy.2022.120526"},{"key":"ref18","first-page":"24611","article-title":"The surprising effectiveness of PPO in cooperative, multi-agent games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yu","year":"2021"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2021.3088290"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2025.3562137"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1016\/j.enbuild.2025.115408"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-56436-4"},{"issue":"129","key":"ref23","first-page":"1","article-title":"On the approximation of cooperative heterogeneous multi-agent reinforcement learning (MARL) using mean field control (MFC)","volume":"23","author":"Subramanian","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref24","first-page":"1","article-title":"Mean-field approximation of cooperative constrained multi-agent reinforcement learning (CMARL)","author":"Mondal","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511803161"},{"key":"ref26","first-page":"9260","article-title":"Action-sufficient state representation learning for control with structural constraints","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Huang"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00788"},{"key":"ref28","first-page":"654","article-title":"More efficient off-policy evaluation through regularized targeted learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bibaut"},{"key":"ref29","first-page":"2626","article-title":"Explainable reinforcement learning through action masking","volume-title":"Proc. AAAI Conf. Artif. Intell.","author":"Madumal"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2022.3158814"},{"key":"ref31","volume-title":"Pecan Street Database","year":"2024"},{"key":"ref32","volume-title":"NOAA Data","author":"National Oceanic","year":"2024"},{"key":"ref33","article-title":"GPT-4 technical report","author":"Achiam","year":"2023","journal-title":"arXiv:2303.08774"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2012.2230637"},{"key":"ref35","first-page":"387","article-title":"Deterministic policy gradient algorithms","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Silver"},{"key":"ref36","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref37","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref38","article-title":"Decomposed soft actor-critic method for cooperative multi-agent reinforcement learning","author":"Pu","year":"2021","journal-title":"arXiv:2104.06655"},{"key":"ref39","first-page":"6379","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc. NIPS","author":"Lowe"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1103\/PhysRevE.69.066138"}],"container-title":["IEEE Internet of Things Journal"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6488907\/11513275\/11415646.pdf?arnumber=11415646","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,11]],"date-time":"2026-05-11T19:48:21Z","timestamp":1778528901000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11415646\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,15]]},"references-count":40,"journal-issue":{"issue":"10"},"URL":"https:\/\/doi.org\/10.1109\/jiot.2026.3668848","relation":{},"ISSN":["2327-4662","2372-2541"],"issn-type":[{"value":"2327-4662","type":"electronic"},{"value":"2372-2541","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,15]]}}}