{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T08:03:34Z","timestamp":1781597014218,"version":"3.54.5"},"reference-count":66,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2022,11,1]],"date-time":"2022-11-01T00:00:00Z","timestamp":1667260800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,11,1]],"date-time":"2022-11-01T00:00:00Z","timestamp":1667260800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,11,1]],"date-time":"2022-11-01T00:00:00Z","timestamp":1667260800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2020AAA0107400"],"award-info":[{"award-number":["2020AAA0107400"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["12071145"],"award-info":[{"award-number":["12071145"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shanghai Municipal Science and Technology Major Project","award":["2021SHZDZX0102"],"award-info":[{"award-number":["2021SHZDZX0102"]}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["18DZ2270700"],"award-info":[{"award-number":["18DZ2270700"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["20511101100"],"award-info":[{"award-number":["20511101100"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Open Research Projects of Zhejiang Lab","award":["2021KE0AB03"],"award-info":[{"award-number":["2021KE0AB03"]}]},{"name":"Shenzhen Institute of Artificial Intelligence and Robotics for Society"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2022,11,1]]},"DOI":"10.1109\/tpami.2021.3102140","type":"journal-article","created":{"date-parts":[[2021,8,4]],"date-time":"2021-08-04T20:34:05Z","timestamp":1628109245000},"page":"8618-8634","source":"Crossref","is-referenced-by-count":25,"title":["Structured Cooperative Reinforcement Learning With Time-Varying Composite Action Space"],"prefix":"10.1109","volume":"44","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2985-1098","authenticated-orcid":false,"given":"Wenhao","family":"Li","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology and the MOE Key Laboratory for Advanced Theory and Application in Statistics and Data Science, East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3064-5128","authenticated-orcid":false,"given":"Xiangfeng","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology and the MOE Key Laboratory for Advanced Theory and Application in Statistics and Data Science, East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Jin","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology and the MOE Key Laboratory for Advanced Theory and Application in Statistics and Data Science, East China Normal University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dijun","family":"Luo","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Tencent Inc, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongyuan","family":"Zha","sequence":"additional","affiliation":[{"name":"School of Data Science and AIRS, Chinese University of Hong Kong, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref2","article-title":"Hyper-meta reinforcement learning with sparse reward","author":"Hua","year":"2020"},{"key":"ref3","article-title":"Skew-Fit: State-covering self-supervised reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Pong"},{"key":"ref4","article-title":"The ingredients of real-world robotic reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhu"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2966414"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013598"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00941"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1613\/jair.3912"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ADPRL.2011.5967381"},{"key":"ref10","first-page":"1185","article-title":"Generalized value functions for large action sets","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Pazis"},{"key":"ref11","article-title":"Deep reinforcement learning in large discrete action spaces","author":"Dulac-Arnold","year":"2015"},{"key":"ref12","article-title":"Efficient entropy for policy gradient with multidimensional action space","author":"Zhang","year":"2018"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10226"},{"key":"ref14","article-title":"Hierarchical approaches for reinforcement learning in parameterized action space","author":"Wei","year":"2018"},{"key":"ref15","article-title":"Parametrized deep Q-networks learning: Reinforcement learning with discrete-continuous hybrid action space","author":"Xiong","year":"2018"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/316"},{"key":"ref17","article-title":"Growing action spaces","author":"Farquhar","year":"2019"},{"key":"ref18","article-title":"Learning action-transferable policy with action embedding","author":"Chen","year":"2019"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2019.xv.001"},{"key":"ref20","article-title":"Generalization to new actions in reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Jain"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5739"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919872545"},{"key":"ref23","article-title":"F2A2: Flexible fully-decentralized approximate actor-critic for cooperative multi-agent reinforcement learning","author":"Li","year":"2020"},{"key":"ref24","article-title":"EMI: Exploration with mutual information","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kim"},{"key":"ref25","first-page":"1","article-title":"Towards a neural statistician","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Edwards"},{"key":"ref26","article-title":"Neural machine translation by jointly learning to align and translate","author":"Bahdanau","year":"2014"},{"key":"ref27","first-page":"1","article-title":"Graph attention networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Veli\u010dkovi\u0107"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.18356\/27095525"},{"key":"ref29","article-title":"Artificial intelligence for social good: A survey","author":"Shi","year":"2020"},{"key":"ref30","first-page":"1751","article-title":"Crops selection for optimal soil planning using multiobjective evolutionary algorithms","volume-title":"Proc. AAAI Conf. Artif. Intell.","author":"L\u00fccken"},{"key":"ref31","article-title":"Microsoft and icrisats intelligent cloud pilot for agriculture in andhra pradesh increase crop yield for farmers","year":"2017"},{"key":"ref32","first-page":"2819","article-title":"Estimating reference evapotranspiration for irrigation management in the texas high plains","volume-title":"Proc. Int. Joint Conf. Artif. Intell.","author":"Holman"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.11172"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v25i1.7811"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3314344.3332485"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3287098.3287100"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/11527862_14"},{"key":"ref39","article-title":"Deep reinforcement learning in parameterized action space","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hausknecht"},{"key":"ref40","first-page":"2944","article-title":"Learning continuous control policies by stochastic value gradients","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Heess"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1153"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-42911-3_48"},{"key":"ref43","article-title":"The natural language of actions","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tennenholtz"},{"key":"ref44","article-title":"Relational forward models for multi-agent learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Tacchetti"},{"key":"ref45","article-title":"Relational inductive biases, deep learning, and graph networks","author":"Battaglia","year":"2018"},{"key":"ref46","article-title":"Deep multi-agent reinforcement learning with relevance graphs","author":"Malysheva","year":"2018"},{"key":"ref47","first-page":"1","article-title":"Graph convolutional reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jiang"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6214"},{"key":"ref49","article-title":"Pic: Permutation invariant critic for multi-agent deep reinforcement learning","volume-title":"Proc. 3rd Conf. Robot Learn.","author":"Liu"},{"key":"ref50","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lowe"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6211"},{"key":"ref52","first-page":"709","article-title":"Dynamic programming for partially observable stochastic games","volume-title":"Proc. AAAI Conf. Artif. Intell.","author":"Hansen"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6216"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2008.2005141"},{"key":"ref55","article-title":"Gated graph sequence neural networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00794"},{"key":"ref57","article-title":"Learning implicit credit assignment for multi-agent actor-critic","volume-title":"Proc. Conf. Neural Inf. Process. Syst.","author":"Zhou"},{"key":"ref58","article-title":"All learning is local: Multi-agent learning in global reward games","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chang"},{"key":"ref59","article-title":"Value-decomposition networks for cooperative multi-agent learning","author":"Sunehag","year":"2017"},{"key":"ref60","article-title":"Qmix: Monotonic value function factorisation for deep multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Rashid"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref62","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref63","article-title":"Actor-attention-critic for multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Iqbal"},{"key":"ref64","article-title":"Hypernetworks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ha"},{"key":"ref65","first-page":"1","article-title":"Dropedge: Towards deep graph convolutional networks on node classification","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Rong"},{"key":"ref66","article-title":"Actor-attention-critic for multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Iqbal"},{"key":"ref67","article-title":"Towards understanding linear value decomposition in cooperative multi-agent Q-learning","author":"Wang","year":"2020"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/34\/9910243\/09507301.pdf?arnumber=9507301","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,11]],"date-time":"2024-01-11T23:23:51Z","timestamp":1705015431000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9507301\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,1]]},"references-count":66,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2021.3102140","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,11,1]]}}}