{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T02:05:35Z","timestamp":1783562735580,"version":"3.55.0"},"reference-count":80,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key R&amp;D Program of China","award":["2023YFF0905400"],"award-info":[{"award-number":["2023YFF0905400"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2341229"],"award-info":[{"award-number":["U2341229"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62206108"],"award-info":[{"award-number":["62206108"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476109"],"award-info":[{"award-number":["62476109"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007847","name":"Natural Science Foundation of Jilin Province","doi-asserted-by":"publisher","award":["20240101373JC"],"award-info":[{"award-number":["20240101373JC"]}],"id":[{"id":"10.13039\/100007847","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Jilin Province Budgetary Capital Construction Fund Plan","award":["2024C008-5"],"award-info":[{"award-number":["2024C008-5"]}]},{"DOI":"10.13039\/501100001348","name":"Agency for Science, Technology and Research","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001348","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100008629","name":"Info-communications Media Development Authority","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100008629","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Artificial Intelligence Research, Institute of High Performance Computing"},{"DOI":"10.13039\/501100001348","name":"Agency for Science, Technology and Research","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001348","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001475","name":"Nanyang Technological University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001475","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tpami.2024.3455257","type":"journal-article","created":{"date-parts":[[2024,9,6]],"date-time":"2024-09-06T17:53:09Z","timestamp":1725645189000},"page":"11392-11408","source":"Crossref","is-referenced-by-count":2,"title":["Diversifying Policies With Non-Markov Dispersion to Expand the Solution Space"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3192-8736","authenticated-orcid":false,"given":"Bohao","family":"Qu","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence, Engineering Research Center of Knowledge-Driven Human-Machine Intelligence, Ministry of Education, Jilin University, Changchun, Jilin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7391-0334","authenticated-orcid":false,"given":"Xiaofeng","family":"Cao","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Engineering Research Center of Knowledge-Driven Human-Machine Intelligence, Ministry of Education, Jilin University, Changchun, Jilin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2697-8093","authenticated-orcid":false,"given":"Yi","family":"Chang","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Engineering Research Center of Knowledge-Driven Human-Machine Intelligence, Ministry of Education, Jilin University, Changchun, Jilin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8095-4637","authenticated-orcid":false,"given":"Ivor W.","family":"Tsang","sequence":"additional","affiliation":[{"name":"Institute of High Performance Computing (IHPC) and Centre for Frontier AI Research (CFAR), Agency for Science, Technology and Research, A*STAR, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4480-169X","authenticated-orcid":false,"given":"Yew-Soon","family":"Ong","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"key":"ref3","first-page":"12 333","article-title":"DouZero: Mastering doudizhu with self-play deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zha"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i4.20394"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref6","article-title":"Dota 2 with large scale deep reinforcement learning","author":"Berner","year":"2019"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.32657\/10356\/90191"},{"key":"ref9","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref10","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-2939-8"},{"key":"ref12","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref13","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-03194-1_4"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913495721"},{"key":"ref16","first-page":"651","article-title":"Scalable deep reinforcement learning for vision-based robotic manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Kalashnikov"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919887447"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2016.00040"},{"key":"ref19","first-page":"18050","article-title":"Effective diversity in population based reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Parker-Holder"},{"key":"ref20","doi-asserted-by":"crossref","first-page":"331","DOI":"10.1016\/S0927-0507(05)80172-0","article-title":"Markov decision processes","volume":"2","author":"Puterman","year":"1990","journal-title":"Handbooks Operations Res. Manage. Sci."},{"key":"ref21","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"Puterman","year":"2014"},{"key":"ref22","first-page":"6596","article-title":"Learning non-Markovian decision-making from state-only sequences","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Qin"},{"key":"ref23","first-page":"8280","article-title":"Off-policy reinforcement learning with delayed rewards","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Han"},{"key":"ref24","first-page":"279","article-title":"Decision-making with non-Markovian rewards: From LTL to automata-based reward shaping","volume-title":"Proc. Multi-Disciplinary Conf. Reinforcement Learn. Decis. Mak.","author":"Camacho"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5814"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11572"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i10.17096"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-060117-105102"},{"key":"ref29","first-page":"1146","article-title":"Stabilising experience replay for deep multi-agent reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Foerster"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.arcontrol.2022.03.003"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(99)00052-1"},{"key":"ref32","first-page":"421","article-title":"An analysis of direct reinforcement learning in non-Markovian domains","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Pendrith"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/0004-3702(94)00012-P"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref35","first-page":"1160","article-title":"Rewarding behaviors","volume-title":"Proc. Conf. Assoc. Advance. Artif. Intell.","author":"Bacchus"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1609\/socs.v8i1.18421"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2017.8206234"},{"key":"ref38","first-page":"452","article-title":"Teaching multiple tasks to an RL agent using LTL","volume-title":"Proc. 17th Int. Conf. Auton. Agents MultiAgent Syst.","author":"Toro Icarte"},{"key":"ref39","first-page":"2107","article-title":"Using reward machines for high-level task specification and decomposition in reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Icarte"},{"key":"ref40","first-page":"112","article-title":"Structured solution methods for non-Markovian decision processes","volume-title":"Proc. Conf. Assoc. Advance. Artif. Intell.","author":"Bacchus"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1676"},{"key":"ref42","first-page":"15523","article-title":"Learning reward machines for partially observable reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Toro Icarte"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2017.2704781"},{"key":"ref44","article-title":"Harnessing distribution ratio estimators for learning agents with quality and diversity","author":"Gangwani","year":"2020"},{"key":"ref45","first-page":"10 510","article-title":"Diversity-driven exploration strategy for deep reinforcement learning","volume-title":"Proc. 32nd Int. Conf. Neural Inf. Process. Syst.","author":"Hong"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/821"},{"key":"ref47","article-title":"Non-local policy optimization via diversity-regularized collaborative exploration","author":"Peng","year":"2020"},{"key":"ref48","first-page":"7483","article-title":"Learning novel policies for tasks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang"},{"key":"ref49","article-title":"Improving exploration in evolution strategies for deep reinforcement learning via a population of novelty-seeking agents","author":"Conti","year":"2017"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1038\/nature14422"},{"key":"ref51","article-title":"Illuminating search spaces by mapping elites","author":"Mouret","year":"2015"},{"key":"ref52","first-page":"753","article-title":"Ridge rider: Finding diverse solutions by following eigenvectors of the Hessian","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Parker-Holder"},{"key":"ref53","article-title":"Modelling behavioural diversity for learning in open-ended games","author":"Nieves","year":"2021"},{"key":"ref54","article-title":"Novel policy seeking with constrained optimization","author":"Sun","year":"2020"},{"key":"ref55","article-title":"Diversity is all you need: Learning skills without a reward function","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Eysenbach"},{"key":"ref56","article-title":"Dynamical distance learning for semi-supervised and unsupervised skill discovery","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hartikainen"},{"key":"ref57","first-page":"1039","article-title":"gep-pg: Decoupling exploration and exploitation in deep reinforcement learning algorithms","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Colas"},{"key":"ref58","article-title":"Intrinsically motivated goal exploration processes with automatic curriculum learning","author":"Forestier","year":"2017"},{"key":"ref59","article-title":"Continuously discovering novel strategies via reward-switching policy optimization","author":"Zhou","year":"2022"},{"key":"ref60","article-title":"Population-guided parallel policy search for reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Jung"},{"key":"ref61","first-page":"1802","article-title":"Learning policy representations in multiagent systems","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Grover"},{"key":"ref62","article-title":"Meta-learning with latent embedding optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Rusu"},{"key":"ref63","first-page":"3556","article-title":"Representations for stable off-policy reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ghosh"},{"key":"ref64","article-title":"Relational forward models for multi-agent learning","author":"Tacchetti","year":"2018"},{"key":"ref65","article-title":"Deep interactive Bayesian reinforcement learning via meta-learning","author":"Zintgraf","year":"2021"},{"key":"ref66","first-page":"4218","article-title":"Machine theory of mind","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Rabinowitz"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/766"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1176"},{"key":"ref69","article-title":"Adaptive input representations for neural language modeling","author":"Baevski","year":"2018"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1561\/2200000044"},{"key":"ref71","first-page":"171","article-title":"\u00c9tude sur les propri\u00e9t\u00e9s des fonctions enti\u00e8res et en particulier d\u2019une fonction consid\u00e9r\u00e9e par riemann","volume":"9","author":"Hadamard","year":"1893","journal-title":"J. de math\u00e9matiques pures et appliqu\u00e9es"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.2307\/2331979"},{"key":"ref73","volume-title":"An Introduction to Multivariate Statistical Analysis","author":"Anderson","year":"1962"},{"key":"ref74","volume-title":"Applied Multivariate Statistical Analysis","volume":"6","author":"Johnson","year":"2014"},{"key":"ref75","first-page":"749","article-title":"Uber die abgrenzung der eigenwerte einer matrix","volume":"7","author":"Gerschgorin","year":"1931","journal-title":"lzv. Akad. Nauk. USSR. Otd. Fiz-Mat. Nauk"},{"key":"ref76","article-title":"OpenAI gym","author":"Brockman","year":"2016"},{"key":"ref77","article-title":"QD-RL: Efficient mixing of quality and diversity in reinforcement learning","author":"Cideron","year":"2020"},{"key":"ref78","first-page":"351","article-title":"Playing atari with deep reinforcement learning","volume":"21","author":"Mnih","year":"2013","journal-title":"Comput. Sci."},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.5555\/3291168.3291210"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/10746266\/10668823.pdf?arnumber=10668823","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:34:33Z","timestamp":1732667673000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10668823\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":80,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3455257","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}