{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:56:35Z","timestamp":1784300195651,"version":"3.55.0"},"reference-count":173,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2020,3,1]],"date-time":"2020-03-01T00:00:00Z","timestamp":1583020800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,3,1]],"date-time":"2020-03-01T00:00:00Z","timestamp":1583020800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,3,1]],"date-time":"2020-03-01T00:00:00Z","timestamp":1583020800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Elite Research travel"},{"name":"The Danish Ministry for Higher Education and Science"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Games"],"published-print":{"date-parts":[[2020,3]]},"DOI":"10.1109\/tg.2019.2896986","type":"journal-article","created":{"date-parts":[[2019,2,13]],"date-time":"2019-02-13T20:31:37Z","timestamp":1550089897000},"page":"1-20","source":"Crossref","is-referenced-by-count":166,"title":["Deep Learning for Video Game Playing"],"prefix":"10.1109","volume":"12","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5381-5498","authenticated-orcid":false,"given":"Niels","family":"Justesen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1011-2976","authenticated-orcid":false,"given":"Philip","family":"Bontrager","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3128-4598","authenticated-orcid":false,"given":"Julian","family":"Togelius","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3607-8400","authenticated-orcid":false,"given":"Sebastian","family":"Risi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref170","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2014.2339221"},{"key":"ref172","article-title":"A deep compositional framework for human-like language acquisition in virtual environment","author":"yu","year":"2017"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-63519-4"},{"key":"ref173","first-page":"3562","article-title":"Learn what not to learn: Action elimination with deep reinforcement learning","author":"zahavy","year":"2018","journal-title":"Advances in Neural IInformation Processing Systems"},{"key":"ref168","author":"wymann","year":"2000"},{"key":"ref169","article-title":"Player modeling","volume":"6","author":"yannakakis","year":"2013","journal-title":"Dagstuhl Follow-Ups"},{"key":"ref39","author":"goodfellow","year":"2016","journal-title":"Deep Learning"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-009-9112-y"},{"key":"ref33","first-page":"1146","article-title":"Stabilising experience replay for deep multi-agent reinforcement learning","author":"foerster","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref32","first-page":"2974","article-title":"Counterfactual multi-agent policy gradients","author":"foerster","year":"0","journal-title":"Proc 32nd AAAI Conf Artif Intell"},{"key":"ref31","article-title":"PathNet: Evolution channels gradient descent in super neural networks","author":"fernando","year":"0"},{"key":"ref30","first-page":"1407","article-title":"IMPALA: Scalable distributed deep-RL with importance weighted actor-learner architectures","author":"espeholt","year":"2018","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/144"},{"key":"ref36","article-title":"Learning visual predictive models of physics for playing billiards","author":"fragkiadaki","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref35","article-title":"Noisy networks for exploration","author":"fortunato","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref34","article-title":"Learning to communicate to solve riddles with deep distributed recurrent q-networks","author":"foerster","year":"2016"},{"key":"ref28","article-title":"Learning to act by predicting the future","author":"dosovitskiy","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ACC.2012.6315022"},{"key":"ref29","first-page":"1329","article-title":"Benchmarking deep reinforcement learning for continuous control","author":"duan","year":"0","journal-title":"Proc 33rd Int Conf Mach Learn"},{"key":"ref20","article-title":"Transfer deep reinforcement learning in 3D environments: An empirical study","author":"chaplot","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref22","first-page":"1281","article-title":"Intrinsically motivated reinforcement learning","author":"chentanez","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.312"},{"key":"ref24","first-page":"5032","article-title":"Improving exploration in evolution strategies for deep reinforcement learning via a population of novelty-seeking agents","author":"conti","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref23","first-page":"4302","article-title":"Deep reinforcement learning from human preferences","author":"christiano","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref101","first-page":"1","article-title":"Language understanding for textbased games using deep reinforcement learning","author":"narasimhan","year":"0","journal-title":"Proc Conf Empirical Methods Natural Lang Process"},{"key":"ref26","article-title":"Playing Atari with six neurons","author":"cuccu","year":"2018"},{"key":"ref100","article-title":"Massively parallel methods for deep reinforcement learning","author":"nair","year":"2015"},{"key":"ref25","article-title":"TextWorld: A learning environment for text-based games","author":"c\u00f4t\u00e9","year":"2018"},{"key":"ref50","article-title":"Half field offense: An environment for multiagent learning and ad hoc teamwork","author":"hausknecht","year":"0","journal-title":"The AAMAS Workshop on Adaptive and Learning Agents"},{"key":"ref51","first-page":"29","article-title":"Deep recurrent q-learning for partially observable MDPs","author":"hausknecht","year":"0","journal-title":"Proc AAAI Fall Symp Sequential Decis Making Intell Agents"},{"key":"ref154","article-title":"Episodic exploration for deep deterministic policies: An application to StarCraft micromanagement tasks","author":"usunier","year":"2016"},{"key":"ref153","first-page":"30","article-title":"Ontogenetic and phylogenetic reinforcement learning","volume":"23","author":"togelius","year":"2009","journal-title":"K&#x00FC;nstliche Intell"},{"key":"ref156","first-page":"5396","article-title":"Hybrid reward architecture for reinforcement learning","author":"van seijen","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref155","first-page":"2094","article-title":"Deep reinforcement learning with double q-learning","author":"van hasselt","year":"0","journal-title":"Proc 30th AAAI Conf Artif Intell"},{"key":"ref150","first-page":"1553","article-title":"A deep hierarchical approach to lifelong learning in minecraft","author":"tessler","year":"2017","journal-title":"Proc 31st AAAI Conf Artif Intell"},{"key":"ref152","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref151","first-page":"2656","article-title":"ELF: An extensive, lightweight and flexible research platform for real-time strategy games","author":"tian","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0172395"},{"key":"ref147","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"ref148","doi-asserted-by":"publisher","DOI":"10.1109\/ICIST.2018.8426160"},{"key":"ref149","first-page":"4497","article-title":"Distral: Robust multitask reinforcement learning","author":"teh","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-32323-2"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ITW.2010.5593336"},{"key":"ref57","first-page":"3223","article-title":"Deep q-learning from demonstrations","author":"hester","year":"0","journal-title":"Proc 32nd AAAI Conf Artif Intell"},{"key":"ref56","article-title":"Rainbow: Combining improvements in deep reinforcement learning","author":"hessel","year":"0","journal-title":"Proc AAAI"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1153"},{"key":"ref53","article-title":"On-policy vs. off-policy updates for deep reinforcement learning","author":"hausknecht","year":"0","journal-title":"the Workshop on Deep Reinforcement Learning Frontiers and Challenges in IJCAI' 16"},{"key":"ref52","article-title":"Deep reinforcement learning in parameterized action space","author":"hausknecht","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref40","article-title":"Neural turing machines","author":"graves","year":"2014"},{"key":"ref167","article-title":"Training agent for first-person shooter game with actor-critic curriculum learning","author":"wu","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref166","first-page":"5285","article-title":"Scalable trust-region method for deep reinforcement learning using kronecker-factored approximation","author":"wu","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref165","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1989.1.2.270"},{"key":"ref164","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022672621406"},{"key":"ref163","doi-asserted-by":"publisher","DOI":"10.1177\/105971239700600202"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022676722315"},{"key":"ref161","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","author":"wang","year":"0","journal-title":"Proc 33rd Int Conf Mach Learn"},{"key":"ref160","article-title":"Sample efficient actor-critic with experience replay","author":"wang","year":"2017"},{"key":"ref4","first-page":"9","article-title":"Combining strategic learning and tactical search in real-time strategy games","author":"barriga","year":"0","journal-title":"Proc AAAI Conf Artif Intell Interactive Digit Entertain"},{"key":"ref3","first-page":"9","article-title":"Overview of RoboCup-98","volume":"21","author":"asada","year":"2000","journal-title":"AI Mag"},{"key":"ref6","article-title":"DeepMind lab","author":"beattie","year":"2016"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1023\/A:1025696116075"},{"key":"ref159","first-page":"114","article-title":"Portfolio online evolution in StarCraft","author":"wang","year":"0","journal-title":"Proc Artif Intell Interactive Digit Entertainment Conf"},{"key":"ref8","first-page":"1471","article-title":"Unifying count-based exploration and intrinsic motivation","author":"bellemare","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2013.2294713"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1613\/jair.3912"},{"key":"ref157","article-title":"Starcraft II: A new challenge for reinforcement learning","author":"vinyals","year":"2017"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1145\/3205455.3205517"},{"key":"ref9","first-page":"449","article-title":"A distributional perspective on reinforcement learning","author":"bellemare","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref46","first-page":"2613","article-title":"Double q-learning","author":"hasselt","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref45","first-page":"2450","article-title":"Recurrent world models facilitate policy evolution","author":"ha","year":"2018","journal-title":"Advances in Neural IInformation Processing Systems"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2009.2038365"},{"key":"ref47","first-page":"164","article-title":"Second order derivatives for network pruning: Optimal brain surgeon","author":"hassibi","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref42","first-page":"3338","article-title":"Deep learning for real-time Atari game play using offline Monte-Carlo tree search planning","author":"guo","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2018.8490442"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/518"},{"key":"ref43","first-page":"1379","article-title":"Dynamic network surgery for efficient DNNs","author":"guo","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2011.6032024"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2014.09.003"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.1109\/TAMD.2010.2056368"},{"key":"ref124","article-title":"Measuring intelligence through games","author":"schaul","year":"2011"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2016.7860433"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1145\/3071178.3071303"},{"key":"ref129","article-title":"Proximal policy optimization algorithms","author":"schulman","year":"2017"},{"key":"ref71","article-title":"Beating Atari with natural language guided reinforcement learning","author":"kaplan","year":"2017"},{"key":"ref128","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref70","first-page":"1809","article-title":"Schema networks: Zero-shot transfer with a generative causal model of intuitive physics","author":"kansky","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1145\/2463372.2463509"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-42716-4"},{"key":"ref77","first-page":"3675","article-title":"Hierarchical deep reinforcement learning: Integrating temporal abstraction and intrinsic motivation","author":"kulkarni","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1611835114"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2017.8080433"},{"key":"ref133","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of go with deep neural networks and tree search","volume":"529","author":"silver","year":"2016","journal-title":"Nature"},{"key":"ref134","first-page":"387","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"0","journal-title":"Proc 31st Int Conf Mach Learn"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2013.6633634"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/CEC.2017.7969556"},{"key":"ref132","article-title":"Loss is its own reward: Self-supervision for reinforcement learning","author":"shelhamer","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref79","first-page":"2140","article-title":"Playing FPS games with deep reinforcement learning","author":"lample","year":"0","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2016.7860439"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2016.7860436"},{"key":"ref138","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45603-1_22"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1109\/TEVC.2005.856210"},{"key":"ref60","article-title":"Generative agents for player decision modeling in games","author":"holmg\u00e5rd","year":"0","journal-title":"Proc Int Conf Found Digit Games"},{"key":"ref139","first-page":"145","article-title":"Deep neuroevolution: Genetic algorithms are a competitive alternative for training deep neural networks for reinforcement learning","author":"such","year":"0","journal-title":"Proc 11th Annu Conf Genetic Evol Comput"},{"key":"ref62","article-title":"SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and $< $ 0.5 MB model size","author":"iandola","year":"2016"},{"key":"ref61","article-title":"Distributed prioritized experience replay","author":"horgan","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref63","article-title":"Reinforcement learning with unsupervised auxiliary tasks","author":"jaderberg","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2016.2535416"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2018.2846639"},{"key":"ref65","first-page":"4246","article-title":"The Malmo platform for artificial intelligence experimentation","author":"johnson","year":"0","journal-title":"Proc Int Joint Conf Artif Intell"},{"key":"ref141","article-title":"TStarBots: Defeating the cheating level builtin AI in StarCraft II in the full game","author":"sun","year":"2018"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-31204-0_38"},{"key":"ref142","volume":"1","author":"sutton","year":"1998","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref67","first-page":"187","article-title":"Continual online evolution for in-game build order adaptation in StarCraft","author":"justesen","year":"0","journal-title":"Proc Genetic Evol Comput Conf"},{"key":"ref143","first-page":"1057","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume":"99","author":"sutton","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2017.8080430"},{"key":"ref144","author":"sweetser","year":"2008","journal-title":"Emergence in Games"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"ref69","article-title":"Illuminating generalization in deep reinforcement learning through procedural level generation","author":"justesen","year":"0","journal-title":"Proc Deep Reinforcement Learn Workshop"},{"key":"ref145","article-title":"TorchCraft: A library for machine learning research on real-time strategy games","author":"synnaeve","year":"2016"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2017.8080408"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2012.2184109"},{"key":"ref95","article-title":"Learning to navigate in complex environments","author":"mirowski","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref108","article-title":"Actor-mimic: Deep multitask and transfer reinforcement learning","author":"parisotto","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref94","article-title":"Efficient estimation of word representations in vector space","author":"mikolov","year":"2013"},{"key":"ref107","first-page":"2721","article-title":"Count-based exploration with neural density models","author":"ostrovski","year":"0","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref93","first-page":"155","article-title":"Computational intelligence in games","author":"miikkulainen","year":"2006","journal-title":"Computational Intelligence Principles and Practice"},{"key":"ref106","first-page":"4026","article-title":"Deep exploration via bootstrapped DQN","author":"osband","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2019.2934906"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1016\/j.entcom.2012.10.001"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1613\/jair.5699"},{"key":"ref104","first-page":"58","article-title":"The combinatorial multi-armed bandit problem and its application to real-time strategy games","author":"ontan\u00f3n","year":"0","journal-title":"Proc AAAI Conf Artif Intell Interactive Digit Entertain"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/MCI.2006.1597057"},{"key":"ref103","first-page":"2863","article-title":"Action-conditional video prediction using deep networks in Atari games","author":"oh","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref102","first-page":"2790","article-title":"Control of memory, active perception, and action in Minecraft","volume":"48","author":"oh","year":"0","journal-title":"Proc 33rd Int Conf Mach Learn"},{"key":"ref111","article-title":"Multiagent bidirectionally-coordinated nets for learning to play StarCraft combat games","author":"peng","year":"2017"},{"key":"ref112","article-title":"Observe and look further: Achieving consistent performance on Atari","author":"pohlen","year":"2018"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.70"},{"key":"ref98","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref99","article-title":"Learning and game AI","volume":"6","author":"mu\u00f1oz-avila","year":"2013","journal-title":"Dagstuhl Follow-Ups"},{"key":"ref96","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref97","article-title":"Playing Atari with deep reinforcement learning","author":"mnih","year":"0","journal-title":"Proc Int Conf Neural Inf Process Syst Deep Learn Workshop"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1613\/jair.3912"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553380"},{"key":"ref12","first-page":"-115i","article-title":"Making a science of model search: Hyperparameter optimization in hundreds of dimensions for vision architectures","author":"bergstra","year":"0","journal-title":"Proc 30th Int Conf Int Conf Mach Learn"},{"key":"ref13","first-page":"2546","article-title":"Algorithms for hyper-parameter optimization","author":"bergstra","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref14","article-title":"Playing doom with SLAM-augmented deep reinforcement learning","author":"bhatti","year":"2016"},{"key":"ref15","article-title":"Playing SNES in the retro learning environment","author":"bhonker","year":"2017"},{"key":"ref118","volume":"37","author":"rummery","year":"1994","journal-title":"On-line Q-learning using connectionist systems"},{"key":"ref16","article-title":"Deep apprenticeship learning for playing video games","author":"bogdanovic","year":"0","journal-title":"Proc 29th AAAI Conf Artif Intell"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1989.1.4.541"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/5236.003.0018"},{"key":"ref17","article-title":"OpenAI Gym","author":"brockman","year":"2016"},{"key":"ref81","doi-asserted-by":"crossref","first-page":"436","DOI":"10.1038\/nature14539","article-title":"Deep learning","volume":"521","author":"lecun","year":"2015","journal-title":"Nature"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2012.2186810"},{"key":"ref84","first-page":"329","article-title":"Exploiting open-endedness to solve problems through the search for novelty","author":"lehman","year":"0","journal-title":"Proc 11th Int Conf Artif Life"},{"key":"ref119","article-title":"Policy distillation","author":"rusu","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref19","first-page":"807","article-title":"All learning is local: Multi-agent learning in global reward games","author":"chang","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1007\/s11023-007-9079-x"},{"key":"ref114","first-page":"63","article-title":"Combining search-based procedural content generation and social gaming in the Petalz video game","author":"risi","year":"0","journal-title":"Proc AAAI Conf Artif Intell Interactive Digit Entertain"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2017.8080444"},{"key":"ref116","first-page":"1","article-title":"Deep reinforcement learning for general video game AI","author":"rodriguez torrado","year":"0","journal-title":"Proc IEEE Conf Comput Intell Games"},{"key":"ref80","author":"le cun","year":"1987"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2015.2494596"},{"key":"ref120","article-title":"Progressive neural networks","author":"rusu","year":"2016"},{"key":"ref89","article-title":"Reinforcement learning for robots using neural networks","author":"lin","year":"1993"},{"key":"ref121","article-title":"Evolution strategies as a scalable alternative to reinforcement learning","author":"salimans","year":"2017"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2013.6633610"},{"key":"ref123","article-title":"Prioritized experience replay","author":"schaul","year":"0","journal-title":"Proc Int Conf Learn Represent"},{"key":"ref85","first-page":"464","article-title":"Multi-agent reinforcement learning in sequential social dilemmas","author":"leibo","year":"0","journal-title":"Proc 16th Conf Auton Agents MultiAgent Syst"},{"key":"ref86","first-page":"430","article-title":"Learning physical intuition of block towers by example","author":"lerer","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref87","article-title":"Deep reinforcement learning: An overview","author":"li","year":"2017"},{"key":"ref88","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"0","journal-title":"Proc Int Conf Learn Represent"}],"container-title":["IEEE Transactions on Games"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7782673\/9039767\/08632747.pdf?arnumber=8632747","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T12:56:59Z","timestamp":1651064219000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8632747\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,3]]},"references-count":173,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tg.2019.2896986","relation":{},"ISSN":["2475-1502","2475-1510"],"issn-type":[{"value":"2475-1502","type":"print"},{"value":"2475-1510","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,3]]}}}