{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,5]],"date-time":"2026-02-05T23:20:21Z","timestamp":1770333621138,"version":"3.49.0"},"reference-count":35,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100002507","name":"2021 Research Grant from Kangwon National University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100002507","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2022]]},"DOI":"10.1109\/access.2022.3175299","type":"journal-article","created":{"date-parts":[[2022,5,16]],"date-time":"2022-05-16T20:23:08Z","timestamp":1652732588000},"page":"63394-63402","source":"Crossref","is-referenced-by-count":2,"title":["Modular Reinforcement Learning for Playing the Game of Tron"],"prefix":"10.1109","volume":"10","author":[{"given":"Mingi","family":"Jeon","sequence":"first","affiliation":[{"name":"Department of Computer Science and Engineering, Kangwon National University, Chuncheon, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9277-7797","authenticated-orcid":false,"given":"Jay","family":"Lee","sequence":"additional","affiliation":[{"name":"Data &#x0026; Investment Division, Hana Bank, Seoul, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5406-5104","authenticated-orcid":false,"given":"Sang-Ki","family":"Ko","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, Kangwon National University, Chuncheon, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","first-page":"105","article-title":"Learning from Monte Carlo rollouts with opponent models for playing Tron","author":"knegt","year":"2018","journal-title":"Proc 10th Int Conf Agents Artif Intell Revised Sel Papers"},{"key":"ref35","article-title":"Mish: A self regularized non-monotonic neural activation function","author":"misra","year":"2020","journal-title":"Proc 31st Brit Mach Vis Conf"},{"key":"ref12","author":"sloane","year":"2011","journal-title":"Google AI Challenge Post-Mortem"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2019.2903261"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.5220\/0006536300290040"},{"key":"ref14","article-title":"ColosseumRL: A framework for multiagent reinforcement learning in n-player games","author":"shmakov","year":"2019","journal-title":"arXiv 1912 04451"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1002\/1097-0037(200010)36:3<156::AID-NET2>3.0.CO;2-L"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/0004-3702(75)90019-3"},{"key":"ref11","article-title":"Endgame detection in Tron","author":"kang","year":"2012"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1137\/0211056"},{"key":"ref10","article-title":"Modular lifelong reinforcement learning via neural composition","author":"mendez","year":"2022","journal-title":"Proc 10th Int Conf Learn Represent"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/321356.321357"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2014.6932889"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2012.6374162"},{"key":"ref19","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"2016","journal-title":"Proc 4th Int Conf Learn Represent"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ITW.2010.5593331"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1512\/iumj.1957.6.56038"},{"key":"ref23","first-page":"5279","article-title":"Scalable trust-region method for deep reinforcement learning using Kronecker-factored approximation","author":"wu","year":"2017","journal-title":"Proc Annu Conf Neural Inf Process Syst"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1137\/S0363012901385691"},{"key":"ref25","article-title":"Learning from delayed rewards","author":"watkins","year":"1989"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"key":"ref22","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume":"48","author":"mnih","year":"2016","journal-title":"Proc 33nd Int Conf Mach Learn"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-03051-4"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2996209"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-020-01758-5"},{"key":"ref29","volume":"24","author":"willem","year":"1997","journal-title":"Minimax Theorems"},{"key":"ref8","first-page":"318","article-title":"On the difficulty of modular reinforcement learning for real-world partial programming","author":"bhat","year":"2006","journal-title":"Proc 21st Nat Conf Artif Intell 18th Innov Appl Artif Intell Conf"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014975"},{"key":"ref9","first-page":"166","article-title":"Modular multitask reinforcement learning with policy sketches","volume":"70","author":"andreas","year":"2017","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref4","article-title":"Multi-agent reinforcement learning: A selective overview of theories and algorithms","author":"zhang","year":"2019","journal-title":"arXiv 1911 10635"},{"key":"ref3","first-page":"6379","article-title":"Multi-agent actor-critic for mixed cooperative-competitive environments","author":"lowe","year":"2017","journal-title":"Proc Annu Conf Neural Inf Process Syst"},{"key":"ref6","article-title":"Neural combinatorial optimization with reinforcement learning","author":"bello","year":"2017","journal-title":"Proc Workshop Track 5th Int Conf Learn Represent"},{"key":"ref5","article-title":"Solving NP-hard problems on graphs with extended AlphaGo zero","author":"abe","year":"2019","journal-title":"arXiv 1905 11623"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/9668973\/09775163.pdf?arnumber=9775163","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,20]],"date-time":"2023-10-20T22:24:56Z","timestamp":1697840696000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9775163\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"references-count":35,"URL":"https:\/\/doi.org\/10.1109\/access.2022.3175299","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]}}}