{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T05:32:16Z","timestamp":1730266336283,"version":"3.28.0"},"reference-count":31,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T00:00:00Z","timestamp":1658102400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,7,18]]},"DOI":"10.1109\/ijcnn55064.2022.9892633","type":"proceedings-article","created":{"date-parts":[[2022,9,30]],"date-time":"2022-09-30T19:56:04Z","timestamp":1664567764000},"page":"1-8","source":"Crossref","is-referenced-by-count":2,"title":["Q-Value Weighted Regression: Reinforcement Learning with Limited Data"],"prefix":"10.1109","author":[{"given":"Piotr","family":"Kozakowski","sequence":"first","affiliation":[{"name":"Faculty of Mathematics, Informatics and Mechanics University of Warsaw,Warsaw,Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lukasz","family":"Kaiser","sequence":"additional","affiliation":[{"name":"OpenAI,San Francisco,United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Henryk","family":"Michalewski","sequence":"additional","affiliation":[{"name":"Google,Warsaw,Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Afroz","family":"Mohiuddin","sequence":"additional","affiliation":[{"name":"Google,Mountain View,United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Katarzyna","family":"Kanska","sequence":"additional","affiliation":[{"name":"Google,Warsaw,Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref31","first-page":"449","article-title":"A distributional perspective on reinforcement learning","volume":"70","author":"bellemare","year":"0","journal-title":"Proceedings of the 34th International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"key":"ref30","article-title":"New method of stochastic approximation type","volume":"1990","author":"polyak","year":"1990","journal-title":"Automation and Remote Control"},{"key":"ref10","article-title":"When to use parametric models in reinforcement learning?","volume":"32","author":"van hasselt","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref12","article-title":"Intriguing properties of neural networks","author":"szegedy","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref13","first-page":"6240","article-title":"Spectrally-normalized margin bounds for neural networks","volume":"30","author":"bartlett","year":"2017","journal-title":"Advances in neural information processing systems"},{"journal-title":"Understanding deep learning requires rethinking generalization","year":"2017","author":"zhang","key":"ref14"},{"journal-title":"Reinforcement Learning An Introduction","year":"2018","author":"sutton","key":"ref15"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref17","article-title":"Policy gradient methods for reinforcement learning with function approximation","volume":"12","author":"sutton","year":"2000","journal-title":"Advances in neural information processing systems"},{"key":"ref18","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume":"48","author":"mnih","year":"2016","journal-title":"Proceedings of The 33rd International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"key":"ref19","article-title":"Continuous control with deep reinforcement learning","author":"lillicrap","year":"2016","journal-title":"ICLR (Poster)"},{"key":"ref28","article-title":"Model based reinforcement learning for atari","author":"lukasz","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1177\/0278364919887447"},{"key":"ref27","article-title":"Data-efficient reinforcement learning with self-predictive representations","author":"schwarzer","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2017.XIII.034"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref5","first-page":"262","article-title":"Sim-to-real robot learning from pixels with progressive nets","volume":"78","author":"rusu","year":"2017","journal-title":"Proceedings of the 1st Annual Conference on Robot Learning ser Proceedings of Machine Learning Research"},{"key":"ref8","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume":"80","author":"haarnoja","year":"0","journal-title":"Proceedings of the 35th International Conference on Machine Learning ser Proceedings of Machine Learning Research"},{"journal-title":"Proximal policy optimization algorithms","year":"2017","author":"schulman","key":"ref7"},{"journal-title":"Starcraft ii A new challenge for reinforcement learning","year":"2017","author":"vinyals","key":"ref2"},{"journal-title":"Advantage-weighted regression Simple and scalable off-policy reinforcement learning","year":"2019","author":"peng","key":"ref9"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1126\/science.aar6404"},{"journal-title":"Reinforcement Learning and Control as Probabilistic Inference Tutorial and Review","year":"2018","author":"levine","key":"ref20"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10269"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/1273496.1273590"},{"journal-title":"AWAC Accelerating Online Reinforcement Learning With Offline Datasets","year":"2020","author":"nair","key":"ref24"},{"key":"ref23","article-title":"Fitted q-iteration by advantage weighted regression","volume":"21","author":"neumann","year":"2009","journal-title":"Advances in neural information processing systems"},{"journal-title":"Offline reinforcement learning Tutorial review and perspectives on open problems","year":"2020","author":"levine","key":"ref26"},{"key":"ref25","first-page":"7768","article-title":"Critic regularized regression","volume":"33","author":"wang","year":"2020","journal-title":"Advances in neural information processing systems"}],"event":{"name":"2022 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2022,7,18]]},"location":"Padua, Italy","end":{"date-parts":[[2022,7,23]]}},"container-title":["2022 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9891857\/9889787\/09892633.pdf?arnumber=9892633","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,3]],"date-time":"2022-11-03T23:00:34Z","timestamp":1667516434000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9892633\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,7,18]]},"references-count":31,"URL":"https:\/\/doi.org\/10.1109\/ijcnn55064.2022.9892633","relation":{},"subject":[],"published":{"date-parts":[[2022,7,18]]}}}