{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T16:21:54Z","timestamp":1781886114234,"version":"3.54.5"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"9-10","license":[{"start":{"date-parts":[[2019,11,8]],"date-time":"2019-11-08T00:00:00Z","timestamp":1573171200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,11,8]],"date-time":"2019-11-08T00:00:00Z","timestamp":1573171200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2020,9]]},"DOI":"10.1007\/s10994-019-05849-4","type":"journal-article","created":{"date-parts":[[2019,11,8]],"date-time":"2019-11-08T18:03:00Z","timestamp":1573236180000},"page":"1699-1725","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":27,"title":["Active deep Q-learning with demonstration"],"prefix":"10.1007","volume":"109","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8734-9894","authenticated-orcid":false,"given":"Si-An","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Voot","family":"Tangkaratt","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hsuan-Tien","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masashi","family":"Sugiyama","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,11,8]]},"reference":[{"issue":"5","key":"5849_CR1","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TSMC.1983.6313077","volume":"13","author":"AG Barto","year":"1983","unstructured":"Barto, A. G., Sutton, R. S., & Anderson, C. W. (1983). Neuronlike adaptive elements that can solve difficult learning control problems. IEEE Trans Systems, Man, and Cybernetics, 13(5), 834\u2013846.","journal-title":"IEEE Trans Systems, Man, and Cybernetics"},{"key":"5849_CR2","unstructured":"Brockman, G., Cheung, V., Pettersson, L., Schneider, J., Schulman, J., Tang, J., & Zaremba, W. (2016). Openai gym. CoRR abs\/1606.01540. \narXiv:1606.01540\n\n."},{"key":"5849_CR3","unstructured":"Brys, T., Harutyunyan, A., Suay, H.B., Chernova, S., Taylor, M.E., & Now\u00e9, A. (2015). Reinforcement learning from demonstration through shaping. In IJCAI AAAI Press, pp. 3352\u20133358."},{"key":"5849_CR4","doi-asserted-by":"publisher","unstructured":"Dagan, I., & Engelson, S.P. (1995). Committee-based sampling for training probabilistic classifiers. In Machine learning, proceedings of the twelfth international conference on machine learning, Tahoe City, California, USA, July 9\u201312, 1995, pp. 150\u2013157, \nhttps:\/\/doi.org\/10.1016\/b978-1-55860-377-6.50027-x\n\n.","DOI":"10.1016\/b978-1-55860-377-6.50027-x"},{"key":"5849_CR5","unstructured":"Fortunato, M., Azar, M.G., Piot, B., Menick, J., Osband, I., Graves, A., Mnih, V., Munos, R., Hassabis, D., Pietquin, O., Blundell, C., & Legg, S. (2017). Noisy networks for exploration. CoRR abs\/1706.10295. \narXiv:1706.10295\n\n."},{"key":"5849_CR6","unstructured":"Gal, Y., Islam, R., & Ghahramani, Z. (2017). Deep bayesian active learning with image data. In Precup, D., Teh, Y. W. (eds.) Proceedings of the 34th International Conference on Machine Learning, ICML 2017, Sydney, NSW, Australia, 6\u201311 August 2017, PMLR, Proceedings of Machine Learning Research (Vol. 70, Pp. 1183\u20131192). \nhttp:\/\/proceedings.mlr.press\/v70\/gal17a.html\n\n."},{"key":"5849_CR7","first-page":"1573","volume":"16","author":"A Geramifard","year":"2015","unstructured":"Geramifard, A., Dann, C., Klein, R. H., Dabney, W., & How, J. P. (2015). Rlpy: a value-function-based reinforcement learning framework for education and research. Journal of Machine Learning Research, 16, 1573\u20131578.","journal-title":"Journal of Machine Learning Research"},{"key":"5849_CR8","unstructured":"Hester, T., Vecerik, M., Pietquin, O., Lanctot, M., Schaul, T., Piot, B., Horgan, D., Quan, J., Sendonaris, A., Osband, I., Dulac-Arnold, G., Agapiou, J., Leibo, J.Z., & Gruslys, A. (2018). Deep Q-learning from demonstrations. In McIlraith SA, Weinberger KQ (eds) AAAI, AAAI Press. Retrieved April 6, 2018, from \nhttps:\/\/www.aaai.org\/ocs\/index.php\/AAAI\/AAAI18\/paper\/view\/16976\n\n."},{"key":"5849_CR9","unstructured":"Hosu, I., Rebedea, T. (2016). Playing atari games with deep reinforcement learning and human checkpoint replay. CoRR abs\/1607.05077."},{"issue":"1","key":"5849_CR10","first-page":"3925","volume":"15","author":"K Judah","year":"2014","unstructured":"Judah, K., Fern, A. P., Dietterich, T. G., et al. (2014). Active lmitation learning: formal and practical reductions to iid learning. The Journal of Machine Learning Research, 15(1), 3925\u20133963.","journal-title":"The Journal of Machine Learning Research"},{"key":"5849_CR11","unstructured":"Kang, B., Jie, Z., Feng, J. (2018). Policy optimization with demonstrations. In Dy. J,, Krause, A. (Eds.) Proceedings of the 35th International Conference on Machine Learning, PMLR, Stockholmsm\u00e4ssan, Stockholm Sweden, Proceedings of Machine Learning Research (Vol. 80, pp. 2469\u20132478). Retrieved April 3, 2018, from \nhttp:\/\/proceedings.mlr.press\/v80\/kang18a.html\n\n."},{"key":"5849_CR12","doi-asserted-by":"publisher","unstructured":"Krawczyk, B., & Wozniak, M. (2017). Online query by committee for active learning from drifting data streams. In 2017 International Joint Conference on Neural Networks, IJCNN 2017, Anchorage, AK, USA, May 14\u201319, 2017, IEEE, pp 2120\u20132127. \nhttps:\/\/doi.org\/10.1109\/IJCNN.2017.7966111\n\n.","DOI":"10.1109\/IJCNN.2017.7966111"},{"key":"5849_CR13","unstructured":"Lipton, Z.C., Gao, J., Li, L., Li, X., Ahmed, F., & Deng, L. (2016). Efficient exploration for dialog policy learning with deep BBQ networks $${\\backslash }$$ & replay buffer spiking. CoRR abs\/1608.05081"},{"issue":"2","key":"5849_CR14","doi-asserted-by":"publisher","first-page":"203","DOI":"10.1016\/0004-3702(82)90040-6","volume":"18","author":"TM Mitchell","year":"1982","unstructured":"Mitchell, T. M. (1982). Generalization as search. Artificial Intelligence, 18(2), 203\u2013226. \nhttps:\/\/doi.org\/10.1016\/0004-3702(82)90040-6\n\n.","journal-title":"Artificial Intelligence"},{"issue":"7540","key":"5849_CR15","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A. A., Veness, J., Bellemare, M. G., et al. (2015). Human-level control through deep reinforcement learning. Nature, 518(7540), 529\u2013533. \nhttps:\/\/doi.org\/10.1038\/nature14236\n\n.","journal-title":"Nature"},{"key":"5849_CR16","unstructured":"Moore, A.W. (1990). Efficient memory-based learning for robot control. Tech. rep."},{"key":"5849_CR17","unstructured":"Osband, I., Blundell, C., Pritzel, A., Roy, B.V. (2016). Deep exploration via bootstrapped DQN. In Lee, D. D., Sugiyama, M., von Luxburg, U., Guyon, I., Garnett, R. (Eds.) NIPS (pp. 4026\u20134034). Retrieved April 7, 2018, from \nhttp:\/\/papers.nips.cc\/paper\/6501-deep-exploration-via-bootstrapped-dqn\n\n."},{"key":"5849_CR18","doi-asserted-by":"crossref","unstructured":"Piot, B., Geist, M., Pietquin, O. (2014). Boosted bellman residual minimization handling expert demonstrations. In ECML\/PKDD (2), Springer, Lecture Notes in Computer Science (vol 8725, pp. 549\u2013564).","DOI":"10.1007\/978-3-662-44851-9_35"},{"key":"5849_CR19","unstructured":"Plappert, M., Houthooft, R., Dhariwal, P., Sidor, S., Chen, R.Y., Chen, X., Asfour, T., Abbeel, P., Andrychowicz, M. (2017). Parameter space noise for exploration. CoRR abs\/1706.01905. \narXiv:1706.01905"},{"key":"5849_CR20","unstructured":"Ross, S., Gordon, G.J., Bagnell, D. (2011). A reduction of imitation learning and structured prediction to no-regret online learning. In AISTATS, JMLR.org, JMLR Proceedings (Vol.\u00a015, pp. 627\u2013635)."},{"key":"5849_CR21","unstructured":"Schaal, S. (1996). Learning from demonstration. In NIPS (pp. 1040\u20131046). MIT Press ."},{"key":"5849_CR22","unstructured":"Schaul, T., Quan, J., Antonoglou, I., Silver, D. (2016). Prioritized experience replay. In International Conference on Learning Representations. Puerto Rico"},{"key":"5849_CR23","unstructured":"Settles, B. (2009). Active learning literature survey. Computer Sciences Technical Report 1648, University of Wisconsin\u2013Madison"},{"key":"5849_CR24","unstructured":"Shon, A.P., Verma, D., Rao, R.P.N. (2007). Active imitation learning. In AAAI (pp. 756\u2013762) AAAI Press."},{"issue":"7587","key":"5849_CR25","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., Huang, A., Maddison, C. J., Guez, A., Sifre, L., van den Driessche, G., et al. (2016). Mastering the game of go with deep neural networks and tree search. Nature, 529(7587), 484\u2013489.","journal-title":"Nature"},{"key":"5849_CR26","unstructured":"Suay, H.B., Brys, T., Taylor, M.E., Chernova, S. (2016). Learning from demonstration for shaping through inverse reinforcement learning. In AAMAS (pp. 429\u2013437). ACM."},{"key":"5849_CR27","unstructured":"Subramanian, K., Jr CLI, Thomaz, A.L. (2016). Exploration from demonstration for interactive reinforcement learning. In AAMAS (pp 447\u2013456). ACM."},{"key":"5849_CR28","first-page":"3309","volume":"70","author":"W Sun","year":"2017","unstructured":"Sun, W., Venkatraman, A., Gordon, G. J., Boots, B., & Bagnell, J. A. (2017). Deeply aggrevated: Differentiable imitation learning for sequential prediction. ICML, PMLR, Proceedings of Machine Learning Research, 70, 3309\u20133318.","journal-title":"ICML, PMLR, Proceedings of Machine Learning Research"},{"key":"5849_CR29","unstructured":"Sutton, R.S. (1995). Generalization in reinforcement learning: Successful examples using sparse coarse coding. In NIPS (pp. 1038\u20131044). MIT Press"},{"key":"5849_CR30","volume-title":"Reinforcement learning\u2013an introduction. Adaptive computation and machine learning.","author":"RS Sutton","year":"1998","unstructured":"Sutton, R. S., & Barto, A. G. (1998). Reinforcement learning\u2013an introduction. Adaptive computation and machine learning. Cambridge: MIT Press."},{"key":"5849_CR31","unstructured":"Taylor, M.E., Suay, H.B., Chernova, S. (2011). Integrating reinforcement learning with human demonstrations of varying ability. In AAMAS, IFAAMAS (pp. 617\u2013624)."},{"key":"5849_CR32","unstructured":"van Hasselt, H., Guez, A., & Silver, D. (2016). Deep reinforcement learning with double q-learning. In Schuurmans D, Wellman MP (eds) Proceedings of the Thirtieth AAAI Conference on Artificial Intelligence, February 12\u201317, 2016 (pp. 2094\u20132100). Phoenix, AZ: AAAI Press. Retrieved April 11, 2018, from \nhttp:\/\/www.aaai.org\/ocs\/index.php\/AAAI\/AAAI16\/paper\/view\/12389\n\n."},{"key":"5849_CR33","unstructured":"Vecerik, M., Hester, T., Scholz, J., Wang, F., Pietquin, O., Piot, B., Heess, N., Roth\u00f6rl, T., Lampe, T., & Riedmiller, M.A. (2017). Leveraging demonstrations for deep reinforcement learning on robotics problems with sparse rewards. CoRR abs\/1707.08817. \narXiv:1707.08817\n\n."},{"key":"5849_CR34","doi-asserted-by":"crossref","unstructured":"Wang, Z., & Taylor, M.E. (2017). Improving reinforcement learning with confidence-based demonstrations. In IJCAI ijcai.org (pp 3027\u20133033).","DOI":"10.24963\/ijcai.2017\/422"},{"key":"5849_CR35","unstructured":"Watter, M., Springenberg, J.T., Boedecker, J., & Riedmiller, M.A. (2015). Embed to control: A locally linear latent dynamics model for control from raw images. In NIPS (pp 2746\u20132754)."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-019-05849-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10994-019-05849-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-019-05849-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,8]],"date-time":"2020-11-08T01:18:15Z","timestamp":1604798295000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10994-019-05849-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,8]]},"references-count":35,"journal-issue":{"issue":"9-10","published-print":{"date-parts":[[2020,9]]}},"alternative-id":["5849"],"URL":"https:\/\/doi.org\/10.1007\/s10994-019-05849-4","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,11,8]]},"assertion":[{"value":"26 November 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 June 2019","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 September 2019","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 November 2019","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}