{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T19:27:11Z","timestamp":1740166031752,"version":"3.37.3"},"reference-count":24,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2020,2,27]],"date-time":"2020-02-27T00:00:00Z","timestamp":1582761600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,2,27]],"date-time":"2020-02-27T00:00:00Z","timestamp":1582761600000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Prog Artif Intell"],"published-print":{"date-parts":[[2020,6]]},"DOI":"10.1007\/s13748-020-00204-4","type":"journal-article","created":{"date-parts":[[2020,2,27]],"date-time":"2020-02-27T21:02:26Z","timestamp":1582837346000},"page":"155-169","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Coaching: accelerating reinforcement learning through human-assisted approach"],"prefix":"10.1007","volume":"9","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7659-3517","authenticated-orcid":false,"given":"Nakarin","family":"Suppakun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2477-3994","authenticated-orcid":false,"given":"Thavida","family":"Maneewarn","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,2,27]]},"reference":[{"key":"204_CR1","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A.A., Veness, J., Bellemare, M.G., Graves, A., Riedmiller, M., Fidjeland, A.K., Ostrovski, G., Petersen, S., Beattie, C., Sadik, A., Antonoglou, I., King, H., Kumaran, D., Wierstra, D., Legg, S., Hassabis, D.: Human-level control through deep reinforcement learning. Nature 518, 529\u2013533 (2015). https:\/\/doi.org\/10.1038\/nature14236","journal-title":"Nature"},{"key":"204_CR2","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., Huang, A., Maddison, C.J., Guez, A., Sifre, L., van den Driessche, G., Schrittwieser, J., Antonoglou, I., Panneershelvam, V., Lanctot, M., Dieleman, S., Grewe, D., Nham, J., Kalchbrenner, N., Sutskever, I., Lillicrap, T., Leach, M., Kavukcuoglu, K., Graepel, T., Hassabis, D.: Mastering the game of Go with deep neural networks and tree search. Nature 529, 484\u2013489 (2016). https:\/\/doi.org\/10.1038\/nature16961","journal-title":"Nature"},{"key":"204_CR3","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., Schrittwieser, J., Simonyan, K., Antonoglou, I., Huang, A., Guez, A., Hubert, T., Baker, L., Lai, M., Bolton, A., Chen, Y., Lillicrap, T., Hui, F., Sifre, L., van den Driessche, G., Graepel, T., Hassabis, D.: Mastering the game of go without human knowledge. Nature 550, 354\u2013359 (2017). https:\/\/doi.org\/10.1038\/nature24270","journal-title":"Nature"},{"key":"204_CR4","doi-asserted-by":"crossref","unstructured":"Warnell, G., Waytowich, N., Lawhern, V., Stone, P.: Deep tamer: Interactive agent shaping in high-dimensional state spaces. In: Proceedings of Thirty-Second AAAI Conference on Artificial Intelligence (AAAI-18), New Orlearns, Louisiana, USA (2018)","DOI":"10.1609\/aaai.v32i1.11485"},{"issue":"3\u20134","key":"204_CR5","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1016\/S0921-8890(02)00168-9","volume":"38","author":"F Kaplan","year":"2002","unstructured":"Kaplan, F., Oudeyer, P.Y., Kubinyi, E., Miklosi, A.: Robotic clicker training. Robot. Auton. Syst. 38(3\u20134), 197\u2013206 (2002)","journal-title":"Robot. Auton. Syst."},{"key":"204_CR6","doi-asserted-by":"crossref","unstructured":"Thomaz, A.L., Hoffman, G., Breazeal, C.: Reinforcement learning with human teachers: understanding how people want to teach robots. In: Proceedings of the 15th IEEE International Symposium on Robot and Human Interactive Communication (ROMAN 2006), Hatfield, UK., pp. 352\u2013357 (2006)","DOI":"10.1109\/ROMAN.2006.314459"},{"key":"204_CR7","unstructured":"Thomaz, A.L., Breazeal, C.: Reinforcement learning with human teacher: evidence of feedback and guidance with implications for learning performance. In: Proceedings of the 21st National Conference on Artificial Intelligence (AAAI 06). Boston, Massachusetts, USA, pp. 1000\u20131005 (2006)"},{"issue":"6\u20134","key":"204_CR8","doi-asserted-by":"publisher","first-page":"716","DOI":"10.1016\/j.artint.2007.09.009","volume":"172","author":"AL Thomaz","year":"2008","unstructured":"Thomaz, A.L., Breazeal, C.: Teachable robots: understanding human teaching behavior to build more effective robot learners. Artif. Intell. 172(6\u20134), 716\u2013737 (2008). https:\/\/doi.org\/10.1016\/j.artint.2007.09.009","journal-title":"Artif. Intell."},{"key":"204_CR9","unstructured":"Griffith, S., Subramanian, K., Scholz, J., Isbell, C.L., Thomaz, A.L.: Policy shaping: integrating human feedback with reinforcement learning. Poster session presented at: Advances in Neural Information Processing Systems 26 (NIPS 2013) (2013)"},{"key":"204_CR10","doi-asserted-by":"crossref","unstructured":"Judah, K., Roy, S., Fern, A., Dietterich, T.G.: Reinforcement Learning Via Practice and Critique Advice. In: Proceedings of the 24th AAAI Conference on Artificial Intelligence (AAAI 2010). Atlanta, Georgia, USA (2010)","DOI":"10.1609\/aaai.v24i1.7690"},{"key":"204_CR11","doi-asserted-by":"crossref","unstructured":"Tenorio-Gonzalez, A., Morales, E., Villaseor-Pineda, L.: Dynamic reward shaping: training a robot by voice. Advances in Artificial Intelligence\u2013IBERAMIA. pp. 483\u2013492 (2010)","DOI":"10.1007\/978-3-642-16952-6_49"},{"key":"204_CR12","unstructured":"Leon, L.A., Tenorio, A.C., Morales, E.F.: Human interaction for effective reinforcement learning. In: Proceedings of the European Conference Machine Learning and Principles and Practice of Knowledge Discovery in Databases (ECMLPKDD 13). Prague (2013)"},{"key":"204_CR13","doi-asserted-by":"crossref","unstructured":"Knox, W.B., Stone, P.: Interactively shaping agents via human reinforcement: the TAMER framework. In: Proceedings of the 5th International Conference on Knowledge Capture (K-CAP 09). Redondo Beach, California, USA, pp. 9-16. (2009)","DOI":"10.1145\/1597735.1597738"},{"key":"204_CR14","unstructured":"Knox, W.B., Stone, P.: Combining manual feedback with subsequent MDP reward signals for reinforcement learning. In: Proceedings of the 9th International Conference on Autonomous Agents and Multiagent System (AAMAS10). Toronto, Canada, 1, pp. 5\u201312. (2010)"},{"key":"204_CR15","unstructured":"Knox, W.B., Stone, P.: Reinforcement learning from simultaneous human and MDP reward. In: Proceedings of the 11th International Conference on Autonomous Agents and Multiagent System (AAMAS 12). Valencia, Spain, 1, pp. 475\u2013482. (2012)"},{"key":"204_CR16","doi-asserted-by":"crossref","unstructured":"Sridharan, M.: Augmented reinforcement learning for interaction with non-expert humans in agent domains. In: Proceedings of IEEE International Conference on Machine Learning Applications. Honolulu, HI, USA, pp. 424\u2013429 (2011)","DOI":"10.1109\/ICMLA.2011.37"},{"key":"204_CR17","doi-asserted-by":"crossref","unstructured":"Celemin, C., Ruiz-del-Solar, J.: COACH: Learning continuous actions from COrrective Advice Communicated by Humans. In: Proceedings of 2015 International Conference on Advanced Robotics (ICAR), Istanbul, pp. 581\u2013586 (2015)","DOI":"10.1109\/ICAR.2015.7251514"},{"key":"204_CR18","doi-asserted-by":"crossref","unstructured":"Celemin, C., Ruiz-del-Solar, J.: Teaching Agents with Corrective Human Feedback for Challenging Problem. In: Proceedings of 2016 IEEE Latin American Conference on Computational Intelligence (LA-CCI). Cartagena, pp. 1\u20136 (2016)","DOI":"10.1109\/LA-CCI.2016.7885734"},{"key":"204_CR19","unstructured":"Vien, N.A., Ertel, W.: Learning via human feedback in continuous state and action spaces. In: Proceedings of AAAI 2012 Fall Symposium Series, Robots Learning Interactively from Human Teachers (RLIHT). Arlington, USA, pp. 65\u201372 (2012)"},{"key":"204_CR20","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1007\/978-3-642-32177-1_3","volume-title":"Innovations in Intelligent Machines -3","author":"Masakazu Hirkoawa","year":"2013","unstructured":"Hirkoawa, M., Suzuki, K.: Coaching robots: online behavior learning from human subjective feedback. In: Innovations in Intelligent Machines-3, vol. 442, pp. 37\u201351 (2013)"},{"key":"204_CR21","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: International Conference on Machine Learning (ICML) (2009)","DOI":"10.1145\/1553374.1553380"},{"key":"204_CR22","unstructured":"CMUSphinx Documentation\u2014CMUSphinx Open Source Speech Recognition. https:\/\/cmusphinx.github.io\/wiki\/(2018). Accessed 31 Dec 2018"},{"key":"204_CR23","unstructured":"Walker, W., Lamere, P., Kwok, P., Raj, B., Singh, R., Gouvea, E., Wolf, P., W\u00f6lfel, J.: Sphinx-4: a flexible open source framework for speech recognition. Technical Report. Sun Microsystems. Mountain View, CA, USA (2004)"},{"key":"204_CR24","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. The MIT Press, Cambridge (1998)"}],"container-title":["Progress in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13748-020-00204-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s13748-020-00204-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s13748-020-00204-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,16]],"date-time":"2022-10-16T19:56:59Z","timestamp":1665950219000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s13748-020-00204-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,2,27]]},"references-count":24,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2020,6]]}},"alternative-id":["204"],"URL":"https:\/\/doi.org\/10.1007\/s13748-020-00204-4","relation":{},"ISSN":["2192-6352","2192-6360"],"issn-type":[{"type":"print","value":"2192-6352"},{"type":"electronic","value":"2192-6360"}],"subject":[],"published":{"date-parts":[[2020,2,27]]},"assertion":[{"value":"24 June 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 February 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 February 2020","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}