{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T20:11:17Z","timestamp":1725999077017},"publisher-location":"Cham","reference-count":29,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030010539"},{"type":"electronic","value":"9783030010546"}],"license":[{"start":{"date-parts":[[2018,11,9]],"date-time":"2018-11-09T00:00:00Z","timestamp":1541721600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-01054-6_13","type":"book-chapter","created":{"date-parts":[[2018,11,8]],"date-time":"2018-11-08T14:47:05Z","timestamp":1541688425000},"page":"187-200","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Proposal and Evaluation of an Indirect Reward Assignment Method for Reinforcement Learning by Profit Sharing Method"],"prefix":"10.1007","author":[{"given":"Kazuteru","family":"Miyazaki","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Naoki","family":"Kodama","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hiroaki","family":"Kobayashi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,11,9]]},"reference":[{"key":"13_CR1","doi-asserted-by":"crossref","unstructured":"Abbeel, P., Ng, A.Y.: Exploration and apprenticeship learning in reinforcement learning. In: Proceedings of 22nd International Conference on Machine Learning, pp. 1\u20138 (2005)","DOI":"10.1145\/1102351.1102352"},{"key":"13_CR2","unstructured":"Chrisman, L.: Reinforcement learning with perceptual aliasing: the perceptual distinctions approach. In: Proceedings of the 10th National Conference on Artificial Intelligence, pp. 183\u2013188 (1992)"},{"key":"13_CR3","unstructured":"Francois-Lavet, V., Fonteneau, R., Emst, D.: How to discount deep reinforcement learning: towards new dynamic strategies. In: NIPS 2015 Deep Reinforcement Learning Workshop (2015)"},{"issue":"1","key":"13_CR4","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1023\/B:MACH.0000019802.64038.6c","volume":"55","author":"A Gosavi","year":"2004","unstructured":"Gosavi, A.: A reinforcement learning algorithm based on policy iteration for average reward: empirical results with yield management and convergence analysis. Mach. Learn. 55(1), 5\u201329 (2004)","journal-title":"Mach. Learn."},{"key":"13_CR5","unstructured":"Gu, S., Lillicrap, T., Sutskever, I., Levine, S.: Continuous deep Q-learning with model-based acceleration. \narXiv:1603\n\n (2016)"},{"key":"13_CR6","unstructured":"Hausknecht, M., Stone, P.: Deep recurrent Q-learning for partially observable MDPs. \narXiv:1507\n\n (2015)"},{"issue":"6","key":"13_CR7","doi-asserted-by":"publisher","first-page":"758","DOI":"10.20965\/jaciii.2012.p0758","volume":"16","author":"S Kuroda","year":"2013","unstructured":"Kuroda, S., Miyazaki, K., Kobayashi, H.: Introduction of fixed mode states into online reinforcement learning with penalties and rewards and its application to biped robot waist trajectory generation. J. Adv. Comput. Intell. Intell. Inf. 16(6), 758\u2013768 (2013)","journal-title":"J. Adv. Comput. Intell. Intell. Inf."},{"issue":"6","key":"13_CR8","doi-asserted-by":"publisher","first-page":"691","DOI":"10.20965\/jaciii.2009.p0691","volume":"13","author":"T Matsui","year":"2009","unstructured":"Matsui, T., Goto, T., Izumi, K.: Acquiring a government bond trading strategy using reinforcement learning. J. Adv. Comput. Intell. Intell. Inf. 13(6), 691\u2013696 (2009)","journal-title":"J. Adv. Comput. Intell. Intell. Inf."},{"key":"13_CR9","doi-asserted-by":"crossref","unstructured":"Merrick, K., Maher, M.L.: Motivated reinforcement learning for adaptive characters in open-ended simulation games. In: Proceedings of the International Conference on Advanced in Computer Entertainment Technology, pp. 127\u2013134 (2007)","DOI":"10.1145\/1255047.1255073"},{"key":"13_CR10","unstructured":"Miyazaki, K., Yamamura, M., Kobayashi, S: On the rationality of profit sharing in reinforcement learning. In: Proceedings of the 3rd International Conference on Fuzzy Logic, Neural Nets and Soft Computing, pp. 285\u2013288 (1994)"},{"issue":"1","key":"13_CR11","doi-asserted-by":"publisher","first-page":"155","DOI":"10.1016\/S0004-3702(96)00062-8","volume":"91","author":"K Miyazaki","year":"1997","unstructured":"Miyazaki, K., Yamamura, M., Kobayashi, S.: k-certainty exploration method: an action selector to identify the environment in reinforcement learning. Artif. Intell. 91(1), 155\u2013171 (1997)","journal-title":"Artif. Intell."},{"key":"13_CR12","unstructured":"Miyazaki, K., Kobayashi, S.: Learning deterministic policies in partially observable Markov decision processes. In: Proceedings of the 5th International Conference on Intelligent Autonomous System, pp. 250\u2013257 (1998)"},{"key":"13_CR13","unstructured":"Miyazaki, K., Tsuboi, S., Kobayashi, S.: Reinforcement learning for penalty avoiding policy making and its extensions and an application to the Othello game. In: 7th International Conference on Information Systems Analysis and Synthesis (ISAS 2000), vol. 3, pp. 40\u201344 (2001)"},{"issue":"2","key":"13_CR14","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1007\/BF03037252","volume":"19","author":"K Miyazaki","year":"2001","unstructured":"Miyazaki, K., Kobayashi, S.: Rationality of reward sharing in multi-agent reinforcement learning. New Gener. Comput. 19(2), 157\u2013172 (2001)","journal-title":"New Gener. Comput."},{"issue":"6","key":"13_CR15","doi-asserted-by":"publisher","first-page":"624","DOI":"10.20965\/jaciii.2009.p0624","volume":"13","author":"K Miyazaki","year":"2009","unstructured":"Miyazaki, K., Kobayashi, S.: Exploitation-oriented learning PS-r$$^{\\#}$$. J. Adv. Comput. Intell. Intell. Inf. 13(6), 624\u2013630 (2009)","journal-title":"J. Adv. Comput. Intell. Intell. Inf."},{"key":"13_CR16","unstructured":"Miyazaki, K., Muraoka, H., Kobayashi, H.: Proposal of a propagation algorithm of the expected failure probability and the effectiveness on multi-agent environments. In: SICE Annual Conference 2013, pp. 1067\u20131072 (2013)"},{"issue":"5","key":"13_CR17","doi-asserted-by":"publisher","first-page":"849","DOI":"10.20965\/jaciii.2017.p0849","volume":"21","author":"K Miyazaki","year":"2017","unstructured":"Miyazaki, K.: Exploitation-oriented learning with deep learning - introducing profit sharing to a deep Q-network. J. Adv. Comput. Intell. Intell. Inf. 21(5), 849\u2013855 (2017)","journal-title":"J. Adv. Comput. Intell. Intell. Inf."},{"key":"13_CR18","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1016\/j.procs.2014.11.079","volume":"41","author":"K Miyazaki","year":"2014","unstructured":"Miyazaki, K., Takeno, J.: The necessity of a secondary system in machine consciousness. Procedia Comput. Sci. 41, 15\u201322 (2014)","journal-title":"Procedia Comput. Sci."},{"key":"13_CR19","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Graves, A., Antonoglou, I., Wierstra, D., Riedmiller, M.: Playing Atari with deep reinforcement learning. In: NIPS Deep Learning Workshop 2013 (2013)"},{"key":"13_CR20","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A.A., Veness, J., Bellemare, M.G., Graves, A., Riedmiller, M., Fidjeland, A.K., Ostrovski, G., Petersen, S., Beattie, C., Sadik, A., Antonoglou, I., King, H., Kumaran, D., Wierstra, D., Legg, S., Hassabis, D.: Human-level control through deep reinforcement learning. Nature 518, 529\u2013533 (2015)","journal-title":"Nature"},{"key":"13_CR21","unstructured":"Nair, A., Srinivasan, P., Blackwell, S., Alcicek, C., Fearon, R., Maria, A.D., Suleyman, M., Beattie, C., Petersen, S., Legg, S., Mnih, V., Silver, D.: Massively parallel methods for deep reinforcement learning. In: ICML Deep Learning Workshop (2015)"},{"key":"13_CR22","unstructured":"Ng, A.Y., Harada, D., Russell, S.J.: Policy invariance under reward transformations: theory and application to reward shaping. In: Proceedings of 16th International Conference on Machine Learning, pp. 278\u2013287 (1999)"},{"key":"13_CR23","unstructured":"Osband, I., Blundell, C., Pritzel, A., Roy, B.V.: Deep exploration via bootstrapped DQN. \narXiv:1602\n\n (2016)"},{"key":"13_CR24","unstructured":"Randl$$\\phi $$v, J., Alstr$$\\phi $$m, P.: Learning to drive a bicycle using reinforcement learning and shaping. In: Proceedings of the 15th International Conference on Machine Learning, pp. 463\u2013471 (1998)"},{"key":"13_CR25","series-title":"A Bradford Book","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. A Bradford Book. MIT Press, Cambridge (1998)"},{"issue":"3","key":"13_CR26","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1177\/105971230501300301","volume":"13","author":"P Stone","year":"2005","unstructured":"Stone, P., Sutton, R.S., Kuhlamann, G.: Reinforcement learning toward RoboCup soccer keepaway. Adapt. Behav. 13(3), 165\u2013188 (2005)","journal-title":"Adapt. Behav."},{"issue":"6","key":"13_CR27","doi-asserted-by":"publisher","first-page":"675","DOI":"10.20965\/jaciii.2009.p0675","volume":"13","author":"T Watanabe","year":"2009","unstructured":"Watanabe, T., Miyazaki, K., Kobayashi, H.: A new improved penalty avoiding rational policy making algorithm for keepaway with continuous state spaces. J. Adv. Comput. Intell. Intell. Inf. 13(6), 675\u2013682 (2009)","journal-title":"J. Adv. Comput. Intell. Intell. Inf."},{"key":"13_CR28","first-page":"55","volume":"8","author":"CJH Watkins","year":"1992","unstructured":"Watkins, C.J.H., Dayan, P.: Technical note: Q-learning. Mach. Learn. 8, 55\u201368 (1992)","journal-title":"Mach. Learn."},{"issue":"2","key":"13_CR29","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s10015-004-0340-6","volume":"9","author":"J Yoshimoto","year":"2005","unstructured":"Yoshimoto, J., Nishimura, M., Tokita, Y., Ishii, S.: Acrobot control by learning the switching of multiple controllers. J. Artif. Life Rob. 9(2), 67\u201371 (2005)","journal-title":"J. Artif. Life Rob."}],"container-title":["Advances in Intelligent Systems and Computing","Intelligent Systems and Applications"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-01054-6_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2018,11,8]],"date-time":"2018-11-08T14:56:40Z","timestamp":1541689000000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-01054-6_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,11,9]]},"ISBN":["9783030010539","9783030010546"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-01054-6_13","relation":{},"ISSN":["2194-5357","2194-5365"],"issn-type":[{"type":"print","value":"2194-5357"},{"type":"electronic","value":"2194-5365"}],"subject":[],"published":{"date-parts":[[2018,11,9]]},"assertion":[{"value":"IntelliSys","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Proceedings of SAI Intelligent Systems Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"London","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2018","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 September 2018","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 September 2018","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"intellisys2018","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/saiconference.com\/IntelliSys2018\/CallforPapers","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}