{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T20:02:56Z","timestamp":1772654576758,"version":"3.50.1"},"publisher-location":"Cham","reference-count":19,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030226480","type":"print"},{"value":"9783030226497","type":"electronic"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-22649-7_25","type":"book-chapter","created":{"date-parts":[[2019,7,9]],"date-time":"2019-07-09T23:37:42Z","timestamp":1562715462000},"page":"311-321","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Model-Based Multi-objective Reinforcement Learning with Unknown Weights"],"prefix":"10.1007","author":[{"given":"Tomohiro","family":"Yamaguchi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shota","family":"Nagahama","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yoshihiro","family":"Ichikawa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keiki","family":"Takadama","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,6,29]]},"reference":[{"issue":"1\u20132","key":"25_CR1","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1016\/0004-3702(94)00011-O","volume":"72","author":"AG Barto","year":"1995","unstructured":"Barto, A.G., Steven, J., Bradtke, S.J., Singh, S.P.: Learning to act using real-time dynamic programming. Artif. Intell. 72(1\u20132), 81\u2013138 (1995)","journal-title":"Artif. Intell."},{"key":"25_CR2","unstructured":"Gao, Y.: Research on Average Reward. Reinforcement Learning. Algorithms, National Laboratory for Novel Software Technology, Nanjing University, 5 November 2006. http:\/\/lamda.nju.edu.cn\/conf\/MLA06\/files\/Gao.Y.pdf"},{"key":"25_CR3","unstructured":"Herrmann, M.: RL 16: Model-based RL and Multi-Objective Reinforcement Learning, University of Edinburgh, School of Informatics (2015). http:\/\/www.inf.ed.ac.uk\/teaching\/courses\/rl\/slides15\/rl16.pdf"},{"key":"25_CR4","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1007\/s11571-008-9066-9","volume":"3","author":"K Hiraoka","year":"2009","unstructured":"Hiraoka, K., Yoshida, M., Mishima, T.: Parallel reinforcement learning for weighted multi-criteria model with adaptive margin. Cogn. Neurodyn. 3, 17\u201324 (2009)","journal-title":"Cogn. Neurodyn."},{"key":"25_CR5","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"208","DOI":"10.1007\/3-540-45683-X_24","volume-title":"PRICAI 2002: Trends in Artificial Intelligence","author":"T Konda","year":"2002","unstructured":"Konda, T., Tensyo, S., Yamaguchi, T.: LC-learning: phased method for average reward reinforcement learning\u2014preliminary results\u2014. In: Ishizuka, M., Sattar, A. (eds.) PRICAI 2002. LNCS (LNAI), vol. 2417, pp. 208\u2013217. Springer, Heidelberg (2002). https:\/\/doi.org\/10.1007\/3-540-45683-X_24"},{"issue":"3","key":"25_CR6","doi-asserted-by":"publisher","first-page":"385","DOI":"10.1109\/TSMC.2014.2358639","volume":"45","author":"C Liu","year":"2015","unstructured":"Liu, C., Xu, X., Hu, D.: Multiobjective reinforcement learning: a comprehensive overview. IEEE Trans. Syst. Man Cybern. Syst. 45(3), 385\u2013398 (2015)","journal-title":"IEEE Trans. Syst. Man Cybern. Syst."},{"key":"25_CR7","first-page":"3253","volume":"13","author":"DJ Lizotte","year":"2012","unstructured":"Lizotte, D.J., Bowling, M., Murphy, S.A.: Linear fitted-Q iteration with multiple reward functions. J. Mach. Learn. Res. 13, 3253\u20133295 (2012)","journal-title":"J. Mach. Learn. Res."},{"key":"25_CR8","first-page":"159","volume":"22","author":"S Mahadevan","year":"1996","unstructured":"Mahadevan, S.: Average reward reinforcement learning: foundations, algorithms, and empirical results. Mach. Learn. 22, 159\u2013196 (1996)","journal-title":"Mach. Learn."},{"key":"25_CR9","first-page":"3663","volume":"15","author":"K Van Moffaert","year":"2014","unstructured":"Van Moffaert, K., Nowe, A.: Multi-objective reinforcement learning using sets of pareto dominating policies. J. Mach. Learn. Res. 15, 3663\u20133692 (2014)","journal-title":"J. Mach. Learn. Res."},{"key":"25_CR10","doi-asserted-by":"crossref","unstructured":"Natarajan, S., Tadepalli, P.: Dynamic preferences in multi-criteria reinforcement learning. In: Proceedings of International Conference on Machine Learning (ICML-2005), pp. 601\u201360 (2005)","DOI":"10.1145\/1102351.1102427"},{"key":"25_CR11","unstructured":"Pinder, J.M.: Multi-objective reinforcement learning framework for unknown stochastic & uncertain environments. Ph.D. Thesis (2016)"},{"key":"25_CR12","doi-asserted-by":"publisher","first-page":"385","DOI":"10.1002\/9780470316887","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"ML Puterman","year":"1994","unstructured":"Puterman, M.L.: Markov Decision Processes: Discrete Stochastic Dynamic Programming, pp. 385\u2013388. Wiley, New York (1994)"},{"key":"25_CR13","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1613\/jair.3987","volume":"48","author":"DM Roijers","year":"2013","unstructured":"Roijers, D.M., Vamplew, P., Whiteson, S., Dazeley, R.: A survey of multi-objective sequential decision-making. J. Artif. Intell. Res. 48, 67\u2013113 (2013)","journal-title":"J. Artif. Intell. Res."},{"key":"25_CR14","unstructured":"Roijers, D.M., Whiteson, S., Vamplew, P., Dazeley, R.: Why multi-objective reinforcement learning? In: European Workshop on Reinforcement Learning, pp. 1\u20132 (2015)"},{"key":"25_CR15","doi-asserted-by":"crossref","unstructured":"Satoh, K., Yamaguchi, T.: Preparing various policies for interactive reinforcement learning. In: SICE-ICASE International Joint Conference 2006 (2006)","DOI":"10.1109\/SICE.2006.315139"},{"key":"25_CR16","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1016\/S0004-3702(98)00002-2","volume":"100","author":"P Tadepalli","year":"1998","unstructured":"Tadepalli, P., Ok, D.: Model-based average reward reinforcement learning. Artif. Intell. 100, 177\u2013224 (1998)","journal-title":"Artif. Intell."},{"key":"25_CR17","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1016\/j.orl.2006.06.005","volume":"35","author":"JN Tsitsiklis","year":"2007","unstructured":"Tsitsiklis, J.N.: NP-hardness of checking the unichain condition in average cost MDPs. Oper. Res. Lett. 35, 319\u2013323 (2007)","journal-title":"Oper. Res. Lett."},{"key":"25_CR18","doi-asserted-by":"crossref","unstructured":"Yang, S. Gao, Y., Bo, A., Wang, H., Chen, X.: Efficient average reward reinforcement learning using constant shifting values. In: Proceedings of the Thirtieth AAAI Conference on Artificial Intelligence (AAAI-16), pp. 2258\u20132264 (2016)","DOI":"10.1609\/aaai.v30i1.10285"},{"key":"25_CR19","doi-asserted-by":"crossref","unstructured":"Wiering, M.A., Withagen, M., Drugan, M.M.: Model-based multiobjective reinforcement learning, In: ADPRL 2014: Proceedings of the IEEE Symposium on Adaptive Dynamic Programming and Reinforcement Learning, pp. 1\u20136 (2014)","DOI":"10.1109\/ADPRL.2014.7010622"}],"container-title":["Lecture Notes in Computer Science","Human Interface and the Management of Information. Information in Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-22649-7_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,19]],"date-time":"2023-07-19T00:21:18Z","timestamp":1689726078000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-22649-7_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030226480","9783030226497"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-22649-7_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"29 June 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"HCII","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Orlando, FL","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31 July 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hcii2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2019.hci.international\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}