{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T17:26:41Z","timestamp":1725989201061},"publisher-location":"Cham","reference-count":25,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030001100"},{"type":"electronic","value":"9783030001117"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-3-030-00111-7_28","type":"book-chapter","created":{"date-parts":[[2018,8,30]],"date-time":"2018-08-30T00:32:07Z","timestamp":1535589127000},"page":"327-340","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Preference-Based Monte Carlo Tree Search"],"prefix":"10.1007","author":[{"given":"Tobias","family":"Joppen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Christian","family":"Wirth","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Johannes","family":"F\u00fcrnkranz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,8,30]]},"reference":[{"key":"28_CR1","unstructured":"Amodei, D., Olah, C., Steinhardt, J., Christiano, P., Schulman, J., Man\u00e9, D.: Concrete problems in AI safety. CoRR abs\/1606.06565 (2016)"},{"issue":"2\u20133","key":"28_CR2","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1023\/A:1013689704352","volume":"47","author":"P Auer","year":"2002","unstructured":"Auer, P., Cesa-Bianchi, N., Fischer, P.: Finite-time analysis of the multiarmed bandit problem. Mach. Learn. 47(2\u20133), 235\u2013256 (2002)","journal-title":"Mach. Learn."},{"issue":"1","key":"28_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TCIAIG.2012.2186810","volume":"4","author":"CB Browne","year":"2012","unstructured":"Browne, C.B., et al.: A survey of Monte Carlo tree search methods. IEEE Trans. Comput. Intell. AI Games 4(1), 1\u201343 (2012)","journal-title":"IEEE Trans. Comput. Intell. AI Games"},{"key":"28_CR4","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1007\/978-3-319-11662-4_3","volume-title":"Algorithmic Learning Theory","author":"R Busa-Fekete","year":"2014","unstructured":"Busa-Fekete, R., H\u00fcllermeier, E.: A survey of preference-based online learning with bandit algorithms. In: Auer, P., Clark, A., Zeugmann, T., Zilles, S. (eds.) ALT 2014. LNCS (LNAI), vol. 8776, pp. 18\u201339. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-11662-4_3"},{"key":"28_CR5","unstructured":"Christiano, P., Leike, J., Brown, T.B., Martic, M., Legg, S., Amodei, D.: Deep reinforcement learning from human preferences. In: Guyon, I., et al. (eds.) Advances in Neural Information Processing Systems 30 (NIPS 2017), Long Beach, CA (2017)"},{"key":"28_CR6","unstructured":"Finnsson, H.: Simulation-based general game playing. Ph.D. thesis, Reykjav\u00edk University (2012)"},{"key":"28_CR7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-14125-6","volume-title":"Preference Learning","year":"2011","unstructured":"F\u00fcrnkranz, J., H\u00fcllermeier, E. (eds.): Preference Learning. Springer, Heidelberg (2011). https:\/\/doi.org\/10.1007\/978-3-642-14125-6"},{"issue":"1\u20132","key":"28_CR8","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1007\/s10994-012-5313-8","volume":"89","author":"J F\u00fcrnkranz","year":"2012","unstructured":"F\u00fcrnkranz, J., H\u00fcllermeier, E., Cheng, W., Park, S.H.: Preference-based reinforcement learning: a formal framework and a policy iteration algorithm. Mach. Learn. 89(1\u20132), 123\u2013156 (2012). https:\/\/doi.org\/10.1007\/s10994-012-5313-8 . Special Issue of Selected Papers from ECML PKDD 2011","journal-title":"Mach. Learn."},{"key":"28_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"269","DOI":"10.1007\/3-540-44719-9_19","volume-title":"Evolutionary Multi-Criterion Optimization","author":"JD Knowles","year":"2001","unstructured":"Knowles, J.D., Watson, R.A., Corne, D.W.: Reducing local optima in single-objective problems by multi-objectivization. In: Zitzler, E., Thiele, L., Deb, K., Coello Coello, C.A., Corne, D. (eds.) EMO 2001. LNCS, vol. 1993, pp. 269\u2013283. Springer, Heidelberg (2001). https:\/\/doi.org\/10.1007\/3-540-44719-9_19"},{"key":"28_CR10","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1007\/11871842_29","volume-title":"Machine Learning: ECML 2006","author":"L Kocsis","year":"2006","unstructured":"Kocsis, L., Szepesv\u00e1ri, C.: Bandit based Monte-Carlo planning. In: F\u00fcrnkranz, J., Scheffer, T., Spiliopoulou, M. (eds.) ECML 2006. LNCS (LNAI), vol. 4212, pp. 282\u2013293. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11871842_29"},{"key":"28_CR11","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1109\/TCIAIG.2009.2018703","volume":"1","author":"CS Lee","year":"2009","unstructured":"Lee, C.S.: The computational intelligence of MoGo revealed in Taiwan\u2019s computer go tournaments. IEEE Trans. Comput. Intell. AI Games 1, 73\u201389 (2009)","journal-title":"IEEE Trans. Comput. Intell. AI Games"},{"issue":"3","key":"28_CR12","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1109\/TCIAIG.2013.2291577","volume":"6","author":"T Pepels","year":"2014","unstructured":"Pepels, T., Winands, M.H., Lanctot, M.: Real-time Monte Carlo tree search in Ms Pac-Man. IEEE Trans. Comput. Intell. AI Games 6(3), 245\u2013257 (2014)","journal-title":"IEEE Trans. Comput. Intell. AI Games"},{"key":"28_CR13","doi-asserted-by":"crossref","unstructured":"Perez-Liebana, D., Mostaghim, S., Lucas, S.M.: Multi-objective tree search approaches for general video game playing. In: IEEE Congress on Evolutionary Computation (CEC 2016), pp. 624\u2013631. IEEE (2016)","DOI":"10.1109\/CEC.2016.7743851"},{"key":"28_CR14","unstructured":"Ponsen, M., Gerritsen, G., Chaslot, G.: Integrating opponent models with Monte-Carlo tree search in poker. In: Proceedings of Interactive Decision Theory and Game Theory Workshop at the Twenty-Fourth Conference on Artificial Intelligence (AAAI 2010), AAAI Workshops, vol. WS-10-03, pp. 37\u201342 (2010)"},{"key":"28_CR15","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"ML Puterman","year":"2005","unstructured":"Puterman, M.L.: Markov Decision Processes: Discrete Stochastic Dynamic Programming, 2nd edn. Wiley, Hoboken (2005)","edition":"2"},{"issue":"4","key":"28_CR16","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1109\/TCIAIG.2010.2098876","volume":"2","author":"A Rimmel","year":"2010","unstructured":"Rimmel, A., Teytaud, O., Lee, C.S., Yen, S.J., Wang, M.H., Tsai, S.R.: Current frontiers in computer go. IEEE Trans. Comput. Intell. AI Games 2(4), 229\u2013238 (2010)","journal-title":"IEEE Trans. Comput. Intell. AI Games"},{"issue":"7676","key":"28_CR17","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., et al.: Mastering the game of go without human knowledge. Nature 550(7676), 354 (2017)","journal-title":"Nature"},{"key":"28_CR18","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (1998)"},{"key":"28_CR19","first-page":"278","volume":"34","author":"LL Thurstone","year":"1927","unstructured":"Thurstone, L.L.: A law of comparative judgement. Psychol. Rev. 34, 278\u2013286 (1927)","journal-title":"Psychol. Rev."},{"key":"28_CR20","doi-asserted-by":"crossref","unstructured":"Weng, P.: Markov decision processes with ordinal rewards: reference point-based preferences. In: Proceedings of the 21st International Conference on Automated Planning and Scheduling (ICAPS 2011) (2011)","DOI":"10.1609\/icaps.v21i1.13448"},{"key":"28_CR21","doi-asserted-by":"crossref","unstructured":"Wirth, C., F\u00fcrnkranz, J., Neumann, G.: Model-free preference-based reinforcement learning. In: Proceedings of the 30th AAAI Conference on Artificial Intelligence (AAAI 2016), pp. 2222\u20132228 (2016)","DOI":"10.1609\/aaai.v30i1.10269"},{"key":"28_CR22","doi-asserted-by":"crossref","unstructured":"Yannakakis, G.N., Cowie, R., Busso, C.: The ordinal nature of emotions. In: Proceedings of the 7th International Conference on Affective Computing and Intelligent Interaction (ACII 2017) (2017)","DOI":"10.1109\/ACII.2017.8273608"},{"issue":"5","key":"28_CR23","doi-asserted-by":"publisher","first-page":"1538","DOI":"10.1016\/j.jcss.2011.12.028","volume":"78","author":"Y Yue","year":"2012","unstructured":"Yue, Y., Broder, J., Kleinberg, R., Joachims, T.: The k-armed dueling bandits problem. J. Comput. Syst. Sci. 78(5), 1538\u20131556 (2012). https:\/\/doi.org\/10.1016\/j.jcss.2011.12.028","journal-title":"J. Comput. Syst. Sci."},{"key":"28_CR24","doi-asserted-by":"crossref","unstructured":"Yue, Y., Joachims, T.: Interactively optimizing information retrieval systems as a dueling bandits problem. In: Proceedings of the 26th Annual International Conference on Machine Learning (ICML 2009), pp. 1201\u20131208 (2009)","DOI":"10.1145\/1553374.1553527"},{"key":"28_CR25","unstructured":"Zoghi, M., Whiteson, S., Munos, R., Rijke, M.: Relative upper confidence bound for the k-armed dueling bandit problem. In: Proceedings of the 31st International Conference on Machine Learning (ICML 2014), pp. 10\u201318 (2014)"}],"container-title":["Lecture Notes in Computer Science","KI 2018: Advances in Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-00111-7_28","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,4]],"date-time":"2023-09-04T14:05:31Z","timestamp":1693836331000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-00111-7_28"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9783030001100","9783030001117"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-00111-7_28","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2018]]}}}