{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T17:00:28Z","timestamp":1765040428333},"publisher-location":"Berlin, Heidelberg","reference-count":16,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642413971"},{"type":"electronic","value":"9783642413988"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-41398-8_37","type":"book-chapter","created":{"date-parts":[[2013,10,15]],"date-time":"2013-10-15T09:09:30Z","timestamp":1381828170000},"page":"427-437","source":"Crossref","is-referenced-by-count":2,"title":["A Policy Iteration Algorithm for Learning from Preference-Based Feedback"],"prefix":"10.1007","author":[{"given":"Christian","family":"Wirth","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Johannes","family":"F\u00fcrnkranz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"37_CR1","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"116","DOI":"10.1007\/978-3-642-33486-3_8","volume-title":"Machine Learning and Knowledge Discovery in Databases","author":"R. Akrour","year":"2012","unstructured":"Akrour, R., Schoenauer, M., Sebag, M.: APRIL: Active preference learning-based reinforcement learning. In: Flach, P.A., De Bie, T., Cristianini, N. (eds.) ECML PKDD 2012, Part II. LNCS, vol.\u00a07524, pp. 116\u2013131. Springer, Heidelberg (2012)"},{"key":"37_CR2","unstructured":"Audibert, J.Y., Bubeck, S.: Minimax policies for adversarial and stochastic bandits. In: Proceedings of the 22nd Conference on Learning Theory (COLT 2009), Montreal, Quebec, Canada, pp. 773\u2013818 (2009)"},{"issue":"2-3","key":"37_CR3","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1023\/A:1013689704352","volume":"47","author":"P. Auer","year":"2002","unstructured":"Auer, P., Cesa-Bianchi, N., Fischer, P.: Finite-time analysis of the multiarmed bandit problem. Machine Learning\u00a047(2-3), 235\u2013256 (2002)","journal-title":"Machine Learning"},{"key":"37_CR4","unstructured":"Auer, P., Cesa-Bianchi, N., Freund, Y., Schapire, R.E.: Gambling in a rigged casino: The adversarial multi-arm bandit problem. In: Proceedings of the 36th Annual Symposium on Foundations of Computer Science, pp. 322\u2013331 (1995)"},{"issue":"3","key":"37_CR5","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1007\/s10994-008-5069-3","volume":"72","author":"C. Dimitrakakis","year":"2008","unstructured":"Dimitrakakis, C., Lagoudakis, M.G.: Rollout sampling approximate policy iteration. Machine Learning\u00a072(3), 157\u2013171 (2008)","journal-title":"Machine Learning"},{"key":"37_CR6","doi-asserted-by":"crossref","unstructured":"F\u00fcrnkranz, J., H\u00fcllermeier, E. (eds.): Preference Learning. Springer (2010)","DOI":"10.1007\/978-3-642-14125-6"},{"issue":"1-2","key":"37_CR7","doi-asserted-by":"crossref","first-page":"123","DOI":"10.1007\/s10994-012-5313-8","volume":"89","author":"J. F\u00fcrnkranz","year":"2012","unstructured":"F\u00fcrnkranz, J., H\u00fcllermeier, E., Cheng, W., Park, S.H.: Preference-based reinforcement learning: a formal framework and a policy iteration algorithm. Machine Learning\u00a089(1-2), 123\u2013156 (2012), special Issue of Selected Papers from ECML PKDD 2011","journal-title":"Machine Learning"},{"key":"37_CR8","doi-asserted-by":"publisher","first-page":"451","DOI":"10.1214\/aos\/1028144844","volume":"26","author":"T. Hastie","year":"1998","unstructured":"Hastie, T., Tibshirani, R.: Classification by pairwise coupling. The Annals of Statistics\u00a026, 451\u2013471 (1998)","journal-title":"The Annals of Statistics"},{"key":"37_CR9","unstructured":"Price, D., Knerr, S., Personnaz, L., Dreyfus, G.: Pairwise neural network classifiers with probabilistic outputs. In: Proceedings of the 7th Conference Advances in Neural Information Processing Systems (NIPS 1994), vol.\u00a07, pp. 1109\u20131116. MIT Press (1994)"},{"key":"37_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1007\/978-3-642-23808-6_3","volume-title":"Machine Learning and Knowledge Discovery in Databases","author":"C.A. Rothkopf","year":"2011","unstructured":"Rothkopf, C.A., Dimitrakakis, C.: Preference elicitation and inverse reinforcement learning. In: Gunopulos, D., Hofmann, T., Malerba, D., Vazirgiannis, M. (eds.) ECML PKDD 2011, Part III. LNCS, vol.\u00a06913, pp. 34\u201348. Springer, Heidelberg (2011)"},{"issue":"3","key":"37_CR11","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1023\/A:1007678930559","volume":"38","author":"S.P. Singh","year":"2000","unstructured":"Singh, S.P., Jaakkola, T., Littman, M.L., Szepesv\u00e1ri, C.: Convergence results for single-step on-policy reinforcement-learning algorithms. Machine Learning\u00a038(3), 287\u2013308 (2000)","journal-title":"Machine Learning"},{"key":"37_CR12","volume-title":"Reinforcement Learning: An Introduction","author":"R.S. Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (1998)"},{"key":"37_CR13","unstructured":"Wirth, C., F\u00fcrnkranz, J.: Learning from trajectory-based action preferences. In: Proceedings of the ICRA 2013 Workshop on Autonomous Learning (to appear, May 2013)"},{"key":"37_CR14","first-page":"975","volume":"5","author":"T.F. Wu","year":"2004","unstructured":"Wu, T.F., Lin, C.J., Weng, R.C.: Probability estimates for multi-class classification by pairwise coupling. Journal of Machine Learning Research\u00a05, 975\u20131005 (2004)","journal-title":"Journal of Machine Learning Research"},{"key":"37_CR15","first-page":"3295","volume":"28","author":"Y. Zhao","year":"2009","unstructured":"Zhao, Y., Kosorok, M., Zeng, D.: Reinforcement learning design for cancer clinical trials. Statistics in Medicine\u00a028, 3295\u20133315 (2009)","journal-title":"Statistics in Medicine"},{"key":"37_CR16","first-page":"1142","volume":"25","author":"A. Wilson","year":"2012","unstructured":"Wilson, A., Fern, A., Tadepalli, P.: A Bayesian Approach for Policy Learning from Trajectory Preference Queries. Advances in Neural Information Processing Systems\u00a025, 1142\u20131150 (2012)","journal-title":"Advances in Neural Information Processing Systems"}],"container-title":["Lecture Notes in Computer Science","Advances in Intelligent Data Analysis XII"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-41398-8_37","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,23]],"date-time":"2019-05-23T16:38:03Z","timestamp":1558629483000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-41398-8_37"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642413971","9783642413988"],"references-count":16,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-41398-8_37","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2013]]}}}