{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T15:37:22Z","timestamp":1778600242513,"version":"3.51.4"},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2016,2,22]],"date-time":"2016-02-22T00:00:00Z","timestamp":1456099200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100000266","name":"Engineering and Physical Sciences Research Council","doi-asserted-by":"publisher","award":["EP\/H012338\/1"],"award-info":[{"award-number":["EP\/H012338\/1"]}],"id":[{"id":"10.13039\/501100000266","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2016,7]]},"DOI":"10.1007\/s10994-016-5547-y","type":"journal-article","created":{"date-parts":[[2016,2,22]],"date-time":"2016-02-22T15:19:37Z","timestamp":1456154377000},"page":"99-127","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":52,"title":["Bayesian policy reuse"],"prefix":"10.1007","volume":"104","author":[{"given":"Benjamin","family":"Rosman","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Majd","family":"Hawasly","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Subramanian","family":"Ramamoorthy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,2,22]]},"reference":[{"issue":"1","key":"5547_CR1","doi-asserted-by":"crossref","first-page":"96","DOI":"10.1111\/j.1748-1090.2006.00096.x","volume":"40","author":"R Amin","year":"2006","unstructured":"Amin, R., Thomas, K., Emslie, R. H., Foose, T. J., & Strien, N. (2006). An overview of the conservation status of and threats to rhinoceros species in the wild. International Zoo Yearbook, 40(1), 96\u2013117.","journal-title":"International Zoo Yearbook"},{"issue":"2","key":"5547_CR2","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1023\/A:1013689704352","volume":"47","author":"P Auer","year":"2002","unstructured":"Auer, P., Cesa-Bianchi, N., & Fischer, P. (2002). Finite-time analysis of the multiarmed bandit problem. Machine Learning, 47(2), 235\u2013256.","journal-title":"Machine Learning"},{"key":"5547_CR3","unstructured":"Brochu, E., Cora, V.M., & De Freitas, N. (2010). A tutorial on Bayesian optimization of expensive cost functions, with application to active user modeling and hierarchical reinforcement learning. arXiv preprint arXiv:1012.2599 ."},{"key":"5547_CR4","unstructured":"Brunskill, E., & Li, L. (2013). Sample complexity of multi-task reinforcement learning. In Proceedings of the 29th conference on uncertainty in artificial intelligence (UAI)."},{"key":"5547_CR5","unstructured":"Bui, L., Johari, R., & Mannor, S. (2012). Clustered bandits. CoRR. arXiv:1206.4169 ."},{"key":"5547_CR6","unstructured":"da Silva, B. C., Konidaris, G. D., & Barto, A. G. (2012). Learning parameterized skills. In Proceedings of the twenty ninth international conference on machine learning."},{"key":"5547_CR7","unstructured":"Dearden, R., Friedman, N., & Andre, D. (1999) Model based Bayesian exploration. In Proceedings of the fifteenth conference on uncertainty in artificial intelligence (pp. 150\u2013159). Morgan Kaufmann."},{"key":"5547_CR8","unstructured":"Engel, Y., & Ghavamzadeh, M. (2007). Bayesian policy gradient algorithms. In Proceedings of the 2006 conference, advances in neural information processing systems (vol. 19, p. 457). MIT Press."},{"key":"5547_CR9","doi-asserted-by":"crossref","unstructured":"Fern\u00e1ndez, F., & Veloso, M. (2006). Probabilistic policy reuse in a reinforcement learning agent. In Proceedings of the fifth international joint conference on autonomous agents and multiagent systems (pp. 720\u2013727). ACM.","DOI":"10.1145\/1160633.1160762"},{"key":"5547_CR10","doi-asserted-by":"crossref","unstructured":"Ginebra, J., & Clayton, M. K. (1995). Response surface bandits. Journal of the Royal Statistical Society, Series B (Methodological), 57(4), 771\u2013784.","DOI":"10.1111\/j.2517-6161.1995.tb02062.x"},{"key":"5547_CR11","doi-asserted-by":"crossref","unstructured":"Gittins, J. C., & Jones, D. M. (1979). A dynamic allocation index for the discounted multiarmed bandit problem. Biometrika, 66(3), 561\u2013565.","DOI":"10.1093\/biomet\/66.3.561"},{"issue":"1","key":"5547_CR12","doi-asserted-by":"crossref","first-page":"4","DOI":"10.1016\/0196-8858(85)90002-8","volume":"6","author":"Tze Leung Lai","year":"1985","unstructured":"Lai, Tze Leung, & Robbins, Herbert. (1985). Asymptotically efficient adaptive allocation rules. Advances in Applied Mathematics, 6(1), 4\u201322.","journal-title":"Advances in Applied Mathematics"},{"key":"5547_CR13","unstructured":"Langford, J., & Zhang, T. (2008). The epoch-greedy algorithm for multi-armed bandits with side information. In Advances in neural information processing systems (pp. 817\u2013824)."},{"key":"5547_CR14","unstructured":"Lazaric, A. (2008). Knowledge transfer in reinforcement learning. PhD thesis, Politecnico di Milano."},{"key":"5547_CR15","unstructured":"Mahmud, M., Hassan, M., Hawasly, M., Rosman, B., & Ramamoorthy, S. (2013). Clustering Markov decision processes for continual transfer. arXiv preprint. arXiv:1311.3959 ."},{"key":"5547_CR16","unstructured":"Mahmud, M., Hassan, M., Rosman, B., Ramamoorthy, S., & Kohli, P. (2014). Adapting interaction environments to diverse users through online action set selection. In AAAI 2014 workshop on machine learning for interactive systems."},{"key":"5547_CR17","unstructured":"Maillard, O. A., & Mannor, S. (2014). Latent bandits. In Proceedings of the 31st international conference on machine learning (pp. 136\u2013144)."},{"issue":"12","key":"5547_CR18","doi-asserted-by":"crossref","first-page":"2787","DOI":"10.1109\/TAC.2009.2031725","volume":"54","author":"AJ Mersereau","year":"2009","unstructured":"Mersereau, A. J., Rusmevichientong, P., & Tsitsiklis, J. N. (2009). A structured multiarmed bandit problem and the greedy policy. IEEE Transactions on Automatic Control, 54(12), 2787\u20132802.","journal-title":"IEEE Transactions on Automatic Control"},{"issue":"2","key":"5547_CR19","doi-asserted-by":"crossref","first-page":"254","DOI":"10.1287\/ijoc.1100.0398","volume":"23","author":"J Ni\u00f1o-Mora","year":"2011","unstructured":"Ni\u00f1o-Mora, J. (2011). Computing a classic index for finite-horizon bandits. INFORMS Journal on Computing, 23(2), 254\u2013267.","journal-title":"INFORMS Journal on Computing"},{"key":"5547_CR20","doi-asserted-by":"crossref","unstructured":"Pandey, S., Chakrabarti, D., & Agarwal, D. (2007). Multi-armed bandit problems with dependent arms. In Proceedings of the 24th international conference on Machine learning (pp 721\u2013728). ACM.","DOI":"10.1145\/1273496.1273587"},{"key":"5547_CR21","unstructured":"Powell, W. B. (2010). The knowledge gradient for optimal learning. Wiley Encyclopedia of Operations Research and ManagementScience."},{"key":"5547_CR22","doi-asserted-by":"crossref","unstructured":"Rosman, B., Ramamoorthy, S., Mahmud, M., Hassan, M., & Kohli, P. (2014). On user behaviour adaptation under interface change. In International conference on intelligent user interfaces.","DOI":"10.1145\/2557500.2557535"},{"issue":"1","key":"5547_CR23","first-page":"2533","volume":"15","author":"A Slivkins","year":"2014","unstructured":"Slivkins, A. (2014). Contextual bandits with similarity information. The Journal of Machine Learning Research, 15(1), 2533\u20132568.","journal-title":"The Journal of Machine Learning Research"},{"key":"5547_CR24","unstructured":"Srinivas, N., Krause, A., Kakade, S. M., & Seeger, M. (2009). Gaussian process optimization in the bandit setting: No regret and experimental design. arXiv preprint. arXiv:0912.3995 ."},{"key":"5547_CR25","doi-asserted-by":"crossref","unstructured":"Strehl, A. L., Mesterharm, C., Littman, M. L., & Hirsh, H. (2006). Experience-efficient learning in associative bandit problems. In Proceedings of the 23rd international conference on machine learning (pp. 889\u2013896). ACM.","DOI":"10.1145\/1143844.1143956"},{"key":"5547_CR26","first-page":"1633","volume":"10","author":"ME Taylor","year":"2009","unstructured":"Taylor, M. E., & Stone, P. (2009). Transfer learning for reinforcement learning domains: A survey. The Journal of Machine Learning Research, 10, 1633\u20131685.","journal-title":"The Journal of Machine Learning Research"},{"key":"5547_CR27","doi-asserted-by":"crossref","unstructured":"Wilson, A., Fern, A., Ray, S., & Tadepalli, P. (2007). Multi-task reinforcement learning: A hierarchical bayesian approach. In Proceedings of the 24th international conference on Machine learning (pp 1015\u20131022). ACM.","DOI":"10.1145\/1273496.1273624"},{"key":"5547_CR28","unstructured":"Wingate, D., Goodman, N. D., Roy, D. M., Kaelbling, L. P., & Tenenbaum, J. B. (2011). Bayesian policy search with policy priors. In Proceedings of the twenty-second international joint conference on artificial intelligence-volume two (pp. 1565\u20131570). AAAI Press."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-016-5547-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10994-016-5547-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-016-5547-y","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,4]],"date-time":"2019-09-04T17:43:58Z","timestamp":1567619038000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10994-016-5547-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,2,22]]},"references-count":28,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2016,7]]}},"alternative-id":["5547"],"URL":"https:\/\/doi.org\/10.1007\/s10994-016-5547-y","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,2,22]]}}}