{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T18:57:09Z","timestamp":1784573829023,"version":"3.55.0"},"publisher-location":"Berlin, Heidelberg","reference-count":25,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783642244117","type":"print"},{"value":"9783642244124","type":"electronic"}],"license":[{"start":{"date-parts":[[2011,1,1]],"date-time":"2011-01-01T00:00:00Z","timestamp":1293840000000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2011]]},"DOI":"10.1007\/978-3-642-24412-4_16","type":"book-chapter","created":{"date-parts":[[2011,10,6]],"date-time":"2011-10-06T04:41:17Z","timestamp":1317876077000},"page":"174-188","source":"Crossref","is-referenced-by-count":250,"title":["On Upper-Confidence Bound Policies for Switching Bandit Problems"],"prefix":"10.1007","author":[{"given":"Aur\u00e9lien","family":"Garivier","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eric","family":"Moulines","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","reference":[{"issue":"4","key":"16_CR1","doi-asserted-by":"crossref","first-page":"1054","DOI":"10.2307\/1427934","volume":"27","author":"R. Agrawal","year":"1995","unstructured":"Agrawal, R.: Sample mean based index policies with O(logn) regret for the multi-armed bandit problem. Adv. in Appl. Probab.\u00a027(4), 1054\u20131078 (1995)","journal-title":"Adv. in Appl. Probab."},{"key":"16_CR2","series-title":"Lecture Notes in Artificial Intelligence","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1007\/978-3-540-75225-7_15","volume-title":"Algorithmic Learning Theory","author":"J.Y. Audibert","year":"2007","unstructured":"Audibert, J.Y., Munos, R., Szepesvari, A.: Tuning bandit algorithms in stochastic environments. In: Hutter, M., Servedio, R.A., Takimoto, E. (eds.) ALT 2007. LNCS (LNAI), vol.\u00a04754, pp. 150\u2013165. Springer, Heidelberg (2007)"},{"issue":"1","key":"16_CR3","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1137\/S0097539701398375","volume":"32","author":"P. Auer","year":"2002","unstructured":"Auer, P., Cesa-Bianchi, N., Freund, Y., Schapire, R.E.: The nonstochastic multiarmed bandit problem. SIAM J. Comput.\u00a032(1), 48\u201377 (2002)","journal-title":"SIAM J. Comput."},{"issue":"Spec. Issue Com","key":"16_CR4","first-page":"397","volume":"3","author":"P. Auer","year":"2002","unstructured":"Auer, P.: Using confidence bounds for exploitation-exploration trade-offs. J. Mach. Learn. Res.\u00a03(Spec. Issue Comput. Learn. Theory), 397\u2013422 (2002)","journal-title":"J. Mach. Learn. Res."},{"issue":"2\/3","key":"16_CR5","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1023\/A:1013689704352","volume":"47","author":"P. Auer","year":"2002","unstructured":"Auer, P., Cesa-Bianchi, N., Fischer, P.: Finite-time analysis of the multiarmed bandit problem. Machine Learning\u00a047(2\/3), 235\u2013256 (2002)","journal-title":"Machine Learning"},{"key":"16_CR6","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511546921","volume-title":"Prediction, Learning, and Games","author":"N. Cesa-Bianchi","year":"2006","unstructured":"Cesa-Bianchi, N., Lugosi, G.: Prediction, Learning, and Games. Cambridge University Press, New York (2006)"},{"issue":"6","key":"16_CR7","doi-asserted-by":"publisher","first-page":"1865","DOI":"10.1214\/aos\/1017939242","volume":"27","author":"N. Cesa-Bianchi","year":"1999","unstructured":"Cesa-Bianchi, N., Lugosi, G.: On prediction of individual sequences. Ann. Statist.\u00a027(6), 1865\u20131895 (1999)","journal-title":"Ann. Statist."},{"issue":"3","key":"16_CR8","doi-asserted-by":"publisher","first-page":"562","DOI":"10.1287\/moor.1060.0206","volume":"31","author":"N. Cesa-Bianchi","year":"2006","unstructured":"Cesa-Bianchi, N., Lugosi, G., Stoltz, G.: Regret minimization under partial monitoring. Math. Oper. Res.\u00a031(3), 562\u2013580 (2006)","journal-title":"Math. Oper. Res."},{"key":"16_CR9","unstructured":"Cesa-Bianchi, N., Lugosi, G., Stoltz, G.: Competing with typical compound actions (2008)"},{"key":"16_CR10","series-title":"Applications of Mathematics","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4612-0711-5","volume-title":"A probabilistic theory of pattern recognition","author":"L. Devroye","year":"1996","unstructured":"Devroye, L., Gy\u00f6rfi, L., Lugosi, G.: A probabilistic theory of pattern recognition. Applications of Mathematics, vol.\u00a031. Springer, New York (1996)"},{"key":"#cr-split#-16_CR11.1","doi-asserted-by":"crossref","unstructured":"2. Freund, Y., Schapire, R.E.: A decision-theoretic generalization of on-line learning and an application to boosting. J. Comput. System Sci. 55(1, part 2), 119-139 (1997)","DOI":"10.1006\/jcss.1997.1504"},{"key":"#cr-split#-16_CR11.2","unstructured":"3. In: Vit\u00e1nyi, P.M.B. (ed.) EuroCOLT 1995. LNCS, vol.\u00a0904. Springer, Heidelberg (1995)"},{"issue":"5","key":"16_CR12","doi-asserted-by":"publisher","first-page":"2305","DOI":"10.1214\/009053604000000580","volume":"32","author":"C.D. Fuh","year":"2004","unstructured":"Fuh, C.D.: Asymptotic operating characteristics of an optimal change point detection in hidden Markov models. Ann. Statist.\u00a032(5), 2305\u20132339 (2004)","journal-title":"Ann. Statist."},{"key":"16_CR13","unstructured":"Garivier, A., Capp\u00e9, O.: The kl-ucb algorithm for bounded stochastic bandits and beyond. In: Proceedings of the 24rd Annual International Conference on Learning Theory (2011)"},{"key":"16_CR14","unstructured":"Hartland, C., Gelly, S., Baskiotis, N., Teytaud, O., Sebag, M.: Multi-armed bandit, dynamic environments and meta-bandits. In: nIPS-2006 Workshop, Online Trading Between Exploration and Exploitation, Whistler, Canada (2006)"},{"issue":"2","key":"16_CR15","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1023\/A:1007424614876","volume":"32","author":"M. Herbster","year":"1998","unstructured":"Herbster, M., Warmuth, M.: Tracking the best expert. Machine Learning\u00a032(2), 151\u2013178 (1998)","journal-title":"Machine Learning"},{"key":"16_CR16","unstructured":"Honda, J., Takemura, A.: An asymptotically optimal bandit algorithm for bounded support models. In: Proceedings of the 23rd Annual International Conference on Learning Theory (2010)"},{"key":"16_CR17","unstructured":"Kocsis, L., Szepesv\u00e1ri, C.: Discounted UCB. In: 2nd PASCAL Challenges Workshop, Venice, Italy (April 2006)"},{"issue":"2","key":"16_CR18","doi-asserted-by":"publisher","first-page":"913","DOI":"10.1016\/j.amc.2007.07.043","volume":"196","author":"D.E. Koulouriotis","year":"2008","unstructured":"Koulouriotis, D.E., Xanthopoulos, A.: Reinforcement learning and evolutionary algorithms for non-stationary multi-armed bandit problems. Applied Mathematics and Computation\u00a0196(2), 913\u2013922 (2008)","journal-title":"Applied Mathematics and Computation"},{"key":"16_CR19","unstructured":"Lai, L., El Gamal, H., Jiang, H., Poor, H.V.: Cognitive medium access: Exploration, exploitation and competition (2007)"},{"issue":"1","key":"16_CR20","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1016\/0196-8858(85)90002-8","volume":"6","author":"T.L. Lai","year":"1985","unstructured":"Lai, T.L., Robbins, H.: Asymptotically efficient adaptive allocation rules. Adv. in Appl. Math.\u00a06(1), 4\u201322 (1985)","journal-title":"Adv. in Appl. Math."},{"issue":"1","key":"16_CR21","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1214\/009053605000000859","volume":"34","author":"Y. Mei","year":"2006","unstructured":"Mei, Y.: Sequential change-point detection when unknown parameters are present in the pre-change distribution. Ann. Statist.\u00a034(1), 92\u2013122 (2006)","journal-title":"Ann. Statist."},{"key":"16_CR22","unstructured":"Slivkins, A., Upfal, E.: Adapting to a changing environment: the brownian restless bandits. In: Proceedings of the Conference on 21st Conference on Learning Theory, pp. 343\u2013354 (2008)"},{"key":"16_CR23","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1017\/S0021900200040420","volume":"25A","author":"P. Whittle","year":"1988","unstructured":"Whittle, P.: Restless bandits: activity allocation in a changing world. J. Appl. Probab. Special\u00a025A, 287\u2013298 (1988) a celebration of applied probability","journal-title":"J. Appl. Probab. Special"},{"key":"16_CR24","first-page":"1177","volume-title":"ICML 2009: Proceedings of the 26th Annual International Conference on Machine Learning","author":"J.Y. Yu","year":"2009","unstructured":"Yu, J.Y., Mannor, S.: Piecewise-stationary bandit problems with side observations. In: ICML 2009: Proceedings of the 26th Annual International Conference on Machine Learning, pp. 1177\u20131184. ACM, New York (2009)"}],"container-title":["Lecture Notes in Computer Science","Algorithmic Learning Theory"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-24412-4_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,4,10]],"date-time":"2019-04-10T03:14:50Z","timestamp":1554866090000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-24412-4_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2011]]},"ISBN":["9783642244117","9783642244124"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-24412-4_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2011]]}}}