{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,15]],"date-time":"2024-09-15T15:18:48Z","timestamp":1726413528783},"reference-count":22,"publisher":"Springer Science and Business Media LLC","issue":"1-2","license":[{"start":{"date-parts":[[2010,6,10]],"date-time":"2010-06-10T00:00:00Z","timestamp":1276128000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Ann Math Artif Intell"],"published-print":{"date-parts":[[2010,10]]},"DOI":"10.1007\/s10472-010-9190-1","type":"journal-article","created":{"date-parts":[[2010,6,9]],"date-time":"2010-06-09T09:47:05Z","timestamp":1276076825000},"page":"3-24","source":"Crossref","is-referenced-by-count":3,"title":["A dynamic programming strategy to balance exploration and exploitation in the bandit problem"],"prefix":"10.1007","volume":"60","author":[{"given":"Olivier","family":"Caelen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gianluca","family":"Bontempi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2010,6,10]]},"reference":[{"key":"9190_CR1","unstructured":"Audibert, J.-Y., Munos, R., Szepesv\u00e1ri, C.: Use of variance estimation in the multi-armed bandit problem. In: NIPS 2006 Workshop on On-line Trading of Exploration and Exploitation (2006)"},{"issue":"2\/3","key":"9190_CR2","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1023\/A:1013689704352","volume":"47","author":"P Auer","year":"2002","unstructured":"Auer, P., Cesa-Bianchi, N., Fischer, P.: Finite-time analysis of the multiarmed bandit problem. Mach. Learn. 47(2\/3), 235\u2013256 (2002)","journal-title":"Mach. Learn."},{"key":"9190_CR3","first-page":"322","volume-title":"Proceedings of the 36th Annual Symposium on Foundations of Computer Science","author":"P Auer","year":"1995","unstructured":"Auer, P., Cesa-Bianchi, N., Freund, Y., Schapire, R.E.: Gambling in a rigged casino: the adversarial multi-armed bandit problem. In: Proceedings of the 36th Annual Symposium on Foundations of Computer Science, pp. 322\u2013331. IEEE Computer Society, Los Alamitos (1995)"},{"issue":"1","key":"9190_CR4","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/S0167-9236(03)00061-7","volume":"38","author":"R Azoulay-Schwartz","year":"2004","unstructured":"Azoulay-Schwartz, R., Kraus, S., Wilkenfeld, J.: Exploitation vs. exploration: choosing a supplier in an environment of incomplete information. Decis. Support Syst. 38(1), 1\u201318 (2004)","journal-title":"Decis. Support Syst."},{"key":"9190_CR5","volume-title":"Dynamic Programming\u2014Deterministic and Stochastic Models","author":"DP Bertsekas","year":"1987","unstructured":"Bertsekas, D.P.: Dynamic Programming\u2014Deterministic and Stochastic Models. Prentice-Hall, Upper Saddle River (1987)"},{"key":"9190_CR6","volume-title":"Neuro-Dynamic Programming","author":"DP Bertsekas","year":"1996","unstructured":"Bertsekas, D.P., Tsitsiklis, J.N.: Neuro-Dynamic Programming. Athena Scientific, Belmont (1996)"},{"key":"9190_CR7","volume-title":"Pattern Recognition and Machine Learning (Information Science and Statistics)","author":"CM Bishop","year":"2006","unstructured":"Bishop, C.M.: Pattern Recognition and Machine Learning (Information Science and Statistics). Springer, New York (2006)"},{"key":"9190_CR8","series-title":"Lecture Notes in Computer Science","first-page":"56","volume-title":"Learning and Intelligent OptimizatioN LION 2007 II","author":"O Caelen","year":"2007","unstructured":"Caelen, O., Bontempi, G.: Improving the exploration strategy in bandit algorithms. In: Maniezzo, V., Battiti, R., Watson, J.-P. (eds.) Learning and Intelligent OptimizatioN LION 2007 II. Lecture Notes in Computer Science, vol. 5313, pp. 56\u201368. Springer, New York (2007)"},{"key":"9190_CR9","unstructured":"Caelen, O., Bontempi, G.: On the evolution of the expected gain of a greedy action in the bandit problem. Technical Report 589, D\u00e9partement d\u2019Informatique, Universit\u00e9 Libre de Bruxelles, Brussels, Belgium (2008)"},{"key":"9190_CR10","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4899-4541-9","volume-title":"An Introduction to the Bootstrap","author":"B Efron","year":"1993","unstructured":"Efron, B., Tibshirani, R.J.: An Introduction to the Bootstrap. Chapman and Hall, New York (1993)"},{"key":"9190_CR11","volume-title":"Multi-armed Bandit Allocation Indices","author":"JC Gittins","year":"1989","unstructured":"Gittins, J.C.: Multi-armed Bandit Allocation Indices. Wiley, New York (1989)"},{"key":"9190_CR12","first-page":"421","volume":"23","author":"J Hardwick","year":"1991","unstructured":"Hardwick, J., Stout, Q.: Bandit strategies for ethical sequential allocation. Computing Sci. Stat. 23, 421\u2013424 (1991)","journal-title":"Computing Sci. Stat."},{"key":"9190_CR13","volume-title":"Handbooks in Operations Research and Management Science: Simulation, Chapter Selecting the Best System","author":"S Kim","year":"2006","unstructured":"Kim, S., Nelson, B.: Handbooks in Operations Research and Management Science: Simulation, Chapter Selecting the Best System. Elsevier, Amsterdam (2006)"},{"key":"9190_CR14","doi-asserted-by":"crossref","first-page":"479","DOI":"10.1137\/S1052623499363220","volume":"12","author":"AJ Kleywegt","year":"2001","unstructured":"Kleywegt, A.J., Shapiro, A., Homem de Mello, T.: The sample average approximation method for stochastic discrete optimization. SIAM J. Optim. 12, 479\u2013502 (2001)","journal-title":"SIAM J. Optim."},{"issue":"2","key":"9190_CR15","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1023\/A:1007541107674","volume":"35","author":"N Meuleau","year":"1999","unstructured":"Meuleau, N., Bourgine, P.: Exploration of multi-state environments: local measures and back-propagation of uncertainty. Mach. Learn. 35(2), 117\u2013154 (1999)","journal-title":"Mach. Learn."},{"key":"9190_CR16","first-page":"126","volume-title":"Artificial Neural Networks for Speech and Vision","author":"MP Perrone","year":"1993","unstructured":"Perrone, M.P., Cooper, L.N.: When networks disagree: ensemble methods for hybrid neural networks. In: Mammone, R.J. (ed.) Artificial Neural Networks for Speech and Vision, pp. 126\u2013142. Chapman and Hall, New York (1993)"},{"key":"9190_CR17","doi-asserted-by":"crossref","DOI":"10.1002\/9780470182963","volume-title":"Approximate Dynamic Programming\u2014Solving the Curses of Dimensionality","author":"WB Powell","year":"2007","unstructured":"Powell, W.B.: Approximate Dynamic Programming\u2014Solving the Curses of Dimensionality. Wiley, Princeton (2007)"},{"key":"9190_CR18","doi-asserted-by":"crossref","DOI":"10.1002\/9780470316887","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"ML Puterman","year":"1994","unstructured":"Puterman, M.L.: Markov Decision Processes: Discrete Stochastic Dynamic Programming. Wiley, New York (1994)"},{"issue":"5","key":"9190_CR19","doi-asserted-by":"crossref","first-page":"527","DOI":"10.1090\/S0002-9904-1952-09620-8","volume":"58","author":"H Robbins","year":"1952","unstructured":"Robbins, H.: Some aspects of the sequential design of experiments. Bull. Am. Math. Soc. 58(5), 527\u2013535 (1952)","journal-title":"Bull. Am. Math. Soc."},{"key":"9190_CR20","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT, Cambridge (1998)"},{"key":"9190_CR21","doi-asserted-by":"crossref","unstructured":"Vermorel, J., Mohri, M.: Multi-armed bandit algorithms and empirical evaluation. In: 16th European Conference on Machine Learning (ECML05), pp. 437\u2013448. ecml (2005)","DOI":"10.1007\/11564096_42"},{"key":"9190_CR22","unstructured":"Watkins, C.: Learning from delayed rewards. Ph.D. thesis, Cambridge University (1989)"}],"container-title":["Annals of Mathematics and Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10472-010-9190-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10472-010-9190-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10472-010-9190-1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,29]],"date-time":"2019-05-29T23:18:59Z","timestamp":1559171939000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10472-010-9190-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2010,6,10]]},"references-count":22,"journal-issue":{"issue":"1-2","published-print":{"date-parts":[[2010,10]]}},"alternative-id":["9190"],"URL":"https:\/\/doi.org\/10.1007\/s10472-010-9190-1","relation":{},"ISSN":["1012-2443","1573-7470"],"issn-type":[{"type":"print","value":"1012-2443"},{"type":"electronic","value":"1573-7470"}],"subject":[],"published":{"date-parts":[[2010,6,10]]}}}