{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,16]],"date-time":"2026-01-16T10:12:13Z","timestamp":1768558333023,"version":"3.49.0"},"publisher-location":"Berlin, Heidelberg","reference-count":24,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783540386254","type":"print"},{"value":"9783540386278","type":"electronic"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2006]]},"DOI":"10.1007\/11840817_82","type":"book-chapter","created":{"date-parts":[[2006,8,31]],"date-time":"2006-08-31T18:11:56Z","timestamp":1157047916000},"page":"790-800","source":"Crossref","is-referenced-by-count":9,"title":["Optimal Tuning of Continual Online Exploration in Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Youssef","family":"Achbany","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Francois","family":"Fouss","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luh","family":"Yen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alain","family":"Pirotte","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marco","family":"Saerens","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"82_CR1","doi-asserted-by":"crossref","unstructured":"Achbany, Y., Fouss, F., Yen, L., Pirotte, A., Saerens, M.: Tuning continual exploration in reinforcement learning. Technical report (2005), http:\/\/www.isys.ucl.ac.be\/staff\/francois\/Articles\/Achbany2005a.pdf","DOI":"10.1007\/11840817_82"},{"key":"82_CR2","volume-title":"Nonlinear programming: Theory and algorithms","author":"M.S. Bazaraa","year":"1993","unstructured":"Bazaraa, M.S., Sherali, H.D., Shetty, C.M.: Nonlinear programming: Theory and algorithms. John Wiley and Sons, Chichester (1993)"},{"key":"82_CR3","volume-title":"Neuro-dynamic programming","author":"D.P. Bertsekas","year":"1996","unstructured":"Bertsekas, D.P.: Neuro-dynamic programming. Athena Scientific, Belmont (1996)"},{"key":"82_CR4","volume-title":"Network optimization: continuous and discrete models","author":"D.P. Bertsekas","year":"1998","unstructured":"Bertsekas, D.P.: Network optimization: continuous and discrete models. Athena Scientific, Belmont (1998)"},{"key":"82_CR5","volume-title":"Dynamic programming and optimal control","author":"D.P. Bertsekas","year":"2000","unstructured":"Bertsekas, D.P.: Dynamic programming and optimal control. Athena sientific, Belmont (2000)"},{"key":"82_CR6","unstructured":"Boyan, J.A., Littman, M.L.: Packet routing in dynamically changing networks: A reinforcement learning approach. In: Advances in Neural Information Processing Systems 6 (NIPS6), pp. 671\u2013678 (1994)"},{"key":"82_CR7","volume-title":"Smoothing, forecasting and prediction of discrete time series","author":"R.G. Brown","year":"1962","unstructured":"Brown, R.G.: Smoothing, forecasting and prediction of discrete time series. Prentice-Hall, Englewood Cliffs (1962)"},{"key":"82_CR8","volume-title":"Graph theory: An algorithmic approach","author":"N. Christofides","year":"1975","unstructured":"Christofides, N.: Graph theory: An algorithmic approach. Academic Press, London (1975)"},{"key":"82_CR9","doi-asserted-by":"publisher","DOI":"10.1002\/0471200611","volume-title":"Elements of information theory","author":"T.M. Cover","year":"1991","unstructured":"Cover, T.M., Thomas, J.A.: Elements of information theory. John Wiley and Sons, Chichester (1991)"},{"key":"82_CR10","volume-title":"Entropy optimization principles with applications","author":"J.N. Kapur","year":"1992","unstructured":"Kapur, J.N., Kesavan, H.K.: Entropy optimization principles with applications. Academic Press, London (1992)"},{"key":"82_CR11","volume-title":"Finite markov chains","author":"J.G. Kemeny","year":"1976","unstructured":"Kemeny, J.G., Snell, J.L.: Finite markov chains. Springer, Heidelberg (1976)"},{"key":"82_CR12","volume-title":"An introduction to game theory","author":"M.J. Osborne","year":"2004","unstructured":"Osborne, M.J.: An introduction to game theory. Oxford University Press, Oxford (2004)"},{"key":"82_CR13","volume-title":"Decision analysis","author":"H. Raiffa","year":"1970","unstructured":"Raiffa, H.: Decision analysis. Addison-Wesley, Reading (1970)"},{"key":"82_CR14","unstructured":"Rummery, G., Niranjan, M.: On-line q-learning using connectionist systems. Technical Report CUED\/F-INFENG\/TR 166, Cambridge University Engineering Departement (1994)"},{"key":"82_CR15","series-title":"Lecture Notes in Artificial Intelligence","doi-asserted-by":"publisher","first-page":"353","DOI":"10.1007\/11564096_35","volume-title":"Machine Learning: ECML 2005","author":"G. Shani","year":"2005","unstructured":"Shani, G., Brafman, R., Shimony, S.: Adaptation for changing stochastic environments through online pomdp policy learning. In: Gama, J., Camacho, R., Brazdil, P.B., Jorge, A.M., Torgo, L. (eds.) ECML 2005. LNCS (LNAI), vol.\u00a03720, pp. 353\u2013364. Springer, Heidelberg (2005)"},{"key":"82_CR16","first-page":"123","volume":"22","author":"S. Singh","year":"1996","unstructured":"Singh, S., Sutton, R.: Reinforcement learning with replacing eligibility traces. Machine Learning\u00a022, 123\u2013158 (1996)","journal-title":"Machine Learning"},{"key":"82_CR17","doi-asserted-by":"publisher","DOI":"10.1002\/0471722138","volume-title":"Introduction to stochastic search and optimization","author":"J.C. Spall","year":"2003","unstructured":"Spall, J.C.: Introduction to stochastic search and optimization. Wiley, Chichester (2003)"},{"key":"82_CR18","volume-title":"Reinforcement learning: an introduction","author":"R.S. Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement learning: an introduction. The MIT Press, Cambridge (1998)"},{"key":"82_CR19","unstructured":"Thrun, S.: Efficient exploration in reinforcement learning. Technical report, School of Computer Science, Carnegie Mellon University (1992)"},{"key":"82_CR20","unstructured":"Thrun, S.: The role of exploration in learning control. In: White, D., Sofge, D. (eds.) Handbook for Intelligent Control: Neural, Fuzzy and Adaptive Approaches, Van Nostrand Reinhold, Florence, Kentucky 41022 (1992)"},{"key":"82_CR21","volume-title":"Probabilistic Robotics","author":"S. Thrun","year":"2005","unstructured":"Thrun, S., Burgard, W., Fox, D.: Probabilistic Robotics. MIT Press, Cambridge (2005)"},{"key":"82_CR22","unstructured":"Verbeeck, K.: Coordinated exploration in multi-agent reinforcement learning. PhD thesis, Vrije Universiteit Brussel, Belgium (2004)"},{"key":"82_CR23","unstructured":"Watkins, J.C.: Learning from delayed rewards. PhD thesis, King\u2019s College of Cambridge, UK (1989)"},{"issue":"3-4","key":"82_CR24","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1007\/BF00992698","volume":"8","author":"J.C. Watkins","year":"1992","unstructured":"Watkins, J.C., Dayan, P.: Q-learning. Machine Learning\u00a08(3-4), 279\u2013292 (1992)","journal-title":"Machine Learning"}],"container-title":["Lecture Notes in Computer Science","Artificial Neural Networks \u2013 ICANN 2006"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/11840817_82.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,17]],"date-time":"2020-11-17T19:40:00Z","timestamp":1605642000000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/11840817_82"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2006]]},"ISBN":["9783540386254","9783540386278"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/11840817_82","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2006]]}}}