{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T08:23:35Z","timestamp":1775031815190,"version":"3.50.1"},"publisher-location":"Cham","reference-count":40,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030925109","type":"print"},{"value":"9783030925116","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-92511-6_10","type":"book-chapter","created":{"date-parts":[[2021,12,7]],"date-time":"2021-12-07T14:05:12Z","timestamp":1638885912000},"page":"154-170","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["A Novel Implementation of\u00a0Q-Learning for\u00a0the\u00a0Whittle Index"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7345-7878","authenticated-orcid":false,"given":"Lachlan J.","family":"Gibson","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3376-0260","authenticated-orcid":false,"given":"Peter","family":"Jacko","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3071-8462","authenticated-orcid":false,"given":"Yoni","family":"Nazarathy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,12,8]]},"reference":[{"issue":"3","key":"10_CR1","doi-asserted-by":"publisher","first-page":"681","DOI":"10.1137\/S0363012999361974","volume":"40","author":"J Abounadi","year":"2001","unstructured":"Abounadi, J., Bertsekas, D., Borkar, V.S.: Learning algorithms for Markov decision processes with average cost. SIAM J. Control Optim. 40(3), 681\u2013698 (2001)","journal-title":"SIAM J. Control Optim."},{"issue":"2","key":"10_CR2","doi-asserted-by":"publisher","first-page":"314","DOI":"10.1287\/opre.1070.0505","volume":"57","author":"TW Archibald","year":"2009","unstructured":"Archibald, T.W., Black, D.P., Glazebrook, K.D.: Indexability and index heuristics for a simple class of inventory routing problems. Oper. Res. 57(2), 314\u2013326 (2009)","journal-title":"Oper. Res."},{"key":"10_CR3","doi-asserted-by":"crossref","unstructured":"Avrachenkov, K.E., Borkar, V.S.: Whittle index based Q-learning for restless bandits with average reward. arXiv:2004.14427 (2021)","DOI":"10.1016\/j.automatica.2022.110186"},{"key":"10_CR4","doi-asserted-by":"publisher","first-page":"1014","DOI":"10.1016\/j.peva.2010.08.015","volume":"67","author":"U Ayesta","year":"2010","unstructured":"Ayesta, U., Erausquin, M., Jacko, P.: A modeling framework for optimizing the flow-level scheduling with time-varying channels. Perform. Eval. 67, 1014\u20131029 (2010)","journal-title":"Perform. Eval."},{"issue":"2","key":"10_CR5","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1007\/s10951-015-0456-7","volume":"20","author":"U Ayesta","year":"2015","unstructured":"Ayesta, U., Jacko, P., Novak, V.: Scheduling of multi-class multi-server queueing systems with abandonments. J. Sched. 20(2), 129\u2013145 (2015). https:\/\/doi.org\/10.1007\/s10951-015-0456-7","journal-title":"J. Sched."},{"key":"10_CR6","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-015-3711-7","volume-title":"Bandit Problems: Sequential Allocation of Experiments","author":"DA Berry","year":"1985","unstructured":"Berry, D.A., Fristedt, B.: Bandit Problems: Sequential Allocation of Experiments. Springer, Netherlands (1985). https:\/\/doi.org\/10.1007\/978-94-015-3711-7"},{"key":"10_CR7","doi-asserted-by":"publisher","first-page":"122","DOI":"10.1006\/aama.1996.0007","volume":"17","author":"AN Burnetas","year":"1996","unstructured":"Burnetas, A.N., Katehakis, M.N.: Optimal adaptive policies for sequential allocation problems. Adv. Appl. Math. 17, 122\u2013142 (1996)","journal-title":"Adv. Appl. Math."},{"issue":"1","key":"10_CR8","doi-asserted-by":"publisher","first-page":"222","DOI":"10.1287\/moor.22.1.222","volume":"22","author":"AN Burnetas","year":"1997","unstructured":"Burnetas, A.N., Katehakis, M.N.: Optimal adaptive policies for Markov decision processes. Math. Oper. Res. 22(1), 222\u2013255 (1997)","journal-title":"Math. Oper. Res."},{"key":"10_CR9","first-page":"154","volume":"18","author":"W Cowan","year":"2017","unstructured":"Cowan, W., Honda, J., Katehakis, M.N.: Normal bandits of unknown means and variances. J. Mach. Learn. Res. 18, 154 (2017)","journal-title":"J. Mach. Learn. Res."},{"key":"10_CR10","unstructured":"Cowan, W., Katehakis, M.N.: Minimal-exploration allocation policies: Asymptotic, almost sure, arbitrarily slow growing regret. arXiv:1505.02865v2 (2015)"},{"key":"10_CR11","unstructured":"Duff, M.O.: Q-Learning for bandit problems. In Proceedings of the Twelfth International Conference on Machine Learning, 32 p. CMPSCI Technical Report 95\u201326 (1995)"},{"key":"10_CR12","doi-asserted-by":"crossref","unstructured":"Fu, J., Nazarathy, Y., Moka, S., Taylor, P.G.: Towards q-learning the whittle index for restless bandits. In: 2019 Australian New Zealand Control Conference (ANZCC), pp. 249\u2013254 (2019)","DOI":"10.1109\/ANZCC47194.2019.8945748"},{"issue":"2","key":"10_CR13","doi-asserted-by":"crossref","first-page":"148","DOI":"10.1111\/j.2517-6161.1979.tb01068.x","volume":"41","author":"JC Gittins","year":"1979","unstructured":"Gittins, J.C.: Bandit processes and dynamic allocation indices. J. Roy. Stat. Soc. Ser. B 41(2), 148\u2013177 (1979)","journal-title":"J. Roy. Stat. Soc. Ser. B"},{"key":"10_CR14","volume-title":"Multi-Armed Bandit Allocation Indices","author":"JC Gittins","year":"1989","unstructured":"Gittins, J.C.: Multi-Armed Bandit Allocation Indices. Wiley, New York (1989)"},{"key":"10_CR15","doi-asserted-by":"crossref","unstructured":"Gittins, J.C., Glazebrook, K., Weber, R.: Multi-Armed Bandit Allocation Indices. Wiley, New York (2011)","DOI":"10.1002\/9780470980033"},{"key":"10_CR16","unstructured":"Gittins, J.C., Jones, D.M.: A dynamic allocation index for the sequential design of experiments. In: Gani, J. (ed.) Progress in Statistics, pp. 241\u2013266. North-Holland, Amsterdam (1974)"},{"issue":"1","key":"10_CR17","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1287\/moor.1080.0342","volume":"34","author":"K Glazebrook","year":"2009","unstructured":"Glazebrook, K., Minty, R.J.: A generalized Gittins index for a class of multiarmed bandits with general resource requirements. Math. Oper. Res. 34(1), 26\u201344 (2009)","journal-title":"Math. Oper. Res."},{"key":"10_CR18","doi-asserted-by":"publisher","first-page":"876","DOI":"10.1214\/10-AAP705","volume":"21","author":"KD Glazebrook","year":"2011","unstructured":"Glazebrook, K.D., Hodge, D.J., Kirkbride, C.: General notions of indexability for queueing control and asset management. Ann. Appl. Probab. 21, 876\u2013907 (2011)","journal-title":"Ann. Appl. Probab."},{"issue":"5","key":"10_CR19","doi-asserted-by":"publisher","first-page":"407","DOI":"10.1007\/s10951-013-0325-1","volume":"17","author":"KD Glazebrook","year":"2013","unstructured":"Glazebrook, K.D., Hodge, D.J., Kirkbride, C., Minty, R.J.: Stochastic scheduling: a short history of index policies and new approaches to index generation for dynamic resource allocation. J. Sched. 17(5), 407\u2013425 (2013). https:\/\/doi.org\/10.1007\/s10951-013-0325-1","journal-title":"J. Sched."},{"issue":"4","key":"10_CR20","doi-asserted-by":"publisher","first-page":"769","DOI":"10.1287\/opre.1070.0444","volume":"55","author":"KD Glazebrook","year":"2007","unstructured":"Glazebrook, K.D., Kirkbride, C., Mitchell, H.M., Gaver, D.P., Jacobs, P.A.: Index policies for shooting problems. Oper. Res. 55(4), 769\u2013781 (2007)","journal-title":"Oper. Res."},{"issue":"4","key":"10_CR21","doi-asserted-by":"publisher","first-page":"975","DOI":"10.1287\/opre.1080.0632","volume":"57","author":"KD Glazebrook","year":"2009","unstructured":"Glazebrook, K.D., Kirkbride, C., Ouenniche, J.: Index policies for the admission control and routing of impatient customers to heterogeneous service stations. Oper. Res. 57(4), 975\u2013989 (2009)","journal-title":"Oper. Res."},{"issue":"3","key":"10_CR22","doi-asserted-by":"publisher","first-page":"696","DOI":"10.1287\/opre.2014.1272","volume":"62","author":"D Graczov\u00e1","year":"2014","unstructured":"Graczov\u00e1, D., Jacko, P.: Generalized restless bandits and the knapsack problem for perishable inventories. INFORMS Oper. Res. 62(3), 696\u2013711 (2014)","journal-title":"INFORMS Oper. Res."},{"issue":"2","key":"10_CR23","doi-asserted-by":"publisher","first-page":"202","DOI":"10.1287\/mksc.1080.0459","volume":"28","author":"JR Hauser","year":"2009","unstructured":"Hauser, J.R., Urban, G.L., Liberali, G., Braun, M.: Website morphing. Market. Sci. 28(2), 202\u2013223 (2009)","journal-title":"Market. Sci."},{"issue":"3","key":"10_CR24","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1239\/aap\/1444308876","volume":"47","author":"DJ Hodge","year":"2015","unstructured":"Hodge, D.J., Glazebrook, K.D.: On the asymptotic optimality of greedy index heuristics for multi-action restless bandits. Adv. Appl. Probab. 47(3), 652\u2013667 (2015)","journal-title":"Adv. Appl. Probab."},{"key":"10_CR25","unstructured":"Jacko, P.: Dynamic Priority Allocation in Restless Bandit Models. Lambert Academic Publishing, Sunnyvale (2010)"},{"issue":"1","key":"10_CR26","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/s10479-013-1312-9","volume":"241","author":"P Jacko","year":"2016","unstructured":"Jacko, P.: Resource capacity allocation to stochastic dynamic competitors: knapsack problem for perishable items and index-knapsack heuristic. Ann. Oper. Res. 241(1), 83\u2013107 (2016)","journal-title":"Ann. Oper. Res."},{"issue":"2","key":"10_CR27","doi-asserted-by":"publisher","first-page":"842","DOI":"10.1214\/17-AOS1569","volume":"46","author":"E Kaufmann","year":"2018","unstructured":"Kaufmann, E.: On Bayesian index policies for sequential resource allocation. Ann. Stat. 46(2), 842\u2013865 (2018)","journal-title":"Ann. Stat."},{"issue":"6","key":"10_CR28","doi-asserted-by":"publisher","first-page":"3812","DOI":"10.1109\/TNET.2016.2562564","volume":"24","author":"M Larra\u00f1aga","year":"2016","unstructured":"Larra\u00f1aga, M., Ayesta, U., Verloop, I.M.: Dynamic control of birth-and-death restless bandits: application to resource-allocation problems. IEEE\/ACM Trans. Networking 24(6), 3812\u20133825 (2016)","journal-title":"IEEE\/ACM Trans. Networking"},{"key":"10_CR29","first-page":"1","volume":"49","author":"T Lattimore","year":"2016","unstructured":"Lattimore, T.: Regret analysis of the finite-horizon Gittins index strategy for multi-armed bandits. J. Mach. Learn. Res. 49, 1\u201332 (2016)","journal-title":"J. Mach. Learn. Res."},{"issue":"2","key":"10_CR30","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1007\/s11750-007-0025-0","volume":"15","author":"J Ni\u00f1o-Mora","year":"2007","unstructured":"Ni\u00f1o-Mora, J.: Dynamic priority allocation via restless bandit marginal productivity indices. TOP 15(2), 161\u2013198 (2007)","journal-title":"TOP"},{"key":"10_CR31","doi-asserted-by":"publisher","first-page":"62","DOI":"10.1016\/j.tcs.2014.09.026","volume":"558","author":"R Ortner","year":"2014","unstructured":"Ortner, R., Ryabko, D., Auer, P., Munos, R.: Regret bounds for restless Markov bandits. Theor. Comput. Sci. 558, 62\u201376 (2014)","journal-title":"Theor. Comput. Sci."},{"issue":"2","key":"10_CR32","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1287\/moor.24.2.293","volume":"24","author":"CH Papadimitriou","year":"1999","unstructured":"Papadimitriou, C.H., Tsitsiklis, J.N.: The complexity of optimal queueing network. Math. Oper. Res. 24(2), 293\u2013305 (1999)","journal-title":"Math. Oper. Res."},{"issue":"4","key":"10_CR33","doi-asserted-by":"publisher","first-page":"1221","DOI":"10.1287\/moor.2014.0650","volume":"39","author":"D Russo","year":"2014","unstructured":"Russo, D., van Roy, B.: Learning to optimize via posterior sampling. Math. Oper. Res. 39(4), 1221\u20131243 (2014)","journal-title":"Math. Oper. Res."},{"key":"10_CR34","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (1998)"},{"issue":"4","key":"10_CR35","doi-asserted-by":"publisher","first-page":"1947","DOI":"10.1214\/15-AAP1137","volume":"26","author":"IM Verloop","year":"2016","unstructured":"Verloop, I.M.: Asymptotically optimal priority policies for indexable and nonindexable restless bandits. Ann. Appl. Probab. 26(4), 1947\u20131995 (2016)","journal-title":"Ann. Appl. Probab."},{"issue":"2","key":"10_CR36","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1214\/14-STS504","volume":"30","author":"SS Villar","year":"2015","unstructured":"Villar, S.S., Bowden, J., Wason, J.: Multi-armed bandit models for the optimal design of clinical trials: benefits and challenges. Stat. Sci. 30(2), 199\u2013215 (2015)","journal-title":"Stat. Sci."},{"key":"10_CR37","doi-asserted-by":"publisher","first-page":"969","DOI":"10.1111\/biom.12337","volume":"71","author":"SS Villar","year":"2015","unstructured":"Villar, S.S., Wason, J., Bowden, J.: Response-adaptive randomization for multi-arm clinical trials using the forward looking Gittins index rule. Biometrics 71, 969\u2013978 (2015)","journal-title":"Biometrics"},{"key":"10_CR38","unstructured":"Watkins, C.J.C.H.: Learning From Delayed Rewards. PhD thesis, Cambridge University, UK (1989)"},{"issue":"3","key":"10_CR39","doi-asserted-by":"publisher","first-page":"637","DOI":"10.2307\/3214547","volume":"27","author":"R Weber","year":"1990","unstructured":"Weber, R., Weiss, G.: On an index policy for restless bandits. J. Appl. Probab. 27(3), 637\u2013648 (1990)","journal-title":"J. Appl. Probab."},{"key":"10_CR40","doi-asserted-by":"crossref","unstructured":"Whittle, P.: Restless bandits: activity allocation in a changing world. In: Gani, J. (ed.), A Celebration of Applied Probability, Journal of Applied Probability, vol. 25, Issue A, pp. 287\u2013298 (1988)","DOI":"10.2307\/3214163"}],"container-title":["Lecture Notes of the Institute for Computer Sciences, Social Informatics and Telecommunications Engineering","Performance Evaluation Methodologies and Tools"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-92511-6_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,13]],"date-time":"2024-09-13T23:02:08Z","timestamp":1726268528000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-92511-6_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030925109","9783030925116"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-92511-6_10","relation":{},"ISSN":["1867-8211","1867-822X"],"issn-type":[{"value":"1867-8211","type":"print"},{"value":"1867-822X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"8 December 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"VALUETOOLS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"EAI International Conference on Performance Evaluation Methodologies and Tools","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 October 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31 October 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"valuetools2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Confy +","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"32","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"16","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"50% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}