{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T15:54:57Z","timestamp":1769529297577,"version":"3.49.0"},"reference-count":50,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T00:00:00Z","timestamp":1614556800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T00:00:00Z","timestamp":1614556800000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T00:00:00Z","timestamp":1614556800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T00:00:00Z","timestamp":1614556800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006831","name":"United States Air Force through Air Force","doi-asserted-by":"publisher","award":["FA8702-15-D-0001"],"award-info":[{"award-number":["FA8702-15-D-0001"]}],"id":[{"id":"10.13039\/100006831","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["AST-1547331"],"award-info":[{"award-number":["AST-1547331"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["CNS-1701964"],"award-info":[{"award-number":["CNS-1701964"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000183","name":"Army Research Office","doi-asserted-by":"publisher","award":["W911NF-17-1-0508"],"award-info":[{"award-number":["W911NF-17-1-0508"]}],"id":[{"id":"10.13039\/100000183","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Inform. Theory"],"published-print":{"date-parts":[[2021,3]]},"DOI":"10.1109\/tit.2021.3054854","type":"journal-article","created":{"date-parts":[[2021,1,27]],"date-time":"2021-01-27T21:12:47Z","timestamp":1611781967000},"page":"1759-1781","source":"Crossref","is-referenced-by-count":21,"title":["Learning Algorithms for Minimizing Queue Length Regret"],"prefix":"10.1109","volume":"67","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2128-5004","authenticated-orcid":false,"given":"Thomas","family":"Stahlbuhk","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Brooke","family":"Shrader","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eytan","family":"Modiano","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2002.808401"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/26.780463"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1561\/1300000050"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TCNS.2016.2635380"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2010.2062509"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ITW.2013.6691221"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/18.212277"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/9.182479"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2014.54"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TCOMM.2016.2569093"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2011.2181864"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/DYSPAN.2010.5457857"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2014.2302471"},{"key":"ref2","first-page":"285","article-title":"On the likelihood that one unknown probability exceeds another in view of the evidence of two samples","volume":"25","author":"thompson","year":"1933","journal-title":"Bull Amer Math Soc"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT.2018.8437817"},{"key":"ref20","first-page":"1","article-title":"Safe exploration in Markov decision processes","author":"modolvan","year":"2012","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref22","first-page":"1","article-title":"Safe exploration and optimization of constrained MDPs using Gaussian processes","author":"wachi","year":"2018","journal-title":"Proc AAAI"},{"key":"ref21","first-page":"1","article-title":"Safe model-based reinforcement learning with stability guarantees","author":"berkenkamp","year":"2017","journal-title":"Proc NIPS"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2016.7524557"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2011.110406"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2014.6848225"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.23919\/WIOPT.2017.7959911"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1145\/3179414"},{"key":"ref10","author":"sutton","year":"2018","journal-title":"Reinforcement Learning An Introduction"},{"key":"ref11","first-page":"49","article-title":"Logarithmic online regret bounds for undiscounted reinforcement learning","author":"auer","year":"2006","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref40","first-page":"1804","article-title":"The impact of imperfect scheduling on cross-layer rate control in wireless networks","author":"lin","year":"2005","journal-title":"Proc IEEE 24th Annu Joint Conf IEEE Comput Commun Soc"},{"key":"ref12","first-page":"1563","article-title":"Near-optimal regret bounds for reinforcement learning","volume":"11","author":"jaksch","year":"2010","journal-title":"J Mach Learn Res"},{"key":"ref13","first-page":"263","article-title":"Minimax regret bounds for reinforcement learning","author":"azar","year":"2017","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref14","first-page":"1","article-title":"Near optimal exploration-exploitation in non-communicating Markov decision processes","author":"fruit","year":"2018","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1561\/2200000070"},{"key":"ref16","first-page":"943","article-title":"A Bayesian framework for reinforcement learning","author":"strens","year":"2000","journal-title":"Proc Conf Mach Learn"},{"key":"ref17","first-page":"1","article-title":"(More) efficient reinforcement learning via posterior sampling","author":"osband","year":"2013","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref18","first-page":"604","article-title":"Near-optimal reinforcement learning in factored MDPs","author":"osband","year":"2014","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref19","first-page":"2701","article-title":"Why is posterior sampling better than optimism for reinforcement learning?","author":"osband","year":"2017","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/0196-8858(85)90002-8"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1561\/2200000024"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1023\/A:1013689704352"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.2307\/1427934"},{"key":"ref8","first-page":"1669","article-title":"Regret of queueing bandits","author":"krishnasamy","year":"2016","journal-title":"Proc Neural Inf Process Syst"},{"key":"ref7","first-page":"359","article-title":"The KL-UCB algorithm for bounded stochastic bandits and beyond","author":"garivier","year":"2011","journal-title":"Proc Conf Learn Theory"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1016\/j.adhoc.2018.10.006"},{"key":"ref9","author":"bertsekas","year":"2019","journal-title":"REINFORCEMENT LEARNING AND OPTIMAL CONTROL"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2018.8486307"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2018.8485833"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2017.8056983"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3323679.3326520"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2017.2783846"},{"key":"ref41","first-page":"1","article-title":"Cross-layer congestion control, routing and scheduling design in ad hoc wireless networks","author":"chen","year":"2006","journal-title":"Proc IEEE InfoCom"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2016.2533496"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2011.083111.110194"}],"container-title":["IEEE Transactions on Information Theory"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/18\/9356438\/9336686-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/18\/9356438\/09336686.pdf?arnumber=9336686","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T14:54:28Z","timestamp":1652194468000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9336686\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,3]]},"references-count":50,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tit.2021.3054854","relation":{},"ISSN":["0018-9448","1557-9654"],"issn-type":[{"value":"0018-9448","type":"print"},{"value":"1557-9654","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,3]]}}}