{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,14]],"date-time":"2025-06-14T04:07:31Z","timestamp":1749874051818,"version":"3.41.0"},"reference-count":32,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001809","name":"Zhejiang Provincial Natural Science Foundation of China","doi-asserted-by":"publisher","award":["LR23F010005"],"award-info":[{"award-number":["LR23F010005"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Program of China","award":["2024YFE0200600"],"award-info":[{"award-number":["2024YFE0200600"]}]},{"name":"National Key Laboratory of Wireless Communications Foundation","award":["2023KP01601"],"award-info":[{"award-number":["2023KP01601"]}]},{"DOI":"10.13039\/501100004502","name":"Big Data and Intelligent Computing Key Laboratory of Chongqing University of Posts and Telecommunications","doi-asserted-by":"publisher","award":["BDIC-2023-B-001"],"award-info":[{"award-number":["BDIC-2023-B-001"]}],"id":[{"id":"10.13039\/501100004502","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/access.2025.3577997","type":"journal-article","created":{"date-parts":[[2025,6,9]],"date-time":"2025-06-09T17:35:48Z","timestamp":1749490548000},"page":"100889-100903","source":"Crossref","is-referenced-by-count":0,"title":["Variation-Aware Bernstein-Based Upper Confidence Reinforcement Learning for Environment With Endogenous and Exogenous Uncertainty"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7414-5638","authenticated-orcid":false,"given":"Ruoqi","family":"Wen","sequence":"first","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4297-5060","authenticated-orcid":false,"given":"Rongpeng","family":"Li","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2024.3452790"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/IOTM.001.2400088"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1155\/2024\/8845070"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2021.3063822"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.aej.2022.08.017"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.4613523"},{"key":"ref7","article-title":"Deep reinforcement learning for dynamic resource allocation in wireless networks","author":"Malhotra","year":"2025","journal-title":"arXiv:2502.01129"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/tai.2024.3375258"},{"article-title":"Exploration-exploitation dilemma in reinforcement learning under various form of prior knowledge","year":"2019","author":"Fruit","key":"ref9"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/0196-8858(85)90002-8"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1023\/A:1017984413808"},{"issue":"51","key":"ref13","first-page":"1563","article-title":"Near-optimal regret bounds for reinforcement learning","volume":"11","author":"Jaksch","year":"2010","journal-title":"J. Mach. Learn. Res."},{"key":"ref14","first-page":"35","article-title":"REGAL: A regularization based algorithm for reinforcement learning in weakly communicating MDPs","volume-title":"Proc. UAI","author":"Bartlett"},{"key":"ref15","article-title":"Improved analysis of UCRL2 with empirical Bernstein inequality","author":"Fruit","year":"2020","journal-title":"arXiv:2007.05456"},{"key":"ref16","first-page":"2442","article-title":"Bootstrapping upper confidence bound","volume-title":"Proc. NIPS","author":"Hao"},{"key":"ref17","first-page":"4890","article-title":"Exploration bonus for regret minimization in discrete and continuous average reward MDPs","volume-title":"Proc. NIPS","volume":"32","author":"Qian"},{"key":"ref18","first-page":"1093","article-title":"Tightening exploration in upper confidence reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Bourel"},{"key":"ref19","first-page":"1334","article-title":"Towards achieving sub-linear regret and hard constraint violation in model-free RL","volume-title":"Proc. AISTATS","author":"Ghosh"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20690"},{"key":"ref21","first-page":"288:1","article-title":"A new look at dynamic regret for non-stationary stochastic bandits","volume":"24","author":"Abbasi-Yadkori","year":"2022","journal-title":"J. Mach. Learn. Res."},{"issue":"98","key":"ref22","first-page":"1","article-title":"Adaptivity and non-stationarity: Problem-dependent dynamic regret for online convex optimization","volume":"25","author":"Zhao","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"ref23","first-page":"3924","article-title":"Online Markov decision processes with time-varying transition probabilities and rewards","volume-title":"Proc. ICML","author":"Li"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.23919\/ACC.2019.8815000"},{"key":"ref25","first-page":"509","article-title":"Variational regret bounds for reinforcement learning","volume-title":"Proc. UAI","author":"Gajane"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.3261050"},{"key":"ref27","first-page":"1841","article-title":"Reinforcement learning for non-stationary Markov decision processes: The blessing of (more) optimism","volume-title":"Proc. ICML","author":"Cheung"},{"key":"ref28","first-page":"1498","article-title":"Non-stationary reinforcement learning without prior knowledge: An optimal black-box approach","volume-title":"Proc. COLT","author":"Wei"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/j.tcs.2009.01.016"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1002\/SERIES1345"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1051\/ro\/1985190100711"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780199535255.001.0001"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10820123\/11028620.pdf?arnumber=11028620","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,13]],"date-time":"2025-06-13T17:52:26Z","timestamp":1749837146000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11028620\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":32,"URL":"https:\/\/doi.org\/10.1109\/access.2025.3577997","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2025]]}}}