{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T20:54:02Z","timestamp":1781816042324,"version":"3.54.5"},"reference-count":110,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Fundamental and Interdisciplinary Disciplines Breakthrough Plan of the Ministry of Education of China","award":["JYB2025XDXM116"],"award-info":[{"award-number":["JYB2025XDXM116"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62273305"],"award-info":[{"award-number":["62273305"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62293511"],"award-info":[{"award-number":["62293511"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100012774","name":"Innovationsfonden","doi-asserted-by":"publisher","award":["1063-00031B"],"award-info":[{"award-number":["1063-00031B"]}],"id":[{"id":"10.13039\/100012774","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Netw. Sci. Eng."],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/tnse.2026.3695143","type":"journal-article","created":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T19:50:48Z","timestamp":1779220248000},"page":"9807-9826","source":"Crossref","is-referenced-by-count":0,"title":["Near-Optimal Reinforcement Learning With Shuffle Differential Privacy"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7434-8841","authenticated-orcid":false,"given":"Shaojie","family":"Bai","sequence":"first","affiliation":[{"name":"College of Control Science and Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1934-7421","authenticated-orcid":false,"given":"Mohammad Sadegh","family":"Talebi","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Copenhagen, Copenhagen, Denmark"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6816-6459","authenticated-orcid":false,"given":"Chengcheng","family":"Zhao","sequence":"additional","affiliation":[{"name":"College of Control Science and Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4221-2162","authenticated-orcid":false,"given":"Peng","family":"Cheng","sequence":"additional","affiliation":[{"name":"College of Control Science and Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3155-3145","authenticated-orcid":false,"given":"Jiming","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Control Science and Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2019.2916583"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2021.3117565"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ITW.2013.6691221"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2016.7437026"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3640312"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1561\/9781601988195"},{"key":"ref7","first-page":"9754","article-title":"Private reinforcement learning with pac and regret guarantees","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Vietri","year":"2020"},{"key":"ref8","first-page":"235","article-title":"Differentially private no-regret exploration in adversarial markov decision processes","volume-title":"Proc. Conf. Uncertainty Artif. Intell.","volume":"244","author":"Bai","year":"2024"},{"key":"ref9","first-page":"10561","article-title":"Local differential privacy for regret minimization in reinforcement learning","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","author":"Garcelon","year":"2021"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i6.20588"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-17653-2_13"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3132747.3132769"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-26951-7_22"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1214\/15-AOS1381"},{"key":"ref15","first-page":"503","article-title":"Batched multi-armed bandits problem","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","author":"Gao","year":"2019"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2018.2876749"},{"key":"ref17","first-page":"18031","article-title":"Sample-efficient reinforcement learning with loglog (t) switching cost","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Qiao","year":"2022"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2025.3546100"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2024.3371384"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2023.3321048"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2024.3432765"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2024.3350710"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3224431"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2024.3391289"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2024.3446667"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2022.3167949"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2022.3157274"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2022.3185092"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/j.apenergy.2022.119688"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TCCN.2023.3237578"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2017.2747409"},{"key":"ref32","volume-title":"Markov Decision Processes: Discrete Stochastic Dyn. Program.","author":"Puterman","year":"2014"},{"key":"ref33","first-page":"263","article-title":"Minimax regret bounds for reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Azar","year":"2017"},{"key":"ref34","first-page":"14433","article-title":"Worst-case regret bounds for exploration via randomized value functions","volume":"32","author":"Russo","journal-title":"in Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref35","first-page":"3003","article-title":"(More) efficient reinforcement learning via posterior sampling","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","author":"Osband","year":"2013"},{"issue":"11","key":"ref36","first-page":"2413","article-title":"Reinforcement learning in finite MDPs: PAC analysis","volume":"10","author":"Strehl","year":"2009","journal-title":"J. Mach. Learn. Res."},{"key":"ref37","first-page":"1563","article-title":"Near-optimal regret bounds for reinforcement learning","volume":"11","author":"Jaksch","year":"2010","journal-title":"J. Mach. Learn. Res."},{"key":"ref38","first-page":"1056","article-title":"Tightening exploration in upper confidence reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bourel","year":"2020"},{"key":"ref39","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020"},{"key":"ref40","first-page":"7304","article-title":"Tighter problem-dependent regret bounds in reinforcement learning without domain knowledge using value function bounds","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zanette","year":"2019"},{"key":"ref41","first-page":"1","article-title":"Non-asymptotic gap-dependent regret bounds for tabular MDPs","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Simchowitz","year":"2019"},{"key":"ref42","first-page":"1","article-title":"Is Q-learning provably efficient","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Jin"},{"key":"ref43","first-page":"1","article-title":"Tight regret bounds for model-based reinforcement learning with greedy policies","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Efroni","year":"2019"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1561\/9781680834710"},{"issue":"124","key":"ref45","first-page":"1","article-title":"Deep exploration via randomized value functions","volume":"20","author":"Osband","year":"2019","journal-title":"J. Mach. Learn. Res."},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i8.16813"},{"key":"ref47","first-page":"6358","article-title":"Near-optimal randomized exploration for tabular Markov decision processes","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Xiong","year":"2022"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1785"},{"key":"ref49","first-page":"578","article-title":"Episodic reinforcement learning in finite MDPs: Minimax lower bounds revisited","volume-title":"Proc. Algorithmic Learn. Theory","author":"Domingues","year":"2021"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2022.1309"},{"key":"ref51","first-page":"10978","article-title":"Learning near optimal policies with low inherent bellman error","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zanette","year":"2020"},{"key":"ref52","first-page":"987","article-title":"VOQL: Towards optimal regret in model-free RL with nonlinear function approximation","volume-title":"Proc. Conf. Learn. Theory","author":"Agarwal","year":"2023"},{"key":"ref53","first-page":"1","article-title":"Provably efficient Q-learning with low switching cost","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","author":"Bai","year":"2019"},{"key":"ref54","first-page":"1","article-title":"Sample-efficiency in multi-batch reinforcement learning: The need for dimension-dependent adaptivity","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Johnson","year":"2024"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/11681878_14"},{"key":"ref56","first-page":"2733","article-title":"(Nearly) optimal algorithms for private online learning in full-information and bandit settings","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"26","author":"Thakurta","year":"2013"},{"key":"ref57","first-page":"24.1","article-title":"Differentially private online learning","volume-title":"Proc. Conf. Learn. Theory, Ser.","volume":"23","author":"Jain","year":"2012"},{"key":"ref58","first-page":"32","article-title":"The price of differential privacy for online learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Agarwal","year":"2017"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2023.3329832"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.10896"},{"key":"ref61","first-page":"592","article-title":"(nearly) optimal differentially private stochastic multi-arm bandits","volume-title":"Proc. Conf. Uncertainty Artif. Intell.","author":"Mishra","year":"2015"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10212"},{"key":"ref63","first-page":"5579","article-title":"An optimal private stochastic-mab algorithm based on optimal private stopping rule","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sajed","year":"2019"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2333"},{"key":"ref65","first-page":"1546","article-title":"Optimal rates of (locally) differentially private heavy-tailed multi-armed bandits","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Tao","year":"2022"},{"issue":"314","key":"ref66","first-page":"1","article-title":"Optimal learning policies for differential privacy in multi-armed bandits","volume":"25","author":"Wang","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"ref67","article-title":"Optimal regret of bernoulli bandits under global differential privacy","author":"Azize","year":"2025"},{"key":"ref68","first-page":"844","article-title":"Near-optimal thompson sampling-based algorithms for differentially private stochastic bandits","volume-title":"Proc. Conf. Uncertainty Artif. Intell.","author":"Hu","year":"2022"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/JSAIT.2024.3389954"},{"key":"ref70","first-page":"1","article-title":"Differentially private contextual linear bandits","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","author":"Shariff","year":"2018"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2024.3362978"},{"key":"ref72","first-page":"381","article-title":"Privacy amplification via shuffling for linear contextual bandits","volume-title":"Proc. Algorithmic Learn. Theory","author":"Garcelon","year":"2022"},{"key":"ref73","first-page":"1","article-title":"Shuffle private linear contextual bandits","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Chowdhury","year":"2022"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2022.3142374"},{"key":"ref75","article-title":"Multi-armed bandits with local differential privacy","author":"Ren","year":"2020"},{"key":"ref76","first-page":"26511","article-title":"Generalized linear bandits with local differential privacy","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","volume":"522","author":"Han","year":"2021"},{"key":"ref77","first-page":"24956","article-title":"Differentially private multi-armed bandits in the shuffle model","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","author":"Tenenbaum","year":"2021"},{"key":"ref78","first-page":"1","article-title":"Distributed differential privacy in multi-armed bandits","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chowdhury","year":"2023"},{"issue":"39","key":"ref79","first-page":"1079","article-title":"Action elimination and stopping conditions for the multi-armed bandit and reinforcement learning problems","volume":"7","author":"Even-Dar","year":"2006","journal-title":"J. Mach. Learn. Res."},{"key":"ref80","first-page":"9914","article-title":"Near-optimal differentially private reinforcement learning","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Qiao","year":"2023"},{"key":"ref81","first-page":"37880","article-title":"Differentially private episodic reinforcement learning with heavy-tailed rewards","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"202","author":"Wu","year":"2023"},{"key":"ref82","article-title":"Differentially private exploration in reinforcement learning with linear representation","author":"Luyo","year":"2021"},{"key":"ref83","first-page":"16529","article-title":"Improved regret for differentially private exploration in linear MDP","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ngo","year":"2022"},{"key":"ref84","article-title":"Towards optimal differentially private regret bounds in linear MDPs","author":"Sahu","year":"2025"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1145\/3508028"},{"key":"ref86","first-page":"627","article-title":"Locally differentially private reinforcement learning for linear mixture markov decision processes","volume-title":"Proc. Asian Conf. Mach. Learn.","author":"Liao","year":"2023"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/TCCN.2024.3499342"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/TDSC.2024.3398994"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3427789"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN60899.2024.10650755"},{"key":"ref91","first-page":"1","article-title":"Privacy preserving reinforcement learning for population processes","author":"Yang-Zhao","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.112558"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2023.3312118"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/CCTA60707.2024.10666610"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/CDC42340.2020.9304015"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/CDC51059.2022.9992410"},{"key":"ref97","first-page":"12402","article-title":"Near optimal reward-free reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang","year":"2021"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2620"},{"key":"ref99","first-page":"133743","article-title":"Offline oracle-efficient learning for contextual MDPs via layerwise exploration-exploitation tradeoff","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Qian","year":"2024"},{"key":"ref100","first-page":"1","article-title":"Unifying PAC and regret: Uniform PAC bounds for episodic reinforcement learning","volume-title":"in Proc. Adv. Neural Inf. Process. Syst.","author":"Dann","year":"2017"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1137\/1.9780898719109"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1016\/j.jcss.2007.08.009"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/tnn.1998.712192"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1017\/9781108571401"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1145\/2591796.2591826"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v24i1.7727"},{"key":"ref107","first-page":"1","article-title":"Almost optimal model-free reinforcement learning via reference-advantage decomposition","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang","year":"2020"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.4153\/CJM-1960-030-4"},{"key":"ref109","first-page":"427","article-title":"Revisiting frank-wolfe: Projection-free sparse convex optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Jaggi","year":"2013"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611970791"}],"container-title":["IEEE Transactions on Network Science and Engineering"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6488902\/11264281\/11527031.pdf?arnumber=11527031","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T20:12:11Z","timestamp":1781813531000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11527031\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":110,"URL":"https:\/\/doi.org\/10.1109\/tnse.2026.3695143","relation":{},"ISSN":["2327-4697","2334-329X"],"issn-type":[{"value":"2327-4697","type":"electronic"},{"value":"2334-329X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}