{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T15:55:06Z","timestamp":1783007706933,"version":"3.54.5"},"reference-count":87,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2022YFB3102100"],"award-info":[{"award-number":["2022YFB3102100"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62293511"],"award-info":[{"award-number":["62293511"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402379"],"award-info":[{"award-number":["62402379"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402431"],"award-info":[{"award-number":["62402431"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62441618"],"award-info":[{"award-number":["62441618"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62172243"],"award-info":[{"award-number":["62172243"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72571007"],"award-info":[{"award-number":["72571007"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Gravitation research program"},{"DOI":"10.13039\/501100003246","name":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek","doi-asserted-by":"publisher","award":["024.006.037"],"award-info":[{"award-number":["024.006.037"]}],"id":[{"id":"10.13039\/501100003246","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Zhejiang University Education Foundation Qizhen Scholar Foundation"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Dependable and Secure Comput."],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1109\/tdsc.2025.3611881","type":"journal-article","created":{"date-parts":[[2025,10,6]],"date-time":"2025-10-06T17:40:56Z","timestamp":1759772456000},"page":"1514-1527","source":"Crossref","is-referenced-by-count":4,"title":["Revealing the Risk of Hyper-Parameter Leakage in Deep Reinforcement Learning Models"],"prefix":"10.1109","volume":"23","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-9028-9326","authenticated-orcid":false,"given":"Linkang","family":"Du","sequence":"first","affiliation":[{"name":"Xi&#x2019;an Jiaotong University, Xi&#x2019;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7208-3392","authenticated-orcid":false,"given":"Zhikun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1128-7989","authenticated-orcid":false,"given":"Min","family":"Chen","sequence":"additional","affiliation":[{"name":"Vrije Universiteit Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5790-5025","authenticated-orcid":false,"given":"Mingyang","family":"Sun","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4268-372X","authenticated-orcid":false,"given":"Shouling","family":"Ji","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4221-2162","authenticated-orcid":false,"given":"Peng","family":"Cheng","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3155-3145","authenticated-orcid":false,"given":"Jiming","family":"Chen","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7130-9211","authenticated-orcid":false,"given":"Michael","family":"Backes","sequence":"additional","affiliation":[{"name":"CISPA Helmholtz Center for Information Security, Saarbr&#x00FC;cken, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3612-7348","authenticated-orcid":false,"given":"Yang","family":"Zhang","sequence":"additional","affiliation":[{"name":"CISPA Helmholtz Center for Information Security, Saarbr&#x00FC;cken, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Deep reinforcement learning for robotic manipulation - The state of the art","author":"Amarjyoti","year":"2017"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i5.20463"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2743240"},{"key":"ref4","article-title":"Poisoning deep reinforcement learning agents with in-distribution Triggers","author":"Ashcraft","year":"2021"},{"key":"ref5","article-title":"Adversarial exploitation of policy imitation","author":"Behzadan","year":"2019"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-62416-7_19"},{"key":"ref7","article-title":"Dota 2 with large scale deep reinforcement learning","author":"Berner","year":"2019"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3433210.3453090"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3433210.3453090"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.3233\/faia240734"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2522401"},{"key":"ref12","article-title":"OpenAI baselines","author":"Dhariwal","year":"2017"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/URAI.2018.8441797"},{"key":"ref14","article-title":"Contrastive explanations for comparing preferences of reinforcement learning agents","author":"Gajcin","year":"2021"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3243734.3243834"},{"key":"ref16","first-page":"1","article-title":"Adversarial policies: Attacking deep reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gleave"},{"key":"ref17","first-page":"1","article-title":"Adversarial policies: Attacking deep reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gleave"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3270860"},{"key":"ref19","first-page":"979","article-title":"You are who you know and how you behave: Attribute inference attacks via users\u2019 social friends and behaviors","volume-title":"Proc. USENIX Conf. Secur. Symp.","author":"Gong"},{"key":"ref20","first-page":"1","article-title":"Explaining and harnessing adversarial examples","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Goodfellow"},{"key":"ref21","first-page":"3910","article-title":"Adversarial policy learning in two-player competitive games","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Guo"},{"key":"ref22","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref23","article-title":"Emergence of locomotion behaviours in rich environments","author":"Heess","year":"2017"},{"key":"ref24","article-title":"Stable baselines","author":"Hill","year":"2018"},{"key":"ref25","first-page":"4572","article-title":"Generative adversarial imitation learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Ho"},{"key":"ref26","article-title":"Membership inference attacks against vision-language models","author":"Hu","year":"2025"},{"key":"ref27","article-title":"Adversarial attacks on neural network policies","author":"Huang","year":"2017"},{"key":"ref28","first-page":"1","article-title":"Adversarial attacks on neural network policies","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Huang"},{"key":"ref29","first-page":"7311","article-title":"THEMIS: Towards practical intellectual property protection for post-deployment on-device deep learning models","volume-title":"Proc. USENIX Conf. Secur. Symp.","author":"Huang"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/tai.2021.3111139"},{"key":"ref31","first-page":"1345","article-title":"High accuracy and high fidelity extraction of neural networks","volume-title":"Proc. USENIX Conf. Secur. Symp.","author":"Jagielski"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18072.2020.9218663"},{"key":"ref33","first-page":"1008","article-title":"Actor-critic algorithms","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Konda"},{"key":"ref34","first-page":"1","article-title":"Delving into adversarial attacks on deep policies","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kos"},{"key":"ref35","article-title":"Thieves on sesame street! model extraction of BERT-based APIs","author":"Krishna","year":"2019"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1201\/9781351251389-8"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-022-10348-5"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/525"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/525"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TDSC.2022.3180828"},{"key":"ref41","article-title":"Rethinking adversarial policies: A generalized attack formulation and provable defense in RL","author":"Liu","year":"2023"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2021.3138611"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3658644.3670293"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.06083"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/SP46214.2022.9833623"},{"key":"ref46","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih"},{"key":"ref47","article-title":"Playing atari with deep reinforcement learning","author":"Mnih","year":"2013"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1145\/3640312"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-28954-6_7"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00509"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.65109\/VYTC7933"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.65109\/RNCT3583"},{"key":"ref53","volume-title":"Conditioned Reflexes and Psychiatry","volume":"2","author":"Pavlov","year":"1941"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i7.20772"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1145\/3624010"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CDC49753.2023.10383661"},{"key":"ref57","first-page":"1291","article-title":"Updates-leak: Data set inference and reconstruction attacks in online learning","volume-title":"Proc. USENIX Conf. Secur. Symp.","author":"Salem"},{"key":"ref58","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman"},{"key":"ref59","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2017.41"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1145\/3372297.3417270"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6047"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.5555\/3241094.3241142"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICSC64641.2025.00040"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2018.00038"},{"key":"ref67","first-page":"11323","article-title":"Privacy-preserving Q-learning with functional noise in continuous spaces","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Wang"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/509"},{"key":"ref69","article-title":"Security and ownership verification in deep reinforcement learning","author":"Wang","year":"2022"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.14722\/ndss.2025.240168"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3207346"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2021.3114024"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737416"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.92"},{"key":"ref75","article-title":"Learning from delayed rewards","author":"Watkins","year":"1989"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.14722\/ndss.2024.24441"},{"key":"ref78","first-page":"1883","article-title":"Adversarial policy training against deep reinforcement learning","volume-title":"Proc. USENIX Conf. Secur. Symp.","author":"Wu"},{"key":"ref79","first-page":"5459","article-title":"Learning to explore via meta-policy gradient","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Xu"},{"key":"ref80","first-page":"2209","article-title":"OblivGNN: Oblivious inference on transductive and inductive graph neural network","volume-title":"Proc. USENIX Conf. Secur. Symp.","author":"Xu"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2021.3133824"},{"key":"ref82","article-title":"Design of intentional backdoors in sequential models","author":"Yang","year":"2019"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM48099.2022.10000751"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487174"},{"key":"ref85","article-title":"Thief, beware of what get you there: Towards understanding model extraction attack","author":"Zhang","year":"2021"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00033"},{"key":"ref87","first-page":"62682","article-title":"Stealthy imitation: Reward-guided environment-free policy stealing","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhuang"}],"container-title":["IEEE Transactions on Dependable and Secure Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/8858\/11354469\/11193654.pdf?arnumber=11193654","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,19]],"date-time":"2026-01-19T20:57:46Z","timestamp":1768856266000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11193654\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1]]},"references-count":87,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/tdsc.2025.3611881","relation":{},"ISSN":["1545-5971","1941-0018","2160-9209"],"issn-type":[{"value":"1545-5971","type":"print"},{"value":"1941-0018","type":"electronic"},{"value":"2160-9209","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1]]}}}