{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T16:01:32Z","timestamp":1784995292259,"version":"3.55.0"},"reference-count":87,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T00:00:00Z","timestamp":1670284800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"published-print":{"date-parts":[[2023,7]]},"DOI":"10.1007\/s10462-022-10348-5","type":"journal-article","created":{"date-parts":[[2022,12,6]],"date-time":"2022-12-06T01:03:07Z","timestamp":1670288587000},"page":"7195-7236","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":31,"title":["New challenges in reinforcement learning: a survey of security and privacy"],"prefix":"10.1007","volume":"56","author":[{"given":"Yunjiao","family":"Lei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dayong","family":"Ye","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sheng","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yulei","family":"Sui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3411-7947","authenticated-orcid":false,"given":"Tianqing","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wanlei","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,12,6]]},"reference":[{"key":"10348_CR1","doi-asserted-by":"crossref","unstructured":"Ahmed U, Lin JC-W, Srivastava G (2021) Privacy-preserving deep reinforcement learning in vehicle adhoc networks. In: IEEE consumer electronics magazine","DOI":"10.1109\/MCE.2021.3088408"},{"key":"10348_CR2","doi-asserted-by":"crossref","unstructured":"Ahmed U, Lin JC-W, Srivastava G, Chen H-C (2022) Deep active reinforcement learning for privacy preserve data mining in 5g environments. J Intell Fuzzy Syst, pp 1\u20138","DOI":"10.3233\/JIFS-219262"},{"key":"10348_CR3","doi-asserted-by":"crossref","first-page":"100235","DOI":"10.1016\/j.cosrev.2020.100235","volume":"36","author":"B Alaya","year":"2020","unstructured":"Alaya B, Laouamer L, Msilini N (2020) Homomorphic encryption systems statement: trends and challenges. Comput Sci Rev 36:100235","journal-title":"Comput Sci Rev"},{"key":"10348_CR4","doi-asserted-by":"crossref","first-page":"103500","DOI":"10.1016\/j.artint.2021.103500","volume":"297","author":"S Arora","year":"2021","unstructured":"Arora S, Doshi P (2021) A survey of inverse reinforcement learning: chalenges, methods and progress. Artif Intell 297:103500","journal-title":"Artif Intell"},{"issue":"6","key":"10348_CR5","doi-asserted-by":"crossref","first-page":"26","DOI":"10.1109\/MSP.2017.2743240","volume":"34","author":"K Arulkumaran","year":"2017","unstructured":"Arulkumaran K, Deisenroth MP, Brundage M, Bharath AA (2017) Deep reinforcement learning: a brief survey. IEEE Signal Process Mag 34(6):26\u201338","journal-title":"IEEE Signal Process Mag"},{"key":"10348_CR6","doi-asserted-by":"crossref","first-page":"834","DOI":"10.1109\/TSMC.1983.6313077","volume":"5","author":"AG Barto","year":"1983","unstructured":"Barto AG, Sutton RS, Anderson CW (1983) Neuronlike adaptive elements that can solve difficult learning control problems. IEEE Trans Syst Man Cybern 5:834\u2013846","journal-title":"IEEE Trans Syst Man Cybern"},{"key":"10348_CR7","doi-asserted-by":"crossref","unstructured":"Behzadan V, Munir A (2017) Vulnerability of deep reinforcement learning to policy induction attacks. In: International conference on machine learning and data mining in pattern recognition. Springer, pp 262\u2013275","DOI":"10.1007\/978-3-319-62416-7_19"},{"key":"10348_CR8","doi-asserted-by":"crossref","DOI":"10.1016\/j.adhoc.2021.102541","volume":"119","author":"A Belhadi","year":"2021","unstructured":"Belhadi A, Djenouri Y, Srivastava G, Jolfaei A, Lin JC-W (2021) Privacy reinforcement learning for faults detection in the smart grid. Ad Hoc Netw 119:102541","journal-title":"Ad Hoc Netw"},{"key":"10348_CR9","doi-asserted-by":"crossref","DOI":"10.1002\/9780470058411","volume-title":"Developing multi-agent systems with JADE","author":"FL Bellifemine","year":"2007","unstructured":"Bellifemine FL, Caire G, Greenwood D (2007) Developing multi-agent systems with JADE. Wiley, Hoboken"},{"key":"10348_CR10","volume-title":"Dynamic programming","author":"R Bellman","year":"1957","unstructured":"Bellman R (1957) Dynamic programming. Princeton University Press, Princeton"},{"key":"10348_CR11","volume-title":"Practical grey-box process identification: theory and applications","author":"TP Bohlin","year":"2006","unstructured":"Bohlin TP (2006) Practical grey-box process identification: theory and applications. Springer, New York"},{"key":"10348_CR12","doi-asserted-by":"crossref","unstructured":"Chan PP, Wang Y, Yeung DS (2020) Adversarial attack against deep reinforcement learning with static reward impact map. In: Proceedings of the 15th ACM Asia conference on computer and communications security, pp 334\u2013343","DOI":"10.1145\/3320269.3384715"},{"issue":"1","key":"10348_CR13","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1186\/s42400-018-0018-3","volume":"2","author":"T Chen","year":"2019","unstructured":"Chen T, Liu J, Xiang Y, Niu W, Tong E, Han Z (2019) Adversarial attack and defense in reinforcement learning-from AI security view. Cybersecurity 2(1):1\u201322","journal-title":"Cybersecurity"},{"issue":"2","key":"10348_CR14","doi-asserted-by":"crossref","first-page":"364","DOI":"10.1109\/TNSE.2021.3117565","volume":"9","author":"M Chen","year":"2021","unstructured":"Chen M, Liu A, Liu W, Ota K, Dong M, Xiong NN (2021a) Rdrl: a recurrent deep reinforcement learning scheme for dynamic spectrum access in reconfigurable wireless networks. IEEE Trans Netw Sci Eng 9(2):364\u2013376","journal-title":"IEEE Trans Netw Sci Eng"},{"key":"10348_CR15","doi-asserted-by":"crossref","DOI":"10.1016\/j.comnet.2021.108186","volume":"195","author":"M Chen","year":"2021","unstructured":"Chen M, Liu W, Wang T, Liu A, Zeng Z (2021b) Edge intelligence computing for mobile augmented reality with deep reinforcement learning approach. Comput Netw 195:108186","journal-title":"Comput Netw"},{"key":"10348_CR16","doi-asserted-by":"crossref","first-page":"1659","DOI":"10.1109\/COMST.2021.3073036","volume":"23","author":"W Chen","year":"2021","unstructured":"Chen W, Qiu X, Cai T, Dai H-N, Zheng Z, Zhang Y (2021c) Deep reinforcement learning for internet of things: a comprehensive survey. IEEE Commun Surv Tutor 23:1659","journal-title":"IEEE Commun Surv Tutor"},{"key":"10348_CR17","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.comcom.2021.04.028","volume":"175","author":"M Chen","year":"2021","unstructured":"Chen M, Wang T, Zhang S, Liu A (2021d) Deep reinforcement learning for computation offloading in mobile edge computing environment. Comput Commun 175:1\u201312","journal-title":"Comput Commun"},{"key":"10348_CR18","doi-asserted-by":"crossref","first-page":"107660","DOI":"10.1016\/j.knosys.2021.107660","volume":"235","author":"M Chen","year":"2022","unstructured":"Chen M, Liu W, Wang T, Zhang S, Liu A (2022) A game-based deep reinforcement learning approach for energy-efficient computation in mec systems. Knowl-Based Syst 235:107660","journal-title":"Knowl-Based Syst"},{"issue":"1","key":"10348_CR19","doi-asserted-by":"crossref","first-page":"799","DOI":"10.1002\/int.22648","volume":"37","author":"Z Cheng","year":"2022","unstructured":"Cheng Z, Ye D, Zhu T, Zhou W, Yu PS, Zhu C (2022) Multi-agent reinforcement learning via knowledge transfer with differentially private noise. Int J Intell Syst 37(1):799\u2013828","journal-title":"Int J Intell Syst"},{"key":"10348_CR20","unstructured":"Chowdhury SR, Zhou X (2021) Differentially private regret minimization in episodic markov decision processes. http:\/\/arxiv.org\/abs\/2112.10599"},{"key":"10348_CR21","doi-asserted-by":"crossref","unstructured":"Dai C, Xiao L, Wan X, Chen Y (2019) Reinforcement learning with safe exploration for network security. In: ICASSP 2019-2019 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 3057\u20133061","DOI":"10.1109\/ICASSP.2019.8682983"},{"issue":"3","key":"10348_CR22","doi-asserted-by":"crossref","first-page":"653","DOI":"10.1109\/TNNLS.2016.2522401","volume":"28","author":"Y Deng","year":"2016","unstructured":"Deng Y, Bao F, Kong Y, Ren Z, Dai Q (2016) Deep direct reinforcement learning for financial signal representation and trading. IEEE Trans Neural Netw Learn Syst 28(3):653\u2013664","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"10348_CR23","unstructured":"Fran\u00e7ois-Lavet V (2017) Contributions to deep reinforcement learning and its applications in smartgrids. PhD thesis, Universite de Liege, Liege, Belgique"},{"key":"10348_CR24","unstructured":"Fu J, Luo K, Levine S (2017) Learning robust rewards with adversarial inverse reinforcement learning. http:\/\/arxiv.org\/abs\/1710.11248"},{"key":"10348_CR25","doi-asserted-by":"crossref","unstructured":"Gandhi D, Pinto L, Gupta A (2017) Learning to fly by crashing. In: 2017 IEEE\/RSJ international conference on intelligent robots and systems (IROS). IEEE, pp 3948\u20133955","DOI":"10.1109\/IROS.2017.8206247"},{"key":"10348_CR26","doi-asserted-by":"crossref","unstructured":"Gao H, Huang W, Liu T, Yin Y, Li Y (2022) Ppo2: Location privacy-oriented task offloading to edge computing using reinforcement learning for intelligent autonomous transport systems. In: IEEE transactions on intelligent transportation systems","DOI":"10.1109\/TITS.2022.3169421"},{"key":"10348_CR27","doi-asserted-by":"crossref","unstructured":"Garrett IY, Gerdes RM (2019) Z table: Cost-optimized attack on reinforcement learning. In: 2019 First IEEE international conference on trust, privacy and security in intelligent systems and applications (TPS-ISA). IEEE, pp 10\u201317","DOI":"10.1109\/TPS-ISA48467.2019.00011"},{"issue":"2","key":"10348_CR28","doi-asserted-by":"crossref","first-page":"178","DOI":"10.1287\/ijoc.1080.0305","volume":"21","author":"A Gosavi","year":"2009","unstructured":"Gosavi A (2009) Reinforcement learning: a tutorial survey and recent advances. INFORMS J Comput 21(2):178\u2013192","journal-title":"INFORMS J Comput"},{"key":"10348_CR29","doi-asserted-by":"crossref","unstructured":"Huang Y, Zhu Q (2019) Deceptive reinforcement learning under adversarial manipulations on cost signals. In: International conference on decision and game theory for security. Springer, pp 217\u2013237","DOI":"10.1007\/978-3-030-32430-8_14"},{"key":"10348_CR30","unstructured":"Kaiser L, Babaeizadeh M, Milos P, Osinski B, Campbell RH, Czechowski K, Erhan D, Finn C, Kozakowski P, Levine S et al (2019) Model-based reinforcement learning for atari. http:\/\/arxiv.org\/abs\/1903.00374"},{"issue":"11","key":"10348_CR31","doi-asserted-by":"crossref","first-page":"1238","DOI":"10.1177\/0278364913495721","volume":"32","author":"J Kober","year":"2013","unstructured":"Kober J, Bagnell JA, Peters J (2013) Reinforcement learning in robotics: a survey. Int J Robot Res 32(11):1238\u20131274","journal-title":"Int J Robot Res"},{"key":"10348_CR32","doi-asserted-by":"crossref","unstructured":"Lee XY, Ghadai S, Tan KL, Hegde C, Sarkar S (2020) Spatiotemporally constrained action space attacks on deep reinforcement learning agents. In: Proceedings of the AAAI conference on artificial intelligence, vol 34, pp 4577\u20134584","DOI":"10.1609\/aaai.v34i04.5887"},{"issue":"3","key":"10348_CR33","doi-asserted-by":"crossref","first-page":"1722","DOI":"10.1109\/COMST.2020.2988367","volume":"22","author":"L Lei","year":"2020","unstructured":"Lei L, Tan Y, Zheng K, Liu S, Zhang K, Shen X (2020) Deep reinforcement learning for autonomous internet of things: model, applications and challenges. IEEE Commun Surv Tutor 22(3):1722\u20131760","journal-title":"IEEE Commun Surv Tutor"},{"issue":"1","key":"10348_CR34","first-page":"1334","volume":"17","author":"S Levine","year":"2016","unstructured":"Levine S, Finn C, Darrell T, Abbeel P (2016) End-to-end training of deep visuomotor policies. J Mach Learn Res 17(1):1334\u20131373","journal-title":"J Mach Learn Res"},{"key":"10348_CR35","doi-asserted-by":"crossref","unstructured":"Li L, Chu W, Langford J, Schapire RE (2010) A contextual-bandit approach to personalized news article recommendation. In: Proceedings of the 19th international conference on world wide web, pp 661\u2013670","DOI":"10.1145\/1772690.1772758"},{"key":"10348_CR36","doi-asserted-by":"crossref","unstructured":"Li Z, Kiseleva J, de Rijke M (2019a) Dialogue generation: From imitation learning to inverse reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 33, pp 6722\u20136729","DOI":"10.1609\/aaai.v33i01.33016722"},{"key":"10348_CR37","doi-asserted-by":"crossref","unstructured":"Li S, Wu Y, Cui X, Dong H, Fang F, Russell S (2019b) Robust multi-agent reinforcement learning via minimax deep deterministic policy gradient. In: Proceedings of the AAAI conference on artificial intelligence, vol 33, pp 4213\u20134220","DOI":"10.1609\/aaai.v33i01.33014213"},{"issue":"3","key":"10348_CR38","doi-asserted-by":"crossref","first-page":"1163","DOI":"10.1109\/TCYB.2020.2982168","volume":"51","author":"H Li","year":"2020","unstructured":"Li H, Wu Y, Chen M (2020) Adaptive fault-tolerant tracking control for discrete-time multiagent systems via reinforcement learning algorithm. IEEE Trans Cybern 51(3):1163\u20131174","journal-title":"IEEE Trans Cybern"},{"key":"10348_CR39","unstructured":"Li J, Ren T, Yan D, Su H, Zhu J (2022) Policy learning for robust markov decision process with a mismatched generative mode. http:\/\/arxiv.org\/abs\/2203.06587"},{"key":"10348_CR40","doi-asserted-by":"crossref","unstructured":"Lin JC-W, Fournier-Viger P, Wu L, Gan W, Djenouri Y, Zhang J (2018) Ppsf: an open-source privacy-preserving and security mining framework. In: 2018 IEEE international conference on data mining workshops (ICDMW). IEEE, pp 1459\u20131463","DOI":"10.1109\/ICDMW.2018.00208"},{"key":"10348_CR41","doi-asserted-by":"crossref","unstructured":"Lin J, Dzeparoska K, Zhang SQ, Leon-Garcia A, Papernot N (2020) On the robustness of cooperative multi-agent reinforcement learning. In: 2020 IEEE security and privacy workshops (SPW). IEEE, pp 62\u201368","DOI":"10.1109\/SPW50608.2020.00027"},{"key":"10348_CR42","unstructured":"Littman ML, Dean TL, Kaelbling LP (2013) On the complexity of solving markov decision problems. http:\/\/arxiv.org\/abs\/1302.4971"},{"issue":"1","key":"10348_CR43","doi-asserted-by":"crossref","first-page":"299","DOI":"10.1109\/TASE.2016.2517155","volume":"14","author":"L Liu","year":"2016","unstructured":"Liu L, Wang Z, Zhang H (2016) Adaptive fault-tolerant tracking control for mimo discrete-time systems via reinforcement learning algorithm with less learning parameters. IEEE Trans Autom Sci Eng 14(1):299\u2013313","journal-title":"IEEE Trans Autom Sci Eng"},{"key":"10348_CR44","unstructured":"Liu Z, Yang Y, Miller T, Masters P (2021) Deceptive reinforcement learning for privacy-preserving planning. http:\/\/arxiv.org\/abs\/2102.03022"},{"issue":"3","key":"10348_CR45","doi-asserted-by":"crossref","first-page":"749","DOI":"10.1109\/JSAC.2022.3142348","volume":"40","author":"S Liu","year":"2022","unstructured":"Liu S, Zheng C, Huang Y, Quek TQ (2022) Distributed reinforcement learning for privacy-preserving dynamic edge caching. IEEE J Sel Areas Commun 40(3):749\u2013760","journal-title":"IEEE J Sel Areas Commun"},{"issue":"4","key":"10348_CR46","doi-asserted-by":"crossref","first-page":"3133","DOI":"10.1109\/COMST.2019.2916583","volume":"21","author":"NC Luong","year":"2019","unstructured":"Luong NC, Hoang DT, Gong S, Niyato D, Wang P, Liang Y-C, Kim DI (2019) Applications of deep reinforcement learning in communications and networking: a survey. IEEE Commun Surv Tutor 21(4):3133\u20133174","journal-title":"IEEE Commun Surv Tutor"},{"issue":"3","key":"10348_CR47","doi-asserted-by":"crossref","first-page":"110","DOI":"10.3390\/data4030110","volume":"4","author":"TL Meng","year":"2019","unstructured":"Meng TL, Khushi M (2019) Reinforcement learning in financial markets. Data 4(3):110","journal-title":"Data"},{"issue":"7540","key":"10348_CR48","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"issue":"2","key":"10348_CR49","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/s10994-009-5110-1","volume":"77","author":"G Neu","year":"2009","unstructured":"Neu G, Szepesv\u00e1ri C (2009) Training parsers by inverse reinforcement learning. Mach Learn 77(2):303\u2013337","journal-title":"Mach Learn"},{"key":"10348_CR50","doi-asserted-by":"crossref","unstructured":"Pan X, You Y, Wang Z, Lu C (2017) Virtual to real reinforcement learning for autonomous driving. http:\/\/arxiv.org\/abs\/1704.03952","DOI":"10.5244\/C.31.11"},{"key":"10348_CR51","unstructured":"Pan X, Wang W, Zhang X, Li B, Yi J, Song D (2019) How you act tells a lot: privacy-leaking attack on deep reinforcement learning. In: Proceedings of the 18th international conference on autonomous agents and multiagent systems, pp 368\u2013376"},{"key":"10348_CR52","doi-asserted-by":"crossref","first-page":"203564","DOI":"10.1109\/ACCESS.2020.3036899","volume":"8","author":"J Park","year":"2020","unstructured":"Park J, Kim DS, Lim H (2020) Privacy-preserving reinforcement learning using homomorphic encryption in cloud computing infrastructures. IEEE Access 8:203564\u2013203579","journal-title":"IEEE Access"},{"key":"10348_CR53","unstructured":"Prakash K, Husain F, Paruchuri P, Gujar SP (2021) How private is your RL policy? An inverse RL based analysis framework. http:\/\/arxiv.org\/abs\/2112.05495"},{"key":"10348_CR54","unstructured":"Rakhsha A, Radanovic G, Devidze R, Zhu X, Singla A (2020) Policy teaching via environment poisoning: training-time adversarial attacks against reinforcement learning. In: International conference on machine learning. PMLR, pp 7974\u20137984"},{"issue":"210","key":"10348_CR55","first-page":"1","volume":"22","author":"A Rakhsha","year":"2021","unstructured":"Rakhsha A, Radanovic G, Devidze R, Zhu X, Singla A (2021) Policy teaching in reinforcement learning via environment poisoning attacks. J Mach Learn Res 22(210):1\u201345","journal-title":"J Mach Learn Res"},{"key":"10348_CR56","doi-asserted-by":"crossref","first-page":"56","DOI":"10.1016\/j.future.2021.09.003","volume":"127","author":"Y Ren","year":"2022","unstructured":"Ren Y, Liu W, Liu A, Wang T, Li A (2022) A privacy-protected intelligent crowdsourcing application of iot based on the reinforcement learning. Future Gener Comput Syst 127:56\u201369","journal-title":"Future Gener Comput Syst"},{"key":"10348_CR57","doi-asserted-by":"crossref","unstructured":"Rodr\u00edguez-Barroso N, L\u00f3pez DJ, Luz\u00f3n M, Herrera F, Mart\u00ednez-C\u00e1mara E (2022) Survey on federated learning threats: concepts, taxonomy on attacks and defences, experimental study and challenges. http:\/\/arxiv.org\/abs\/2201.08135","DOI":"10.1016\/j.inffus.2022.09.011"},{"key":"10348_CR58","doi-asserted-by":"crossref","unstructured":"Sakuma J, Kobayashi S, Wright RN (2008) Privacy-preserving reinforcement learning. In: Proceedings of the 25th international conference on machine learning, pp 864\u2013871","DOI":"10.1145\/1390156.1390265"},{"key":"10348_CR59","doi-asserted-by":"crossref","unstructured":"Sehgal A, La H, Louis S, Nguyen H (2019) Deep reinforcement learning using genetic algorithm for parameter optimization. In: 2019 third IEEE international conference on robotic computing (IRC). IEEE, pp 596\u2013601","DOI":"10.1109\/IRC.2019.00121"},{"key":"10348_CR60","doi-asserted-by":"crossref","unstructured":"Sun J, Zhang T, Xie X, Ma L, Zheng Y, Chen K, Liu Y (2020) Stealthy and efficient adversarial attacks against deep reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 34, pp 5883\u20135891","DOI":"10.1609\/aaai.v34i04.6047"},{"key":"10348_CR61","first-page":"22447","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning: an introduction. MIT Press, Cambridge, p 22447"},{"key":"10348_CR62","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton RS, Barto AG (2018) Reinforcement learning: an introduction. MIT Press, Cambridge"},{"key":"10348_CR63","unstructured":"Tessler C, Efroni Y, Mannor S (2019) Action robust reinforcement learning and applications in continuous control. In: International conference on machine learning, PMLR. pp 6215\u20136224"},{"key":"10348_CR64","unstructured":"Tucker A, Gleave A, Russell S (2018) Inverse reinforcement learning for video games. http:\/\/arxiv.org\/abs\/1810.10593"},{"issue":"11","key":"10348_CR65","doi-asserted-by":"crossref","first-page":"8693","DOI":"10.1109\/JIOT.2020.3040957","volume":"8","author":"A Uprety","year":"2020","unstructured":"Uprety A, Rawat DB (2020) Reinforcement learning for IoT security: a comprehensive survey. IEEE Internet Things J 8(11):8693\u20138706","journal-title":"IEEE Internet Things J"},{"key":"10348_CR66","unstructured":"Vietri G, Balle B, Krishnamurthy A, Wu S (2020) Private reinforcement learning with pac and regret guarantees. In: International conference on machine learning. PMLR, pp 9754\u20139764"},{"key":"10348_CR67","unstructured":"Wang B, Hegde N (2019) Privacy-preserving q-learning with functional noise in continuous state spaces. http:\/\/arxiv.org\/abs\/1901.10634"},{"key":"10348_CR68","doi-asserted-by":"crossref","unstructured":"Wang X, Nair S, Althoff M (2020) Falsification-based robust adversarial reinforcement learning. In: 2020 19th IEEE international conference on machine learning and applications (ICMLA). IEEE, pp 205\u2013212","DOI":"10.1109\/ICMLA51294.2020.00042"},{"key":"10348_CR69","unstructured":"Watkins CJCH (1989) Learning from delayed rewards"},{"issue":"3\u20134","key":"10348_CR70","first-page":"279","volume":"8","author":"CJ Watkins","year":"1992","unstructured":"Watkins CJ, Dayan P (1992) Q-learning. Mach Learn 8(3\u20134):279\u2013292","journal-title":"Mach Learn"},{"key":"10348_CR71","doi-asserted-by":"crossref","first-page":"108004","DOI":"10.1016\/j.comnet.2021.108004","volume":"191","author":"Y Wu","year":"2021","unstructured":"Wu Y, Wang Z, Ma Y, Leung VC (2021) Deep reinforcement learning for blockchain in industrial iot: a survey. Comput Netw 191:108004","journal-title":"Comput Netw"},{"issue":"2","key":"10348_CR72","doi-asserted-by":"crossref","first-page":"843","DOI":"10.1109\/SURV.2012.060912.00182","volume":"15","author":"Z Xiao","year":"2012","unstructured":"Xiao Z, Xiao Y (2012) Security and privacy in cloud computing. IEEE Commun Surv Tutor 15(2):843\u2013859","journal-title":"IEEE Commun Surv Tutor"},{"issue":"10","key":"10348_CR73","doi-asserted-by":"crossref","first-page":"4214","DOI":"10.1109\/TCYB.2019.2906574","volume":"50","author":"D Ye","year":"2019","unstructured":"Ye D, Zhu T, Zhou W, Philip SY (2019) Differentially private malicious agent avoidance in multiagent advising learning. IEEE Trans Cybern 50(10):4214\u20134227","journal-title":"IEEE Trans Cybern"},{"key":"10348_CR74","doi-asserted-by":"crossref","first-page":"569","DOI":"10.1109\/TIFS.2020.3016842","volume":"16","author":"D Ye","year":"2020","unstructured":"Ye D, Zhu T, Shen S, Zhou W (2020a) A differentially private game theoretic approach for deceiving cyber adversaries. IEEE Trans Inf Forensic Secur 16:569\u2013584","journal-title":"IEEE Trans Inf Forensic Secur"},{"key":"10348_CR75","unstructured":"Ye D, Shen S, Zhu T, Liu B, Zhou W (2020b) One parameter defense-defending against data inference attacks via differential privacy. IEEE Trans Inf Forensics Secur"},{"key":"10348_CR76","unstructured":"Ye D, Zhu T, Cheng Z, Zhou W, Philip SY (2020c) Differential advising in multiagent reinforcement learning. In: IEEE transactions on cybernetics"},{"key":"10348_CR77","doi-asserted-by":"crossref","unstructured":"Ye D, Zhu T, Shen S, Zhou W, Yu P (2020d) Differentially private multi-agent planning for logistic-like problems. In: IEEE transactions on dependable and secure computing","DOI":"10.1109\/TDSC.2020.3017497"},{"key":"10348_CR78","doi-asserted-by":"crossref","unstructured":"Ye D, Zhu T, Zhu C, Zhou W, Philip SY (2022) Model-based self-advising for multi-agent learning. In: IEEE transactions on neural networks and learning systems","DOI":"10.1109\/TNNLS.2022.3147221"},{"key":"10348_CR79","doi-asserted-by":"crossref","unstructured":"Ying Z, Zhang Y, Cao S, Xu S, Liu X (2020) Oidpr: optimized insulin dosage based on privacy-preserving reinforcement learning. In: 2020 IFIP Networking Conference (Networking). IEEE, pp 655\u2013657","DOI":"10.1002\/ett.3953"},{"issue":"4","key":"10348_CR80","doi-asserted-by":"crossref","first-page":"2238","DOI":"10.1109\/JIOT.2020.3026589","volume":"8","author":"S Yu","year":"2020","unstructured":"Yu S, Chen X, Zhou Z, Gong X, Wu D (2020) When deep reinforcement learning meets federated learning: intelligent multitimescale resource management for multiaccess edge computing in 5g ultradense network. IEEE Internet Things J 8(4):2238\u20132251","journal-title":"IEEE Internet Things J"},{"issue":"1","key":"10348_CR81","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3477600","volume":"55","author":"C Yu","year":"2021","unstructured":"Yu C, Liu J, Nemati S, Yin G (2021) Reinforcement learning in healthcare: a survey. ACM Comput Surv (CSUR) 55(1):1\u201336","journal-title":"ACM Comput Surv (CSUR)"},{"key":"10348_CR82","doi-asserted-by":"crossref","unstructured":"Zhai P, Luo J, Dong Z, Zhang L, Wang S, Yang D (2022) Robust adversarial reinforcement learning with dissipation inequation constraint","DOI":"10.1609\/aaai.v36i5.20481"},{"key":"10348_CR83","unstructured":"Zhang X, Ma Y, Singla A, Zhu X (2020) Adaptive reward-poisoning attacks against reinforcement learning. In: International conference on machine learning. PMLR, pp 11225\u201311234"},{"key":"10348_CR84","doi-asserted-by":"crossref","unstructured":"Zhao Y, Shumailov I, Cui H, Gao X, Mullins R, Anderson R (2020) Blackbox attacks on reinforcement learning agents using approximated temporal information. In: 2020 50th Annual IEEE\/IFIP international conference on dependable systems and networks workshops (DSN-W). IEEE, pp 16\u201324","DOI":"10.1109\/DSN-W50199.2020.00013"},{"key":"10348_CR85","doi-asserted-by":"crossref","unstructured":"Zhou X (2022) Differentially private reinforcement learning with linear function approximation. http:\/\/arxiv.org\/abs\/2201.07052","DOI":"10.1145\/3489048.3522648"},{"issue":"8","key":"10348_CR86","doi-asserted-by":"crossref","first-page":"1619","DOI":"10.1109\/TKDE.2017.2697856","volume":"29","author":"T Zhu","year":"2017","unstructured":"Zhu T, Li G, Zhou W, Philip SY (2017) Differentially private data publishing and analysis: a survey. IEEE Trans Knowl Data Eng 29(8):1619\u20131638","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"10348_CR87","doi-asserted-by":"crossref","unstructured":"Zhu T, Ye D, Wang W, Zhou W, Yu PS (2020) More than privacy: applying differential privacy in key areas of artificial intelligence. http:\/\/arxiv.org\/abs\/2008.01916","DOI":"10.1109\/TKDE.2020.3014246"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-022-10348-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-022-10348-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-022-10348-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,6]],"date-time":"2023-07-06T09:56:37Z","timestamp":1688637397000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-022-10348-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,6]]},"references-count":87,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2023,7]]}},"alternative-id":["10348"],"URL":"https:\/\/doi.org\/10.1007\/s10462-022-10348-5","relation":{},"ISSN":["0269-2821","1573-7462"],"issn-type":[{"value":"0269-2821","type":"print"},{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,6]]},"assertion":[{"value":"6 December 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}