{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,30]],"date-time":"2025-05-30T05:45:49Z","timestamp":1748583949990,"version":"3.37.3"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2022,10,3]],"date-time":"2022-10-03T00:00:00Z","timestamp":1664755200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,10,3]],"date-time":"2022-10-03T00:00:00Z","timestamp":1664755200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072406"],"award-info":[{"award-number":["62072406"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Laboratory of Science and Tecknology on Information System Security","award":["61421110502"],"award-info":[{"award-number":["61421110502"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2001"],"award-info":[{"award-number":["U21B2001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key R&D Programs of Zhejiang Province","award":["2022C01018"],"award-info":[{"award-number":["2022C01018"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62103374"],"award-info":[{"award-number":["62103374"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004731","name":"Natural Science Foundation of Zhejiang Province","doi-asserted-by":"publisher","award":["LGF20F020016"],"award-info":[{"award-number":["LGF20F020016"]}],"id":[{"id":"10.13039\/501100004731","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Lab of Ministry of Public Security","award":["2020DSJSYS003"],"award-info":[{"award-number":["2020DSJSYS003"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2023,5]]},"DOI":"10.1007\/s10489-022-03882-w","type":"journal-article","created":{"date-parts":[[2022,10,3]],"date-time":"2022-10-03T03:24:50Z","timestamp":1664767490000},"page":"12831-12858","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Agent manipulator: Stealthy strategy attacks on deep reinforcement learning"],"prefix":"10.1007","volume":"53","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7153-2755","authenticated-orcid":false,"given":"Jinyin","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueke","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haibin","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shanqing","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"Bao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,3]]},"reference":[{"key":"3882_CR1","first-page":"621","volume":"33","author":"D Ye","year":"2020","unstructured":"Ye D, Chen G, Zhang W, Chen S, Yuan B, Liu B, Chen J, Liu Z, Qiu F, Yu H et al (2020) Towards playing full moba games with deep reinforcement learning. Adv Neural Inf Process Syst 33:621\u2013632","journal-title":"Adv Neural Inf Process Syst"},{"issue":"9","key":"3882_CR2","doi-asserted-by":"publisher","first-page":"3706","DOI":"10.1002\/rnc.4962","volume":"30","author":"Y Yang","year":"2020","unstructured":"Yang Y, Vamvoudakis KG, Modares H (2020) Safe reinforcement learning for dynamical games. Int J Robust Nonlinear Control 30(9):3706\u20133726","journal-title":"Int J Robust Nonlinear Control"},{"key":"3882_CR3","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1016\/j.ins.2018.06.022","volume":"463","author":"X Yang","year":"2018","unstructured":"Yang X, He H, Wei Q, Luo B (2018) Reinforcement learning for robust adaptive control of partially unknown nonlinear systems subject to unmatched uncertainties. Inf Sci 463:307\u2013322","journal-title":"Inf Sci"},{"key":"3882_CR4","doi-asserted-by":"crossref","unstructured":"Fayjie AR, Hossain S, Oualid D, Lee DJ (2018) Driverless car: Autonomous driving using deep reinforcement learning in urban environment. In: 2018 15th international conference on ubiquitous robots UR, IEEE, pp 896\u2013901","DOI":"10.1109\/URAI.2018.8441797"},{"key":"3882_CR5","unstructured":"Prasad N, Cheng LF, Chivers C, Draugelis M, Engelhardt BE (2017) A reinforcement learning approach to weaning of mechanical ventilation in intensive care units. In: 33Rd conference on uncertainty in artificial intelligence"},{"issue":"8","key":"3882_CR6","doi-asserted-by":"publisher","first-page":"6202","DOI":"10.1007\/s10489-021-02218-4","volume":"51","author":"J Lee","year":"2021","unstructured":"Lee J, Koh H, Choe HJ (2021) Learning to trade in financial time series using high-frequency through wavelet transformation and deep reinforcement learning. Appl Intell 51(8):6202\u2013 6223","journal-title":"Appl Intell"},{"issue":"1","key":"3882_CR7","doi-asserted-by":"publisher","first-page":"231","DOI":"10.1007\/s13042-020-01167-7","volume":"12","author":"A Perrusqu\u00eda","year":"2021","unstructured":"Perrusqu\u00eda A, Yu W, Li X (2021) Multi-agent reinforcement learning for redundant robot control in task-space. Int J Mach Learn Cybern 12(1):231\u2013241","journal-title":"Int J Mach Learn Cybern"},{"key":"3882_CR8","doi-asserted-by":"crossref","unstructured":"Nguyen TT, Reddi V (2022) Deep reinforcement learning for cyber security. IEEE Transactions on Neural Networks and Learning Systems, p 1\u201318","DOI":"10.1109\/TNNLS.2021.3121870"},{"key":"3882_CR9","doi-asserted-by":"publisher","first-page":"467","DOI":"10.1016\/j.ins.2020.06.010","volume":"537","author":"PA Andersen","year":"2020","unstructured":"Andersen PA, Goodwin M, Granmo OC (2020) Towards safe reinforcement-learning in industrial grid-warehousing. Inf Sci 537:467\u2013484","journal-title":"Inf Sci"},{"key":"3882_CR10","doi-asserted-by":"crossref","unstructured":"Le N, Rathour VS, Yamazaki K, Luu K, Savvides M (2021) Deep reinforcement learning in computer vision: a comprehensive survey. Artificial Intelligence Review, p 1\u201387","DOI":"10.1007\/s10462-021-10061-9"},{"issue":"7","key":"3882_CR11","doi-asserted-by":"publisher","first-page":"1704","DOI":"10.1109\/TMM.2019.2960636","volume":"22","author":"R Furuta","year":"2019","unstructured":"Furuta R, Inoue N, Yamasaki T (2019) Pixelrl: Fully convolutional network with reinforcement learning for image processing. IEEE Trans Multimed 22(7):1704\u20131719","journal-title":"IEEE Trans Multimed"},{"issue":"2","key":"3882_CR12","doi-asserted-by":"publisher","first-page":"112","DOI":"10.1109\/MNET.011.2000303","volume":"35","author":"Q Liu","year":"2021","unstructured":"Liu Q, Cheng L, Jia AL, Liu C (2021) Deep reinforcement learning for communication flow control in wireless mesh networks. IEEE Netw 35(2):112\u2013119","journal-title":"IEEE Netw"},{"key":"3882_CR13","doi-asserted-by":"publisher","first-page":"62","DOI":"10.1016\/j.ins.2021.01.077","volume":"565","author":"P Chen","year":"2021","unstructured":"Chen P, Lu W (2021) Deep reinforcement learning based moving object grasping. Inf Sci 565:62\u201376","journal-title":"Inf Sci"},{"issue":"9","key":"3882_CR14","doi-asserted-by":"publisher","first-page":"1363","DOI":"10.3390\/electronics9091363","volume":"9","author":"N Vithayathil Varghese","year":"2020","unstructured":"Vithayathil Varghese N, Mahmoud QH (2020) A survey of multi-task deep reinforcement learning. Electronics 9(9):1363","journal-title":"Electronics"},{"key":"3882_CR15","doi-asserted-by":"publisher","first-page":"815","DOI":"10.1016\/j.ins.2020.08.101","volume":"546","author":"F Zou","year":"2021","unstructured":"Zou F, Yen GG, Tang L, Wang C (2021) A reinforcement learning approach for dynamic multi-objective optimization. Inf Sci 546:815\u2013834","journal-title":"Inf Sci"},{"key":"3882_CR16","doi-asserted-by":"publisher","first-page":"205","DOI":"10.1016\/j.ins.2020.05.022","volume":"536","author":"N Pr\u00f6llochs","year":"2020","unstructured":"Pr\u00f6llochs N, Feuerriegel S, Lutz B, Neumann D (2020) Negation scope detection for sentiment analysis: a reinforcement learning framework for replicating human interpretations. Inf Sci 536:205\u2013221","journal-title":"Inf Sci"},{"issue":"3","key":"3882_CR17","doi-asserted-by":"publisher","first-page":"1722","DOI":"10.1109\/COMST.2020.2988367","volume":"22","author":"L Lei","year":"2020","unstructured":"Lei L, Tan Y, Zheng K, Liu S, Zhang K, Shen X (2020) Deep reinforcement learning for autonomous internet of things: model, applications and challenges. IEEE Commun Surv Tutor 22(3):1722\u20131760","journal-title":"IEEE Commun Surv Tutor"},{"key":"3882_CR18","unstructured":"Igl M, Ciosek K, Li Y, Tschiatschek S, Zhang C, Devlin S, Hofmann K (2019) Generalization in reinforcement learning with selective noise injection and information bottleneck. Proceedings of the 33rd International Conference on Neural Information Processing Systems, p 13979\u201313991"},{"issue":"04","key":"3882_CR19","first-page":"6202","volume":"34","author":"J Wang","year":"2020","unstructured":"Wang J, Liu Y, Li B (2020) Reinforcement learning with perturbed rewards. Proc Conf AAAI Artif Intell 34(04):6202\u20136209","journal-title":"Proc Conf AAAI Artif Intell"},{"key":"3882_CR20","unstructured":"Pinto L, Davidson J, Sukthankar R, Gupta A (2017) Robust adversarial reinforcement learning. International Conference on Machine Learning, p 2817\u20132826"},{"key":"3882_CR21","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1016\/j.geb.2016.06.004","volume":"103","author":"M Bravo","year":"2017","unstructured":"Bravo M, Mertikopoulos P (2017) On the robustness of learning in games with stochastically perturbed payoff observations. Games Econ Behav 103:41\u201366","journal-title":"Games Econ Behav"},{"key":"3882_CR22","doi-asserted-by":"crossref","unstructured":"Behzadan V, Munir A (2018) Mitigation of policy manipulation attacks on deep q-networks with parameter-space noise, International Conference on Computer Safety Reliability, and Security, p 406\u2013417","DOI":"10.1007\/978-3-319-99229-7_34"},{"key":"3882_CR23","doi-asserted-by":"publisher","first-page":"107295","DOI":"10.1016\/j.asoc.2021.107295","volume":"105","author":"RRO Al-Nima","year":"2021","unstructured":"Al-Nima RRO, Han T, Al-Sumaidaee SAM, Chen T, Woo WL (2021) Robustness and performance of deep reinforcement learning. Appl Soft Comput 105:107295","journal-title":"Appl Soft Comput"},{"key":"3882_CR24","doi-asserted-by":"crossref","unstructured":"Han Y, Rubinstein BI, Abraham T, Alpcan T, De Vel O, Erfani S, Hubczenko D, Leckie C, Montague P (2018) Reinforcement learning for autonomous defence in software-defined networking. In: International conference on decision and game theory for security, Springer, pp 145\u2013165","DOI":"10.1007\/978-3-030-01554-1_9"},{"key":"3882_CR25","doi-asserted-by":"crossref","unstructured":"Bai X, Niu W, Liu J, Gao X, Xiang Y, Liu J (2018) Adversarial examples construction towards white-box q table variation in dqn pathfinding training. In: 2018 IEEE third international conference on data science in cyberspace (DSC), IEEE, pp 781\u2013787","DOI":"10.1109\/DSC.2018.00126"},{"key":"3882_CR26","doi-asserted-by":"crossref","unstructured":"Lee XY, Ghadai S, Tan KL, Hegde C, Sarkar S (2020) Spatiotemporally constrained action space attacks on deep reinforcement learning agents. In: AAAI, pp 4577\u20134584","DOI":"10.1609\/aaai.v34i04.5887"},{"key":"3882_CR27","unstructured":"Panagiota K, Kacper W, Jha S, Wenchao L (2020) Trojdrl: Trojan attacks on deep reinforcement learning agents. In: Proc. 57th ACM\/IEEE design automation conference (DAC)"},{"key":"3882_CR28","doi-asserted-by":"crossref","unstructured":"Behzadan V, Munir A (2017) Vulnerability of deep reinforcement learning to policy induction attacks. In: International conference on machine learning and data mining in pattern recognition, pp 262\u2013275","DOI":"10.1007\/978-3-319-62416-7_19"},{"key":"3882_CR29","doi-asserted-by":"crossref","unstructured":"Wang B, Yao Y, Shan S, Li H, Viswanath B, Zheng H, Zhao BY (2019) Neural cleanse: identifying and mitigating backdoor attacks in neural networks. In: 2019 IEEE symposium on security and privacy (SP), IEEE, pp 707\u2013723","DOI":"10.1109\/SP.2019.00031"},{"key":"3882_CR30","doi-asserted-by":"crossref","unstructured":"Wang L, Javed Z, Wu X, Guo W, Xing X, Song D (2021) Backdoorl: Backdoor attack against competitive reinforcement learning. In: IJCAI","DOI":"10.24963\/ijcai.2021\/509"},{"key":"3882_CR31","unstructured":"Behzadan V, Hsu W (2019) Adversarial exploitation of policy imitation. In: IJCAI"},{"key":"3882_CR32","unstructured":"Kos J, Song D (2017) Delving into adversarial attacks on deep policies. In: 5Th international conference on learning representations, ICLR"},{"key":"3882_CR33","unstructured":"Tretschk E, Oh SJ, Fritz M (2018) Sequential attacks on agents for long-term adversarial goals. In: 2. ACM Computer science in cars symposium"},{"key":"3882_CR34","unstructured":"Hussenot L, Geist M, Pietquin O (2019) Targeted attacks on deep reinforcement learning agents through adversarial observations, 1\u20139 arXiv:1905.12282"},{"key":"3882_CR35","unstructured":"Huang S, Papernot N, Goodfellow I, Duan Y, Abbeel P (2017) Adversarial attacks on neural network policies. In: 5Th international conference on learning representations, ICLR"},{"key":"3882_CR36","unstructured":"Goodfellow IJ, Shlens J, Szegedy C (2015) Explaining and harnessing adversarial examples. In: The international conference on learning representations, ICLR"},{"key":"3882_CR37","unstructured":"Behzadan V, Munir A (2017) Whatever does not kill deep reinforcement learning, makes it stronger, 1\u20138 arXiv:1712.09344"},{"key":"3882_CR38","unstructured":"Pattanaik A, Tang Z, Liu S, Bommannan G, Chowdhary G (2018) Robust deep reinforcement learning with adversarial attacks. In: 17th International conference on autonomous agents and multiagent systems, AAMAS 2018, pp 2040\u2013 2042"},{"key":"3882_CR39","unstructured":"Gleave A, Dennis M, Wild C, Kant N, Levine S, Russell S (2019) Adversarial policies: Attacking deep reinforcement learning. In: International conference on learning representations"},{"key":"3882_CR40","unstructured":"Sun Y, Huo D, Huang F (2020) Vulnerability-aware poisoning mechanism for online rl with unknown dynamics. In: International conference on learning representations"},{"key":"3882_CR41","first-page":"11225","volume":"119","author":"X Zhang","year":"2020","unstructured":"Zhang X, Ma Y, Singla A, Zhu X (2020) Adaptive reward-poisoning attacks against reinforcement learning. Proceedings of the 37th International Conference on Machine Learning 119:11225\u201311234","journal-title":"Proceedings of the 37th International Conference on Machine Learning"},{"key":"3882_CR42","unstructured":"Behzadan V, Hsu W (2017) Analysis and improvement of adversarial training in dqn agents with adversarially-guided exploration (age), 1\u20139 arXiv:1906.01119"},{"key":"3882_CR43","unstructured":"Rajeswaran A, Ghotra S, Ravindran B, Levine S (2016) Epopt: Learning robust neural network policies using model ensembles. In: Proceedings of the 5th International Conference on Learning Representations, 1\u201315 arXiv:1610.01283"},{"issue":"2","key":"3882_CR44","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1162\/0899766053011528","volume":"17","author":"J Morimoto","year":"2005","unstructured":"Morimoto J, Doya K (2005) Robust reinforcement learning. Neural Comput 17(2):335\u2013359","journal-title":"Neural Comput"},{"key":"3882_CR45","doi-asserted-by":"crossref","unstructured":"Ogunmolu O, Gans N, Summers T (2018) Minimax iterative dynamic game: application to nonlinear robot control tasks. In: 2018 IEEE\/RSJ international conference on intelligent robots and systems (IROS), IEEE, pp 6919\u20136925","DOI":"10.1109\/IROS.2018.8594037"},{"key":"3882_CR46","unstructured":"Gu Z, Jia Z, Choset H (2019) Adversary a3c for robust reinforcement learning, 1\u201312 arXiv:1912.00330"},{"key":"3882_CR47","unstructured":"Behzadan V, Hsu W (2019) Sequential triggers for watermarking of deep reinforcement learning policies, 1\u20134 arXiv:1906.01126"},{"key":"3882_CR48","unstructured":"Lin YC, Liu MY, Sun M, Huang JB (2017) Detecting adversarial attacks on neural network policies with visual foresight, 1\u201310 arXiv:1710.00814"},{"key":"3882_CR49","doi-asserted-by":"crossref","unstructured":"Hessel M, Modayil J, Van Hasselt H, Schaul T, Ostrovski G, Dabney W, Horgan D, Piot B, Azar M, Silver D (2018) Rainbow: Combining improvements in deep reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, Vol. 32","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"3882_CR50","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1613\/jair.3912","volume":"47","author":"MG Bellemare","year":"2013","unstructured":"Bellemare MG, Naddaf Y, Veness J, Bowling M (2013) The arcade learning environment: an evaluation platform for general agents. J Artif Intell Res 47:253\u2013279","journal-title":"J Artif Intell Res"},{"issue":"2","key":"3882_CR51","doi-asserted-by":"publisher","first-page":"336","DOI":"10.1007\/s11263-019-01228-7","volume":"128","author":"RR Selvaraju","year":"2020","unstructured":"Selvaraju RR, Cogswell M, Das A, Vedantam R, Parikh D, Batra D (2020) Grad-cam: Visual explanations from deep networks via gradient-based localization. Int J Comput Vis 128(2):336\u2013 359","journal-title":"Int J Comput Vis"},{"key":"3882_CR52","doi-asserted-by":"crossref","unstructured":"Pei K, Cao Y, Yang J, Jana S (2017) Deepxplore: Automated whitebox testing of deep learning systems. In: proceedings of the 26th symposium on operating systems principles, pp 1\u201318","DOI":"10.1145\/3132747.3132785"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-03882-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-022-03882-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-03882-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,20]],"date-time":"2023-05-20T10:48:30Z","timestamp":1684579710000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-022-03882-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,3]]},"references-count":52,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2023,5]]}},"alternative-id":["3882"],"URL":"https:\/\/doi.org\/10.1007\/s10489-022-03882-w","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2022,10,3]]},"assertion":[{"value":"10 June 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 October 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}