{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T15:22:18Z","timestamp":1782314538643,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":71,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Ministry of Science and Technology of China","award":["2022YFB3102100"],"award-info":[{"award-number":["2022YFB3102100"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1145\/3658644.3670293","type":"proceedings-article","created":{"date-parts":[[2024,12,9]],"date-time":"2024-12-09T12:19:20Z","timestamp":1733746760000},"page":"645-659","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["<i>SUB-PLAY:<\/i>\n            Adversarial Policies against Partially Observed Multi-Agent Reinforcement Learning Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6572-972X","authenticated-orcid":false,"given":"Oubo","family":"Ma","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2311-4943","authenticated-orcid":false,"given":"Yuwen","family":"Pu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9028-9326","authenticated-orcid":false,"given":"Linkang","family":"Du","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4813-8274","authenticated-orcid":false,"given":"Yang","family":"Dai","sequence":"additional","affiliation":[{"name":"Laboratory for Big Data and Decision, Changsha, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9621-6425","authenticated-orcid":false,"given":"Ruo","family":"Wang","sequence":"additional","affiliation":[{"name":"Chinese Aeronautical Establishment, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8510-4025","authenticated-orcid":false,"given":"Xiaolei","family":"Liu","sequence":"additional","affiliation":[{"name":"Institute of Computer Application, China Academy of Engineering Physics, Mianyang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1119-3237","authenticated-orcid":false,"given":"Yingcai","family":"Wu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4268-372X","authenticated-orcid":false,"given":"Shouling","family":"Ji","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,12,9]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Trapit Bansal Jakub Pachocki Szymon Sidor Ilya Sutskever and Igor Mordatch. 2018. Emergent Complexity via Multi-Agent Competition. In ICLR."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-62416-7_19"},{"key":"e_1_3_2_1_3_1","unstructured":"Christopher Berner Greg Brockman Brooke Chan Vicki Cheung Przemys\u0142aw Dkebiak Christy Dennison David Farhi Quirin Fischer Shariq Hashme Chris Hesse et al. 2019. Dota 2 with Large Scale Deep Reinforcement Learning. arXiv (2019)."},{"key":"e_1_3_2_1_4_1","unstructured":"Noam Brown. 2020. Equilibrium Finding for Large Adversarial Imperfect-Information Games. PhD thesis (2020)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Nicholas Carlini and David Wagner. 2017. Towards Evaluating the Robustness of Neural Networks. In S&P.","DOI":"10.1109\/SP.2017.49"},{"key":"e_1_3_2_1_6_1","volume-title":"MARNet: Backdoor Attacks Against Cooperative Multi-Agent Reinforcement Learning","author":"Chen Yanjiao","year":"2022","unstructured":"Yanjiao Chen, Zhicong Zheng, and Xueluan Gong. 2022. MARNet: Backdoor Attacks Against Cooperative Multi-Agent Reinforcement Learning. IEEE Transactions on Dependable and Secure Computing (2022)."},{"key":"e_1_3_2_1_7_1","volume-title":"Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning? arXiv","author":"Dai Yang","year":"2024","unstructured":"Yang Dai, Oubo Ma, Longfei Zhang, Xingxing Liang, Shengchao Hu, Mengzhu Wang, Shouling Ji, Jincai Huang, and Li Shen. 2024. Is Mamba Compatible with Trajectory Optimization in Offline Reinforcement Learning? arXiv (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Malte F Jung, Spencer Kohn, Tyler H Shaw, Richard Pak, and Mark A Neerincx.","author":"De Visser Ewart J","year":"2020","unstructured":"Ewart J De Visser, Marieke MM Peeters, Malte F Jung, Spencer Kohn, Tyler H Shaw, Richard Pak, and Mark A Neerincx. 2020. Towards a Theory of Longitudinal Trust Calibration in Human-Robot Teams. International Journal of Social Robotics (2020)."},{"key":"e_1_3_2_1_9_1","unstructured":"DJI. [n. d.]. Robomaster. https:\/\/www.robomaster.com\/en-US."},{"key":"e_1_3_2_1_10_1","unstructured":"Linkang Du Chen Min Sun Mingyang Ji Shouling Cheng Peng Chen Jiming and Zhang Zhikun. 2024. ORL-AUDITOR: Dataset Auditing in Offline Deep Reinforcement Learning. In NDSS."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Rolando Fernandez Derrik E Asher Anjon Basak Piyush K Sharma Erin G Zaroukian Christopher D Hsu Michael R Dorothy Christopher M Kroninger Luke Frerichs John Rogers et al. 2021. Multi-Agent Coordination for Strategic Maneuver with a Survey of Reinforcement Learning. Technical Report. US Army Combat Capabilities Development Command Army Research Laboratory.","DOI":"10.21236\/AD1154872"},{"key":"e_1_3_2_1_12_1","unstructured":"Wei Fu Weihua Du Jingwei Li Sunli Chen Jingzhao Zhang and Yi Wu. 2023. Iteratively Learn Diverse Strategies with State Distance Information. In NeurIPS."},{"key":"e_1_3_2_1_13_1","volume-title":"Conference on Robot Learning.","author":"Fu Zipeng","year":"2023","unstructured":"Zipeng Fu, Xuxin Cheng, and Deepak Pathak. 2023. Deep Whole-Body Control: Learning a Unified Policy for Manipulation and Locomotion. In Conference on Robot Learning."},{"key":"e_1_3_2_1_14_1","unstructured":"Dibya Ghosh Jad Rahme Aviral Kumar Amy Zhang Ryan P Adams and Sergey Levine. 2021. Why Generalization in RL is Difficult: Epistemic Pomdps and Implicit Partial Observability. In NeurIPS."},{"key":"e_1_3_2_1_15_1","volume-title":"Adversarial Policies: Attacking Deep Reinforcement Learning. In ICLR.","author":"Gleave Adam","year":"2020","unstructured":"Adam Gleave, Michael Dennis, Cody Wild, Neel Kant, Sergey Levine, and Stuart Russell. 2020. Adversarial Policies: Attacking Deep Reinforcement Learning. In ICLR."},{"key":"e_1_3_2_1_16_1","volume-title":"ATTRITION: Attacking Static Hardware Trojan Detection Techniques Using Reinforcement Learning. In CCS.","author":"Gohil Vasudev","year":"2022","unstructured":"Vasudev Gohil, Hao Guo, Satwik Patnaik, and Jeyavijayan Rajendran. 2022. ATTRITION: Attacking Static Hardware Trojan Detection Techniques Using Reinforcement Learning. In CCS."},{"key":"e_1_3_2_1_17_1","volume-title":"Backdoor Attacks, and Defenses","author":"Goldblum Micah","year":"2022","unstructured":"Micah Goldblum, Dimitris Tsipras, Chulin Xie, Xinyun Chen, Avi Schwarzschild, Dawn Song, Aleksander Mkadry, Bo Li, and Tom Goldstein. 2022. Dataset Security for Machine Learning: Data Poisoning, Backdoor Attacks, and Defenses. IEEE Transactions on Pattern Analysis and Machine Intelligence (2022)."},{"key":"e_1_3_2_1_18_1","unstructured":"Ian J Goodfellow Jonathon Shlens and Christian Szegedy. 2015. Explaining and Harnessing Adversarial Examples. In ICLR."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-09996-w"},{"key":"e_1_3_2_1_20_1","unstructured":"Pengjie Gu Mengchen Zhao Jianye Hao and Bo An. 2022. Online Ad Hoc Teamwork under Partial Observability. In ICLR."},{"key":"e_1_3_2_1_21_1","unstructured":"Junfeng Guo Ang Li Lixu Wang and Cong Liu. 2023. PolicyCleanse: Backdoor Detection and Mitigation for Competitive Reinforcement Learning. In ICCV."},{"key":"e_1_3_2_1_22_1","unstructured":"Wenbo Guo Xian Wu Sui Huang and Xinyu Xing. 2021. Adversarial Policy Learning in Two-Player Competitive Games. In ICML."},{"key":"e_1_3_2_1_23_1","volume-title":"PATROL: Provable Defense against Adversarial Policy in Two-player Games. In USENIX Security.","author":"Guo Wenbo","year":"2023","unstructured":"Wenbo Guo, Xian Wu, Lun Wang, Xinyu Xing, and Dawn Song. 2023. PATROL: Provable Defense against Adversarial Policy in Two-player Games. In USENIX Security."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCST.2015.2411632"},{"key":"e_1_3_2_1_25_1","unstructured":"Ping He Yifan Xia Xuhong Zhang and Shouling Ji. 2023. Efficient Query-Based Attack against ML-Based Android Malware Detection under Zero Knowledge Setting. In CCS."},{"key":"e_1_3_2_1_26_1","unstructured":"Johannes Heinrich Marc Lanctot and David Silver. 2015. Fictitious Self-Play in Extensive-Form Games. In ICML."},{"key":"e_1_3_2_1_27_1","volume-title":"Drone Systems for Factory Security and Surveillance. Interdisciplinary Description of Complex Systems: INDECS","author":"Hell P\u00e9ter Miksa","year":"2019","unstructured":"P\u00e9ter Miksa Hell and P\u00e9ter J\u00e1nos Varga. 2019. Drone Systems for Factory Security and Surveillance. Interdisciplinary Description of Complex Systems: INDECS (2019)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Peter Henderson Riashat Islam Philip Bachman Joelle Pineau Doina Precup and David Meger. 2018. Deep Reinforcement Learning that Matters. In AAAI.","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"e_1_3_2_1_29_1","volume-title":"Adversarial Attacks on Neural Network Policies. arXiv","author":"Huang Sandy","year":"2017","unstructured":"Sandy Huang, Nicolas Papernot, Ian Goodfellow, Yan Duan, and Pieter Abbeel. 2017. Adversarial Attacks on Neural Network Policies. arXiv (2017)."},{"key":"e_1_3_2_1_30_1","volume-title":"Text Laundering: Mitigating Malicious Features Through Knowledge Distillation of Large Foundation Models. In Inscrypt.","author":"Jiang Yi","year":"2023","unstructured":"Yi Jiang, Chenghui Shi, Oubo Ma, Youliang Tian, and Shouling Ji. 2023. Text Laundering: Mitigating Malicious Features Through Knowledge Distillation of Large Foundation Models. In Inscrypt."},{"key":"e_1_3_2_1_31_1","volume-title":"Champion-Level Drone Racing Using Deep Reinforcement Learning. Nature","author":"Kaufmann Elia","year":"2023","unstructured":"Elia Kaufmann, Leonard Bauersfeld, Antonio Loquercio, Matthias M\u00fcller, Vladlen Koltun, and Davide Scaramuzza. 2023. Champion-Level Drone Racing Using Deep Reinforcement Learning. Nature (2023)."},{"key":"e_1_3_2_1_32_1","volume-title":"TrojDRL: Evaluation of Backdoor Attacks on Deep Reinforcement Learning. In 57th ACM\/IEEE Design Automation Conference (DAC).","author":"Kiourti Panagiota","year":"2020","unstructured":"Panagiota Kiourti, Kacper Wardega, Susmit Jha, and Wenchao Li. 2020. TrojDRL: Evaluation of Backdoor Attacks on Deep Reinforcement Learning. In 57th ACM\/IEEE Design Automation Conference (DAC)."},{"key":"e_1_3_2_1_33_1","volume-title":"A Survey of Zero-Shot Generalisation in Deep Reinforcement Learning. Journal of Artificial Intelligence Research","author":"Kirk Robert","year":"2023","unstructured":"Robert Kirk, Amy Zhang, Edward Grefenstette, and Tim Rockt\u00e4schel. 2023. A Survey of Zero-Shot Generalisation in Deep Reinforcement Learning. Journal of Artificial Intelligence Research (2023)."},{"key":"e_1_3_2_1_34_1","volume-title":"Chinmay Hegde, and Soumik Sarkar.","author":"Lee Xian Yeow","year":"2020","unstructured":"Xian Yeow Lee, Sambit Ghadai, Kai Liang Tan, Chinmay Hegde, and Soumik Sarkar. 2020. Spatiotemporally Constrained Action Space Attacks on Deep Reinforcement Learning Agents. In AAAI."},{"key":"e_1_3_2_1_35_1","unstructured":"Joel Z Leibo Vinicius Zambaldi Marc Lanctot Janusz Marecki and Thore Graepel. 2017. Multi-Agent Reinforcement Learning in Sequential Social Dilemmas. In AAMAS."},{"key":"e_1_3_2_1_36_1","unstructured":"Zhuohang Li Yi Wu Jian Liu Yingying Chen and Bo Yuan. 2020. AdvPulse: Universal Synchronization-free and Targeted Audio Adversarial Attacks via Subsecond Perturbations. In CCS."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.5555\/3091574.3091594"},{"key":"e_1_3_2_1_38_1","unstructured":"Xiangyu Liu Souradip Chakraborty Yanchao Sun and Furong Huang. 2024. Rethinking Adversarial Policies: A Generalized Attack Formulation and Provable Defense in RL. In ICLR."},{"key":"e_1_3_2_1_39_1","volume-title":"HDRS: A Hybrid Reputation System with Dynamic Update Interval for Detecting Malicious Vehicles in VANETs","author":"Liu Xuejiao","year":"2022","unstructured":"Xuejiao Liu, Oubo Ma, Wei Chen, Yingjie Xia, and Yuxuan Zhou. 2022. HDRS: A Hybrid Reputation System with Dynamic Update Interval for Detecting Malicious Vehicles in VANETs. IEEE Transactions on Intelligent Transportation Systems (2022)."},{"key":"e_1_3_2_1_40_1","volume-title":"OpenAI Pieter Abbeel, and Igor Mordatch","author":"Lowe Ryan","year":"2017","unstructured":"Ryan Lowe, Yi I Wu, Aviv Tamar, Jean Harb, OpenAI Pieter Abbeel, and Igor Mordatch. 2017. Multi-Agent Actor-Critic for Mixed Cooperative-Competitive Environments. In NIPS."},{"key":"e_1_3_2_1_41_1","volume-title":"ABM-V: An Adaptive Backoff Mechanism for Mitigating Broadcast Storm in VANETs","author":"Ma Oubo","year":"2023","unstructured":"Oubo Ma, Xuejiao Liu, and Yingjie Xia. 2023. ABM-V: An Adaptive Backoff Mechanism for Mitigating Broadcast Storm in VANETs. IEEE Transactions on Vehicular Technology (2023)."},{"key":"e_1_3_2_1_42_1","volume-title":"SUB-PLAY: Adversarial Policies against Partially Observed Multi-Agent Reinforcement Learning Systems. arXiv","author":"Ma Oubo","year":"2024","unstructured":"Oubo Ma, Yuwen Pu, Linkang Du, Yang Dai, Ruo Wang, Xiaolei Liu, Yingcai Wu, and Shouling Ji. 2024. SUB-PLAY: Adversarial Policies against Partially Observed Multi-Agent Reinforcement Learning Systems. arXiv (2024)."},{"key":"e_1_3_2_1_43_1","unstructured":"Yuzhe Ma Xuezhou Zhang Wen Sun and Jerry Zhu. 2019. Policy Poisoning in Batch Reinforcement Learning and Control. In NeurIPS."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Suman Maiti Anjana Balabhaskara Sunandan Adhikary Ipsita Koley and Soumyajit Dey. 2023. Targeted Attack Synthesis for Smart Grid Vulnerability Analysis. In CCS.","DOI":"10.1145\/3576915.3623155"},{"key":"e_1_3_2_1_45_1","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A Rusu Joel Veness Marc G Bellemare Alex Graves Martin Riedmiller Andreas K Fidjeland Georg Ostrovski et al. 2015. Human-Level Control through Deep Reinforcement Learning. Nature (2015)."},{"key":"e_1_3_2_1_46_1","volume-title":"Ngoc Duy Nguyen, and Saeid Nahavandi","author":"Nguyen Thanh Thi","year":"2020","unstructured":"Thanh Thi Nguyen, Ngoc Duy Nguyen, and Saeid Nahavandi. 2020. Deep Reinforcement Learning for Multiagent Systems: A Review of Challenges, Solutions, and Applications. IEEE Transactions on Cybernetics (2020)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Nicolas Papernot Patrick McDaniel Somesh Jha Matt Fredrikson Z Berkay Celik and Ananthram Swami. 2016. The Limitations of Deep Learning in Adversarial Settings. In EuroS&P.","DOI":"10.1109\/EuroSP.2016.36"},{"key":"e_1_3_2_1_48_1","unstructured":"Georgios Papoudakis Filippos Christianos and Stefano Albrecht. 2021. Agent Modelling under Partial Observability for Deep Reinforcement Learning. In NeurIPS."},{"key":"e_1_3_2_1_49_1","volume-title":"Gregory Farquhar, Jakob Foerster, and Shimon Whiteson.","author":"Rashid Tabish","year":"2020","unstructured":"Tabish Rashid, Mikayel Samvelyan, Christian Schroeder De Witt, Gregory Farquhar, Jakob Foerster, and Shimon Whiteson. 2020. Monotonic Value Function Factorisation for Deep Multi-Agent Reinforcement Learning. The Journal of Machine Learning Research (2020)."},{"key":"e_1_3_2_1_50_1","volume-title":"The Security of Autonomous Driving: Threats, Defenses, and Future Directions. Proc","author":"Ren Kui","year":"2019","unstructured":"Kui Ren, Qian Wang, Cong Wang, Zhan Qin, and Xiaodong Lin. 2019. The Security of Autonomous Driving: Threats, Defenses, and Future Directions. Proc. IEEE (2019)."},{"key":"e_1_3_2_1_51_1","unstructured":"Unitree Robotics. [n. d.]. New Creature of Embodied AI Unitree Go2. https:\/\/www.unitree.com\/go2."},{"key":"e_1_3_2_1_52_1","volume-title":"Reward is Enough. Artificial Intelligence","author":"Silver David","year":"2021","unstructured":"David Silver, Satinder Singh, Doina Precup, and Richard S Sutton. 2021. Reward is Enough. Artificial Intelligence (2021)."},{"key":"e_1_3_2_1_53_1","unstructured":"Jianwen Sun Tianwei Zhang Xiaofei Xie Lei Ma Yan Zheng Kangjie Chen and Yang Liu. 2020. Stealthy and Efficient Adversarial Attacks against Deep Reinforcement Learning. In AAAI."},{"key":"e_1_3_2_1_54_1","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton Richard S","year":"2018","unstructured":"Richard S Sutton and Andrew G Barto. 2018. Reinforcement Learning: An Introduction. MIT press."},{"key":"e_1_3_2_1_55_1","volume-title":"Neural Cleanse: Identifying and Mitigating Backdoor Attacks in Neural Networks. In S&P.","author":"Wang Bolun","year":"2019","unstructured":"Bolun Wang, Yuanshun Yao, Shawn Shan, Huiying Li, Bimal Viswanath, Haitao Zheng, and Ben Y Zhao. 2019. Neural Cleanse: Identifying and Mitigating Backdoor Attacks in Neural Networks. In S&P."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"crossref","unstructured":"Jinghan Wang Chengyu Song and Heng Yin. 2021. Reinforcement Learning-based Hierarchical Seed Scheduling for Greybox Fuzzing. In NDSS.","DOI":"10.14722\/ndss.2021.24486"},{"key":"e_1_3_2_1_57_1","volume-title":"BACKDOORL: Backdoor Attack against Competitive Reinforcement Learning. In IJCAI.","author":"Wang Lun","year":"2021","unstructured":"Lun Wang, Zaynah Javed, Xian Wu, Wenbo Guo, Xinyu Xing, and Dawn Song. 2021. BACKDOORL: Backdoor Attack against Competitive Reinforcement Learning. In IJCAI."},{"key":"e_1_3_2_1_58_1","volume-title":"Towards Smart Factory for Industry 4.0: A Self-Organized Multi-Agent System with Big Data Based Feedback and Coordination. Computer Networks","author":"Wang Shiyong","year":"2016","unstructured":"Shiyong Wang, Jiafu Wan, Daqiang Zhang, Di Li, and Chunhua Zhang. 2016. Towards Smart Factory for Industry 4.0: A Self-Organized Multi-Agent System with Big Data Based Feedback and Coordination. Computer Networks (2016)."},{"key":"e_1_3_2_1_59_1","unstructured":"Tony Tong Wang Adam Gleave Tom Tseng Kellin Pelrine Nora Belrose Joseph Miller Michael D Dennis Yawen Duan Viktor Pogrebniak Sergey Levine et al. 2023. Adversarial Policies Beat Superhuman Go AIs. In ICML."},{"key":"e_1_3_2_1_60_1","unstructured":"Xian Wu Wenbo Guo Hua Wei and Xinyu Xing. 2021. Adversarial Policy Training against Deep Reinforcement Learning. In USENIX Security."},{"key":"e_1_3_2_1_61_1","volume-title":"RLID-V: Reinforcement Learning-Based Information Dissemination Policy Generation in VANETs","author":"Xia Yingjie","year":"2023","unstructured":"Yingjie Xia, Xuejiao Liu, Jing Ou, and Oubo Ma. 2023. RLID-V: Reinforcement Learning-Based Information Dissemination Policy Generation in VANETs. IEEE Transactions on Intelligent Transportation Systems (2023)."},{"key":"e_1_3_2_1_62_1","volume-title":"SIDE: State Inference for Partially Observable Cooperative Multi-Agent Reinforcement Learning. In AAMAS.","author":"Xu Zhiwei","year":"2022","unstructured":"Zhiwei Xu, Yunpeng Bai, Dapeng Li, Bin Zhang, and Guoliang Fan. 2022. SIDE: State Inference for Partially Observable Cooperative Multi-Agent Reinforcement Learning. In AAMAS."},{"key":"e_1_3_2_1_63_1","volume-title":"Osbert Bastani, Yewen Pu, Armando Solar-Lezama, and Martin Rinard.","author":"Yang Yichen","year":"2021","unstructured":"Yichen Yang, Jeevana Priya Inala, Osbert Bastani, Yewen Pu, Armando Solar-Lezama, and Martin Rinard. 2021. Program Synthesis Guided Reinforcement Learning for Partially Observed Environments. In NeurIPS."},{"key":"e_1_3_2_1_64_1","unstructured":"Yuanshun Yao Huiying Li Haitao Zheng and Ben Y Zhao. 2019. Latent Backdoor Attacks on Deep Neural Networks. In CCS."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"crossref","unstructured":"Yang You Liangwei Li Baisong Guo Weiming Wang and Cewu Lu. 2020. Combinatorial Q-Learning for Dou Di Zhu. In AAAI.","DOI":"10.1609\/aiide.v16i1.7445"},{"key":"e_1_3_2_1_66_1","unstructured":"Honggang Yu Kaichen Yang Teng Zhang Yun-Yun Tsai Tsung-Yi Ho and Yier Jin. 2020. CloudLeak: Large-Scale Deep Learning Models Stealing Through Adversarial Examples. In NDSS."},{"key":"e_1_3_2_1_67_1","volume-title":"AIRS: Explanation for Deep Reinforcement Learning based Security Applications. In USENIX Security.","author":"Yu Jiahao","year":"2023","unstructured":"Jiahao Yu, Wenbo Guo, Qi Qin, Gang Wang, Ting Wang, and Xinyu Xing. 2023. AIRS: Explanation for Deep Reinforcement Learning based Security Applications. In USENIX Security."},{"key":"e_1_3_2_1_68_1","volume-title":"Backdoor Attacks against Deep Reinforcement Learning based Traffic Signal Control Systems. Peer-to-Peer Networking and Applications","author":"Zhang Heng","year":"2023","unstructured":"Heng Zhang, Jun Gu, Zhikun Zhang, Linkang Du, Yongmin Zhang, Yan Ren, Jian Zhang, and Hongran Li. 2023. Backdoor Attacks against Deep Reinforcement Learning based Traffic Signal Control Systems. Peer-to-Peer Networking and Applications (2023)."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2021.3049345"},{"key":"e_1_3_2_1_70_1","unstructured":"Xuezhou Zhang Yuzhe Ma Adish Singla and Xiaojin Zhu. 2020. Adaptive Reward-Poisoning Attacks against Reinforcement Learning. In ICML."},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"crossref","unstructured":"Xin Zhou Xiangyong Wen Zhepei Wang Yuman Gao Haojia Li Qianhao Wang Tiankai Yang Haojian Lu Yanjun Cao Chao Xu et al. 2022. Swarm of Micro Flying Robots in the Wild. Science Robotics (2022). gr","DOI":"10.1126\/scirobotics.abm5954"}],"event":{"name":"CCS '24: ACM SIGSAC Conference on Computer and Communications Security","location":"Salt Lake City UT USA","acronym":"CCS '24","sponsor":["SIGSAC ACM Special Interest Group on Security, Audit, and Control"]},"container-title":["Proceedings of the 2024 on ACM SIGSAC Conference on Computer and Communications Security"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3670293","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3658644.3670293","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T06:04:05Z","timestamp":1755842645000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3658644.3670293"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":71,"alternative-id":["10.1145\/3658644.3670293","10.1145\/3658644"],"URL":"https:\/\/doi.org\/10.1145\/3658644.3670293","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2024-12-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}