{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T13:02:36Z","timestamp":1771074156018,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":21,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,19]]},"DOI":"10.1145\/3788731.3788758","type":"proceedings-article","created":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T11:49:28Z","timestamp":1771069768000},"page":"172-179","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Eval-PPO: Building an Efficient Threat Evaluator Using Proximal Policy Optimization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-4312-5954","authenticated-orcid":false,"given":"Wuzhou","family":"Sun","sequence":"first","affiliation":[{"name":"School of Computer and Artificial Intelligence, Southwest Jiaotong University, Chengdu, Sichuan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1677-531X","authenticated-orcid":false,"given":"Siyi","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence, Southwest Jiaotong University, Chengdu, Sichuan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9121-2687","authenticated-orcid":false,"given":"Haonan","family":"Luo","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence, Southwest Jiaotong University, Chengdu, Sichuan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0672-3790","authenticated-orcid":false,"given":"Pengpeng","family":"Zeng","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Tongji University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8690-8164","authenticated-orcid":false,"given":"Jie","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence, Southwest Jiaotong University, Chengdu, Sichuan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6949-3673","authenticated-orcid":false,"given":"Ji","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence, Southwest Jiaotong University, Chengdu, Sichuan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,2,14]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"Do as I Can, Not as I Say: Grounding Language in Robotic Affordances","author":"Ahn Michael","year":"2022","unstructured":"Michael Ahn, Anthony Brohan, Noah Brown, Yevgen Chebotar, Omar Cortes, Byron David, Chelsea Finn, Chuyuan Fu, Keerthana Gopalakrishnan, Karol Hausman, Alexander Herzog, Daniel Ho, Jasmine Hsu, Julian Ibarz, Brian Ichter, Alex Irpan, Eric Jang, Rosario\u00a0Jauregui Ruano, Kyle Jeffrey, Sally Jesmonth, Nikhil Joshi, Ryan Julian, Dmitry Kalashnikov, Yuheng Kuang, Kuang-Huei Lee, Sergey Levine, Yao Lu, Linda Luu, Carolina Parada, Peter Pastor, Jorn Quiambao, Kanishka Rao, Jarek Rettinghouse, Diego Reyes, Pierre Sermanet, Nicolas Sievers, Clayton Tan, Alexander Toshev, Vincent Vanhoucke, Fei Xia, Ted Xiao, Peng Xu, Sichun Xu, Mengyuan Yan, and Andy Zeng. 2022. Do as I Can, Not as I Say: Grounding Language in Robotic Affordances. arXiv:https:\/\/arXiv.org\/abs\/2204.01691https:\/\/arxiv.org\/abs\/2204.01691"},{"key":"e_1_3_3_1_3_2","volume-title":"Dota 2 with Large Scale Deep Reinforcement Learning","author":"Berner Christopher","year":"2019","unstructured":"Christopher Berner, Greg Brockman, Brooke Chan, Vicki Cheung, Przemys\u0142aw Debiak, Christy Dennison, David Farhi, Quirin Fischer, Pieter He, Jakub Ho, Julia Hockenmaier, Christian Lebiere, Jeff Clune, and Igor Mordatch. 2019. Dota 2 with Large Scale Deep Reinforcement Learning. arXiv:https:\/\/arXiv.org\/abs\/1912.06680https:\/\/arxiv.org\/abs\/1912.06680 Policy gradient methods for reinforcement learning."},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","unstructured":"Nan Ding Keisuke Takeda and Kazuki Fujii. 2022. Deep Reinforcement Learning in a Racket Sport for Player Evaluation With Technical and Tactical Contexts. IEEE Access 10 (2022) 54764\u201354772. 10.1109\/ACCESS.2022.3173938","DOI":"10.1109\/ACCESS.2022.3173938"},{"key":"e_1_3_3_1_5_2","unstructured":"Shengyi Huang Huayu Cui Xiaotian Bai and Jiaji Li. 2022. CleanRL: High-Quality Single-File Implementations of Deep Reinforcement Learning Algorithms. Journal of Machine Learning Research 23 274 (2022) 1\u201318. https:\/\/jmlr.org\/papers\/v23\/21-1342.html"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/IGIC.2013.6659136"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581348"},{"key":"e_1_3_3_1_8_2","first-page":"1928","volume-title":"Proceedings of the 33rd International Conference on Machine Learning","author":"Mnih Volodymyr","year":"2016","unstructured":"Volodymyr Mnih, Adri\u00e0\u00a0Puigdom\u00e8nech Badia, Mehdi Mirza, Alex Graves, Timothy\u00a0P. Lillicrap, Tim Harley, David Silver, and Koray Kavukcuoglu. 2016. Asynchronous Methods for Deep Reinforcement Learning. In Proceedings of the 33rd International Conference on Machine Learning. 1928\u20131937. https:\/\/proceedings.mlr.press\/v48\/mniha16.html"},{"key":"e_1_3_3_1_9_2","volume-title":"Playing Atari with Deep Reinforcement Learning","author":"Mnih Volodymyr","year":"2013","unstructured":"Volodymyr Mnih, Koray Kavukcuoglu, David Silver, Alex Graves, Ioannis Antonoglou, Daan Wierstra, and Martin Riedmiller. 2013. Playing Atari with Deep Reinforcement Learning. arXiv:https:\/\/arXiv.org\/abs\/1312.5602https:\/\/arxiv.org\/abs\/1312.5602"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/CoG51982.2022.9893617"},{"key":"e_1_3_3_1_11_2","volume-title":"Proximal Policy Optimization Algorithms","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. arXiv:https:\/\/arXiv.org\/abs\/1707.06347https:\/\/arxiv.org\/abs\/1707.06347"},{"key":"e_1_3_3_1_12_2","volume-title":"Value-Decomposition Networks for Cooperative Multi-Agent Learning","author":"Sunehag Peter","year":"2017","unstructured":"Peter Sunehag, Guy Lever, Audrunas Gruslys, Wojciech\u00a0M. Czarnecki, Vinicius Zambaldi, Max Jaderberg, Marc Lanctot, Nicolas Sonnerat, Joel\u00a0Z. Leibo, Karl Tuyls, and David Silver. 2017. Value-Decomposition Networks for Cooperative Multi-Agent Learning. arXiv:https:\/\/arXiv.org\/abs\/1706.05296https:\/\/arxiv.org\/abs\/1706.05296"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","unstructured":"Richard\u00a0S. Sutton and Andrew\u00a0G. Barto. 1998. Reinforcement Learning: An Introduction. IEEE Transactions on Neural Networks 9 5 (1998) 1054\u20131054. 10.1109\/TNN.1998.712192","DOI":"10.1109\/TNN.1998.712192"},{"key":"e_1_3_3_1_14_2","volume-title":"Advances in Neural Information Processing Systems 12","author":"Sutton Richard\u00a0S.","year":"1999","unstructured":"Richard\u00a0S. Sutton, David\u00a0A. McAllester, Satinder\u00a0P. Singh, and Yishay Mansour. 1999. Policy Gradient Methods for Reinforcement Learning with Function Approximation. In Advances in Neural Information Processing Systems 12. MIT Press. https:\/\/proceedings.neurips.cc\/paper\/1999\/file\/464d828159cc5aab8f8c0e1c5d5c2e73-Paper.pdf"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","unstructured":"Oriol Vinyals Igor Babuschkin Wojciech\u00a0M. Czarnecki Micha\u00ebl Mathieu Andrew Dudzik Julian Schrittwieser Gabriel de Guez Edward Lockhart Aja Huang Marcin Hubicka Timothy\u00a0P. Lillicrap David Silver and Demis Hassabis. 2019. Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575 (2019) 350\u2013354. 10.1038\/s41586-019-1724-z","DOI":"10.1038\/s41586-019-1724-z"},{"key":"e_1_3_3_1_16_2","volume-title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","author":"Wang Guanzhi","year":"2023","unstructured":"Guanzhi Wang, Yuqi Xie, Yunfan Jiang, Anima Anandkumar, Chaowei Xiao, Yuke Zhu, Linxi Fan, and Ajay Mandlekar. 2023. Voyager: An Open-Ended Embodied Agent with Large Language Models. arXiv:https:\/\/arXiv.org\/abs\/2305.16291https:\/\/arxiv.org\/abs\/2305.16291"},{"key":"e_1_3_3_1_17_2","unstructured":"Jie Wang Zhendong Yang Liansong Zong Xiaobo Zhang Dexian Wang and Ji Zhang. 2025. Dual-branch Prompting for Multimodal Machine Translation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.17588 (2025)."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Qi Xiong Kai Tang Minbo Ma Ji Zhang Jie Xu and Tianrui Li. 2025. Modeling temporal dependencies within the target for long-term time series forecasting. IEEE Transactions on Knowledge and Data Engineering (2025).","DOI":"10.1109\/TKDE.2025.3609415"},{"key":"e_1_3_3_1_19_2","unstructured":"Ji Zhang Xu Luo Lianli Gao Difan Zou Hengtao Shen and Jingkuan Song. 2023. From Channel Bias to Feature Redundancy: Uncovering the\" Less is More\" Principle in Few-Shot Learning. arXiv e-prints (2023) arXiv\u20132310."},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Ji Zhang Jingkuan Song Lianli Gao Nicu Sebe and Heng\u00a0Tao Shen. 2025. Reliable Few-shot Learning under Dual Noises. IEEE Transactions on Pattern Analysis and Machine Intelligence (2025).","DOI":"10.1109\/TPAMI.2025.3584051"},{"key":"e_1_3_3_1_21_2","unstructured":"Ji Zhang Shihan Wu Lianli Gao Jingkuan Song Nicu Sebe and Heng\u00a0Tao Shen. 2025. A Closer Look at Conditional Prompt Tuning for Vision-Language Models. International Journal of Computer Vision (2025)."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","unstructured":"Yiming Zhao Yujing Lu Jian Zhao Wei Zhou and Hongxia Li. 2024. DanZero+: Dominating the GuanDan Game Through Reinforcement Learning. IEEE Transactions on Games 16 4 (2024) 914\u2013926. 10.1109\/TG.2023.3305999","DOI":"10.1109\/TG.2023.3305999"}],"event":{"name":"EILM 2025: 2025 International Conference on Embodied Intelligence and Large Models","location":"Chengdu China","acronym":"EILM 2025"},"container-title":["Proceedings of the 2025 International Conference on Embodied Intelligence and Large Models"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3788731.3788758","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T12:09:58Z","timestamp":1771070998000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3788731.3788758"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,19]]},"references-count":21,"alternative-id":["10.1145\/3788731.3788758","10.1145\/3788731"],"URL":"https:\/\/doi.org\/10.1145\/3788731.3788758","relation":{},"subject":[],"published":{"date-parts":[[2025,12,19]]},"assertion":[{"value":"2026-02-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}