{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T11:44:19Z","timestamp":1775475859886,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":15,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,2,2]]},"DOI":"10.1145\/3795496.3795713","type":"proceedings-article","created":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T10:44:48Z","timestamp":1775472288000},"page":"51-58","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Global Reward Generation Framework based on Large Language Models for Intelligent Air Combat"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-0579-9484","authenticated-orcid":false,"given":"Lianmeng","family":"Zhou","sequence":"first","affiliation":[{"name":"Xinjiang University, School of Computer Science and Technology, Urumqi, Xinjiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8082-539X","authenticated-orcid":false,"given":"Minchi","family":"Kuang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Department of Precision Instrument, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3314-927X","authenticated-orcid":false,"given":"Heng","family":"Shi","sequence":"additional","affiliation":[{"name":"Tsinghua University, Department of Precision Instrument, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6830-1211","authenticated-orcid":false,"given":"Jihong","family":"Zhu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Department of Precision Instrument, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,6]]},"reference":[{"key":"e_1_3_3_1_1_2","volume-title":"An intelligent maneuver decision-making approach for air combat based on deep reinforcement learning and transformer networks.\u00a0Entropy\u00a026, 12","author":"Li Wentao","year":"2024","unstructured":"Wentao Li, Feng Fang, Dongliang Peng, and Shuning Han. 2024. An intelligent maneuver decision-making approach for air combat based on deep reinforcement learning and transformer networks.\u00a0Entropy\u00a026, 12 (2024), 1036."},{"key":"e_1_3_3_1_2_2","volume-title":"Deep reinforcement learning-based air-to-air combat maneuver generation in a realistic environment.\u00a0IEEE Access\u00a011","author":"Bae Jung Ho","year":"2023","unstructured":"Jung Ho Bae, Hoseong Jung, Seogbong Kim, Sungho Kim, and Yong-Duk Kim. 2023. Deep reinforcement learning-based air-to-air combat maneuver generation in a realistic environment.\u00a0IEEE Access\u00a011 (2023), 26427\u201326440."},{"key":"e_1_3_3_1_3_2","volume-title":"Maneuver decision-making through automatic curriculum reinforcement learning without handcrafted reward functions.\u00a0Applied Sciences\u00a013, 16","author":"Wei Yujie","year":"2023","unstructured":"Yujie Wei, Hongpeng Zhang, Yuan Wang, and Changqiang Huang. 2023. Maneuver decision-making through automatic curriculum reinforcement learning without handcrafted reward functions.\u00a0Applied Sciences\u00a013, 16 (2023), 9421."},{"key":"e_1_3_3_1_4_2","volume-title":"In\u00a0Proceedings of the 33rd International Joint Conference on Artificial Intelligence (IJCAI \u201924)","author":"Chen Xinning","year":"2024","unstructured":"Xinning Chen, Xuan Liu, Yanwen Ba, Shigeng Zhang, Bo Ding, and Kenli Li. 2024. Selective learning for sample-efficient training in multi-agent sparse reward tasks. In\u00a0Proceedings of the 33rd International Joint Conference on Artificial Intelligence (IJCAI \u201924). 8384\u20138388."},{"key":"e_1_3_3_1_5_2","volume-title":"Revisiting sparse rewards for goal-reaching reinforcement learning.\u00a0Reinforcement Learning Journal\u00a04","author":"Vasan Gautham","year":"2024","unstructured":"Gautham Vasan, Yan Wang, Fahim Shahriar, James Bergstra, Martin J\u00e4gersand, and A. Rupam Mahmood. 2024. Revisiting sparse rewards for goal-reaching reinforcement learning.\u00a0Reinforcement Learning Journal\u00a04 (2024), 1841\u20131854."},{"key":"e_1_3_3_1_6_2","volume-title":"Comprehensive overview of reward engineering and shaping in advancing reinforcement learning applications.\u00a0IEEE Access\u00a012","author":"Ibrahim Sinan","year":"2024","unstructured":"Sinan Ibrahim, Mostafa Mostafa, Ali Jnadi, Hadi Salloum, and Pavel Osinenko. 2024. Comprehensive overview of reward engineering and shaping in advancing reinforcement learning applications.\u00a0IEEE Access\u00a012 (2024), 175473\u2013175500."},{"key":"e_1_3_3_1_7_2","volume-title":"In\u00a0Proceedings of the International Conference on Learning Representations (ICLR).","author":"Ma Yecheng Jason","year":"2024","unstructured":"Yecheng Jason Ma, William Liang, Guanzhi Wang, De-An Huang, Osbert Bastani, Dinesh Jayaraman, Yuke Zhu, Linxi Fan, and Anima Anandkumar. 2024. Eureka: Human-level reward design via coding large language models. In\u00a0Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_1_8_2","first-page":"11809","article-title":"Tree of thoughts: deliberate problem solving with large language models","volume":"36","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Thomas L. Griffiths, Yuan Cao, and Karthik Narasimhan. 2023. Tree of thoughts: deliberate problem solving with large language models. In\u00a0Advances in Neural Information Processing Systems (NeurIPS), Vol. 36. 11809\u201311822.","journal-title":"In\u00a0Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_3_1_9_2","volume-title":"In\u00a0Proceedings of the 39th International Conference on Machine Learning (ICML). 9118\u20139147","author":"Huang Wenlong","year":"2022","unstructured":"Wenlong Huang, Pieter Abbeel, Deepak Pathak, and Igor Mordatch. 2022. Language models as zero-shot planners: extracting actionable knowledge for embodied agents. In\u00a0Proceedings of the 39th International Conference on Machine Learning (ICML). 9118\u20139147."},{"key":"e_1_3_3_1_10_2","volume-title":"Large language models for UAVs: current state and pathways to the future.\u00a0IEEE Open Journal of Vehicular Technology\u00a05","author":"Javaid Shumaila","year":"2024","unstructured":"Shumaila Javaid, Hamza Fahim, Bin He, and Nasir Saeed. 2024. Large language models for UAVs: current state and pathways to the future.\u00a0IEEE Open Journal of Vehicular Technology\u00a05 (2024), 1166\u201311692."},{"key":"e_1_3_3_1_11_2","volume-title":"In\u00a0Proceedings of the International Conference on Learning Representations (ICLR).","author":"Xie Tianbao","year":"2024","unstructured":"Tianbao Xie, Siheng Zhao, Chen Henry Wu, Yitao Liu, Qian Luo, Victor Zhong, Yanchao Yang, and Tao Yu. 2024. Text2Reward: reward shaping with language models for reinforcement learning. In\u00a0Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_1_12_2","unstructured":"John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal policy optimization algorithms.\u00a0arXiv preprint arXiv:1707.06347\u00a0(2017)."},{"key":"e_1_3_3_1_13_2","volume-title":"Long short-term memory.\u00a0Neural Computation\u00a09, 8","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory.\u00a0Neural Computation\u00a09, 8 (1997), 1735\u20131780."},{"key":"e_1_3_3_1_14_2","first-page":"5998","article-title":"Attention is all you need","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. In\u00a0Advances in Neural Information Processing Systems (NIPS), Vol. 30. 5998\u20136008.","journal-title":"In\u00a0Advances in Neural Information Processing Systems (NIPS)"},{"key":"e_1_3_3_1_15_2","unstructured":"Daya Guo Dejian Yang Haowei Zhang et al. 2025. DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning.\u00a0Nature\u00a0645 (2025) 633\u2013638."}],"event":{"name":"AIACT 2026: 2026 10th International Conference on Artificial Intelligence, Automation and Control Technologies","location":"Sydney Australia","acronym":"AIACT 2026"},"container-title":["Proceedings of the 2026 10th International Conference on Artificial Intelligence, Automation and Control Technologies"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3795496.3795713","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T10:45:25Z","timestamp":1775472325000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3795496.3795713"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,2]]},"references-count":15,"alternative-id":["10.1145\/3795496.3795713","10.1145\/3795496"],"URL":"https:\/\/doi.org\/10.1145\/3795496.3795713","relation":{},"subject":[],"published":{"date-parts":[[2026,2,2]]},"assertion":[{"value":"2026-04-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}