{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,20]],"date-time":"2026-02-20T07:08:20Z","timestamp":1771571300848,"version":"3.50.1"},"reference-count":47,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T00:00:00Z","timestamp":1760400000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T00:00:00Z","timestamp":1760400000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,14]]},"DOI":"10.1109\/ictc66702.2025.11387958","type":"proceedings-article","created":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T20:55:29Z","timestamp":1771534529000},"page":"618-623","source":"Crossref","is-referenced-by-count":0,"title":["Forging Agentic AI: A Comprehensive Survey on the Symbiotic Convergence of Large Language Models and Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Minsoo","family":"Kim","sequence":"first","affiliation":[{"name":"Ajou Univerity,Dept. Aritificial Intelligence Convergence Network,Suwon,South Korea,16499"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joongheon","family":"Kim","sequence":"additional","affiliation":[{"name":"Korea University,Dept. Electrical and Computer Engineering,Seoul,South Korea,02841"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Soyi","family":"Jung","sequence":"additional","affiliation":[{"name":"Ajou University,Dept. Electrical and Computer Engineering,Suwon,South Korea,16499"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref3","article-title":"Reinforcement learning: An introduction","volume":"2","author":"Sutton","year":"2018","journal-title":"MIT Press"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TNSM.2024.3392393"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/7503.003.0006"},{"issue":"159","key":"ref6","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Advances in Neural Information Processing Systems (NeurIPS)","volume":"33","author":"Brown"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3450851"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2025.3576732"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.23919\/JCN.2025.000001"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref11","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. 32nd International Conference on Machine Learning","volume":"37","author":"Schulman"},{"key":"ref12","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref13","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. 33rd International Conference on Machine Learning (ICML)","volume":"48","author":"Mnih"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2023.3283235"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2021.3062418"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1023\/A:1013689704352"},{"key":"ref17","article-title":"R*: Efficient reward design via reward structure evolution and parameter alignment optimization with large language models","volume-title":"Proc. 42nd International Conference on Machine Learning (ICML)","author":"Li"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i15.33679"},{"issue":"2338","key":"ref19","first-page":"53 728","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Advances in Neural Information Processing Systems 36 (NeurIPS 2023)","author":"Rafailov","year":"2023"},{"key":"ref20","article-title":"Hallucination-aware generative pretrained transformer for cooperative aerial mobility control","author":"Ahn","year":"2025"},{"key":"ref21","article-title":"LLM-Explorer: A plug-in reinforcement learning policy exploration enhancement driven by large language models","author":"Hao","year":"2025"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657683"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.3390\/math13121932"},{"key":"ref24","article-title":"Semantically aligned task decomposition in multi-agent reinforcement learning","author":"Li","year":"2023"},{"key":"ref25","article-title":"Option discovery using LLM-guided semantic hierarchical reinforcement learning","author":"Shek","year":"2025"},{"key":"ref26","article-title":"Think twice, act once: A co-evolution framework of LLM and RL for large-scale decision making","volume-title":"Proc. 42nd International Conference on Machine Learning (ICML)","author":"Wan"},{"key":"ref27","article-title":"Reinforce LLM reasoning through multi-agent reflection","volume-title":"Proc. 42nd International Conference on Machine Learning (ICML)","author":"Yuan"},{"issue":"2106","key":"ref28","first-page":"51 348","article-title":"LLM-empowered state representation for reinforcement learning","volume-title":"Proc. 41st International Conference on Machine Learning (ICML)","author":"Wang"},{"issue":"3110","key":"ref29","first-page":"98 009","article-title":"Do LLMs build world representations? Probing through the lens of state abstraction","volume-title":"Proc. 38th Conference on Neural Information Processing Systems (NeurIPS)","author":"Li"},{"key":"ref30","article-title":"Efficient reinforcement learning with large language model priors","volume-title":"Proc. 13th International Conference on Learning Representations (ICLR)","author":"Yan"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2025.XXI.018"},{"key":"ref32","article-title":"Combining LLM decision and RL action selection to improve rl policy for adaptive interventions","author":"Karine","year":"2025"},{"issue":"627","key":"ref33","article-title":"Large language model as a policy teacher for training reinforcement learning agents","volume-title":"Proc. Thirty-Third International Joint Conference on Artificial Intelligence (IJCAI-24)","author":"Zhou"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0495"},{"key":"ref35","first-page":"24 631","article-title":"Prompting decision transformer for few-shot policy generalization","volume-title":"Proc. 39th International Conference on Machine Learning (ICML)","volume":"162","author":"Xu"},{"key":"ref36","article-title":"The synergy of LLMs & RL unlocks offline learning of generalizable language-conditioned policies with low-fidelity data","volume-title":"Proc. 42nd International Conference on Machine Learning (ICML)","author":"Pouplin"},{"key":"ref37","article-title":"RRO: LLM agent optimization through rising reward trajectories","author":"Wang","year":"2025"},{"key":"ref38","article-title":"ELEMENTAL: Interactive learning from demonstrations and vision-language models for reward design in robotics","volume-title":"Proc. 42nd International Conference on Machine Learning (ICML)","author":"Chen"},{"key":"ref39","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","author":"Ahn","year":"2022"},{"issue":"346","key":"ref40","first-page":"8657","article-title":"Guiding pretraining in reinforcement learning with large language models","volume-title":"Proc. 40th International Conference on Machine Learning (ICML)","volume":"202","author":"Du"},{"issue":"48","key":"ref41","first-page":"1009","article-title":"Read and reap the rewards: Learning to play atari with the help of instruction manuals","volume-title":"Proc. Advances in Neural Information Processing Systems (NeurIPS)","volume":"36","author":"Wu"},{"key":"ref42","article-title":"Envgen: Generating and adapting environments via LLMs for training embodied agents","volume-title":"Proc. Conference on Language Modeling (COLM)","author":"Zala"},{"key":"ref43","article-title":"Steer LLM latents for hallucination detection","volume-title":"Proc. 42nd International Conference on Machine Learning (ICML)","author":"Park"},{"key":"ref44","article-title":"Hallucinot: Hallucination detection through context and common knowledge verification","author":"Paudel","year":"2025"},{"key":"ref45","article-title":"Improving data efficiency for LLM reinforcement fine-tuning through difficulty-targeted online data selection and rollout replay","author":"Sun","year":"2025"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2025.3577527"},{"key":"ref47","article-title":"Co-reinforcement learning for unified multimodal understanding and generation","author":"Jiang","year":"2025"}],"event":{"name":"2025 16th International Conference on Information and Communication Technology Convergence (ICTC)","location":"Jeju, Korea, Republic of","start":{"date-parts":[[2025,10,14]]},"end":{"date-parts":[[2025,10,17]]}},"container-title":["2025 16th International Conference on Information and Communication Technology Convergence (ICTC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11387762\/11387791\/11387958.pdf?arnumber=11387958","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,20]],"date-time":"2026-02-20T06:41:16Z","timestamp":1771569676000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11387958\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,14]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/ictc66702.2025.11387958","relation":{},"subject":[],"published":{"date-parts":[[2025,10,14]]}}}