{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T15:40:37Z","timestamp":1784216437393,"version":"3.55.0"},"reference-count":194,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62301559"],"award-info":[{"award-number":["62301559"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Commun. Surv. Tutorials"],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/comst.2025.3614494","type":"journal-article","created":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T17:58:51Z","timestamp":1758823131000},"page":"3055-3088","source":"Crossref","is-referenced-by-count":10,"title":["Decision-Making Large Language Model for Wireless Communication: A Comprehensive Survey on Key Techniques"],"prefix":"10.1109","volume":"28","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2316-3526","authenticated-orcid":false,"given":"Ning","family":"Yang","sequence":"first","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2911-3949","authenticated-orcid":false,"given":"Mingrui","family":"Fan","sequence":"additional","affiliation":[{"name":"School of Computer and Communication Engineering, University of Science and Technology Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3536-1729","authenticated-orcid":false,"given":"Wentao","family":"Wang","sequence":"additional","affiliation":[{"name":"Leicester International Institute, Dalian University of Technology, Panjin, Liaoning, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0236-6482","authenticated-orcid":false,"given":"Haijun","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer and Communication Engineering, University of Science and Technology Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Defines IMT-2030 Next-Generation Intelligent Communication Network Architecture","year":"2023"},{"key":"ref2","volume-title":"Study on Future IMT Technologies (IMT-2030)","year":"2024"},{"key":"ref3","article-title":"Co-creating a cyber-physical world with 6G","year":"2024"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2021.3092597"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2022.3217145"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2023.3333326"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/WCNC61545.2025.10978540"},{"key":"ref8","article-title":"Enhancing LLM reasoning with reward-guided tree search","author":"Jiang","year":"2024","journal-title":"arXiv:2411.11694"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2022.3180787"},{"key":"ref10","volume-title":"OpenAI O1 System Card","year":"2024"},{"key":"ref11","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown"},{"issue":"240","key":"ref12","first-page":"1","article-title":"PaLM: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref13","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv:2302.13971"},{"key":"ref14","first-page":"5383","article-title":"Lifelong language pretraining with distribution-specialized experts","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Chen"},{"key":"ref15","article-title":"Multitask prompted training enables zero-shot task generalization","volume-title":"Proc. 10th Int. Conf. Learn. Represent. (ICLR)","author":"Sanh"},{"key":"ref16","first-page":"15476","article-title":"STaR: Bootstrapping reasoning with reasoning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zelikman"},{"key":"ref17","first-page":"9722","article-title":"Ultrafeedback: Boosting language models with scaled AI feedback","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Cui"},{"key":"ref18","first-page":"46534","article-title":"Self-refine: Iterative refinement with self-feedback","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Madaan"},{"key":"ref19","first-page":"8634","article-title":"Reflexion: Language agents with verbal reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shinn"},{"key":"ref20","volume-title":"Defining AI native: A key enabler for advanced intelligent telecom networks","author":"Roeland","year":"2023"},{"key":"ref21","volume-title":"Samsung AI Forum 2023 Day 2: Discussing Technological Trends and the Future of Generative AI","year":"2023"},{"key":"ref22","volume-title":"Transforo Action","year":"2025"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2024.3405075"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2024.3465447"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/MWC.016.2300600"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/LWC.2024.3462556"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/MWC.015.2300404"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2024.3435752"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2025.3527641"},{"key":"ref30","article-title":"Pushing large language models to the 6G edge: Vision, challenges, and opportunities","author":"Lin","year":"2023","journal-title":"arXiv:2309.16739"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/MWC.005.2400019"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2024.128089"},{"key":"ref33","article-title":"Large language models in 6G security: Challenges and opportunities","author":"Nguyen","year":"2024","journal-title":"arXiv:2403.12239"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/OJCOMS.2024.3522103"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/OJVT.2024.3446799"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/mcom.001.2300807"},{"key":"ref37","article-title":"Large language models for forecasting and anomaly detection: A systematic literature review","author":"Su","year":"2024","journal-title":"arXiv:2402.10350"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2020.2969627"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2020.3001225"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3362433"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TCOMM.2024.3382326"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2020.2981320"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2023.3275945"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.2018.1800030"},{"key":"ref45","article-title":"Evolving LLMs\u2019 self-refinement capability via iterative preference optimization","author":"Zeng","year":"2025","journal-title":"arXiv:2502.05605"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2024.3420755"},{"key":"ref47","article-title":"Magentic-one: A generalist multi-agent system for solving complex tasks","author":"Fourney","year":"2024","journal-title":"arXiv:2411.04468"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/comst.2025.3564333"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.23919\/WiOpt58741.2023.10349870"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.23919\/WiOpt58741.2023.10349878"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM42002.2020.9322150"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/TCOMM.2020.3040283"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/TMLCN.2025.3593184"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM54140.2023.10437725"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.3390\/electronics12030516"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512217"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICC.2019.8761775"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM46510.2021.9685424"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.011.2200707"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1145\/3339825.3394938"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/GCWkshp64532.2024.11101012"},{"key":"ref62","article-title":"An empirical study of NetOps capability of pre-trained large language models","author":"Miao","year":"2023","journal-title":"arXiv:2309.05557"},{"key":"ref63","article-title":"Bidirectional encoder representations from transformers (BERT) for question answering in the telecom domain.: Adapting a BERT-like language model to the telecom domain using the electra pre-training approach","author":"Holm","year":"2021"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/COMSNETS59351.2024.10427044"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CCNC54725.2025.10975994"},{"key":"ref66","article-title":"Natural language processing model for log analysis to retrieve solutions for troubleshooting processes","author":"Marzo i Grimalt","year":"2021"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1145\/3639856.3639892"},{"key":"ref68","first-page":"65030","article-title":"Grammar prompting for domain-specific language generation with large language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Wang"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671647"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/TCCN.2024.3482354"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2024.3352100"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1145\/3474838"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.29012\/jpc.880"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29823"},{"key":"ref75","article-title":"Multimodal LLM integrated semantic communications for 6G immersive experiences","author":"Zhang","year":"2025","journal-title":"arXiv:2507.04621"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.91"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.825"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1145\/3655103.3655110"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2022.3219840"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1016\/j.infsof.2025.107734"},{"key":"ref81","article-title":"StructuredRAG: JSON response formatting with large language models","author":"Shorten","year":"2024","journal-title":"arXiv:2408.11061"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/LWC.2025.3578370"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1907.11692"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2644615"},{"key":"ref87","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2074"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"ref90","article-title":"Conditional positional encodings for vision transformers","volume-title":"Proc. 11th Int. Conf. Learn. Representations","author":"Chu"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1007\/s10618-023-00948-2"},{"key":"ref92","first-page":"17413","article-title":"Scatterbrain: Unifying sparse and low-rank attention","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Chen"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3372060"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-naacl.55"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.304"},{"key":"ref96","first-page":"29335","article-title":"DSelect-k: Differentiable selection in the mixture of experts with applications to multi-task learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hazimeh"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.70"},{"key":"ref98","first-page":"3","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Hu"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.252"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref101","article-title":"Efficiently modeling long sequences with structured state spaces","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Gu"},{"key":"ref102","article-title":"Mamba: Linear-time sequence modeling with selective state spaces","volume-title":"Proc. 1st Conf. Lang. Modeling","author":"Gu"},{"key":"ref103","article-title":"Jamba: A hybrid transformer-mamba language model","author":"Lieber","year":"2024","journal-title":"arXiv:2403.19887"},{"key":"ref104","first-page":"28877","article-title":"Do transformers really perform badly for graph representation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Ying"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.379"},{"key":"ref106","article-title":"GreaseLM: Graph REASoning enhanced language models for question answering","volume-title":"Proc. Int. (ICLR)","author":"Zhang"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1109\/CEC60901.2024.10612166"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671647"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1109\/LWC.2025.3529082"},{"key":"ref110","article-title":"Self-refined generative foundation models for wireless traffic prediction","author":"Hu","year":"2024","journal-title":"arXiv:2408.10390"},{"key":"ref111","article-title":"Retrieval-augmented generation for mobile edge computing via large language model","author":"Ren","year":"2024","journal-title":"arXiv:2412.20820"},{"key":"ref112","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447527"},{"key":"ref113","article-title":"Finetuned language models are zero-shot learners","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Lee"},{"key":"ref114","article-title":"Connecting large language models with evolutionary algorithms yields powerful prompt optimizers","volume-title":"Proc. 12th Int. Conf. Learn. Represent.","author":"Guo"},{"key":"ref115","article-title":"Wireless-friendly window position optimization for RIS-aided outdoor-to-indoor networks based on multi-modal large language model","author":"Hou","year":"2024","journal-title":"arXiv:2410.20691"},{"key":"ref116","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"ref117","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive NLP tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lewis"},{"key":"ref118","article-title":"Llm-empowered resource allocation in wireless communications systems","author":"Lee","year":"1999","journal-title":"arXiv:2408.02944"},{"key":"ref119","article-title":"Does prompt formatting have any impact on LLM performance?","author":"He","year":"2024","journal-title":"arXiv:2411.10541"},{"key":"ref120","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Lee"},{"key":"ref121","article-title":"Automatic chain of thought prompting in large language models","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Zhang"},{"key":"ref122","first-page":"11809","article-title":"Tree of thoughts: Deliberate problem solving with large language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yao"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1754"},{"key":"ref124","article-title":"Quiet-STaR: Language models can teach themselves to think before speaking","volume-title":"Proc. 1st Conf. Lang. Modeling","author":"Zelikman"},{"key":"ref125","article-title":"ART: Automatic multi-step reasoning and tool-use for large language models","author":"Paranjape","year":"2023","journal-title":"arXiv:2303.09014"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.23919\/JCIN.2024.10582829"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1109\/VTC2024-Spring62846.2024.10683177"},{"key":"ref128","article-title":"CPL: Critical plan step learning boosts LLM generalization in reasoning tasks","author":"Wang","year":"2024","journal-title":"arXiv:2409.08642"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.313"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.720"},{"key":"ref131","article-title":"C2P: Featuring large language models with causal reasoning","author":"Bagheri","year":"2024","journal-title":"arXiv:2407.18069"},{"key":"ref132","article-title":"Towards CausalGPT: A multi-agent approach for faithful knowledge reasoning via promoting causal consistency in LLMs","author":"Tang","year":"2023","journal-title":"arXiv:2308.11914"},{"key":"ref133","article-title":"To CoT or not to CoT? Chain-of-thought helps mainly on math and symbolic reasoning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Sprague"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.165"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.673"},{"key":"ref136","article-title":"Efficient reasoning with hidden thinking","author":"Shen","year":"2025","journal-title":"arXiv:2501.19201"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1016\/j.swevo.2013.06.001"},{"key":"ref138","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1317"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1137"},{"key":"ref140","article-title":"Large-language-model enabled semantic communication systems","author":"Wang","year":"2024","journal-title":"arXiv:2407.14112"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.1109\/CEC60901.2024.10611913"},{"key":"ref142","doi-asserted-by":"publisher","DOI":"10.1109\/ICCWorkshops59551.2024.10615340"},{"key":"ref143","article-title":"UoMo: A universal model of mobile traffic forecasting for wireless network optimization","author":"Chai","year":"2024","journal-title":"arXiv:2410.15322"},{"key":"ref144","article-title":"CE-CoLLM: Efficient and adaptive large language models through cloud-edge collaboration","author":"Jin","year":"2024","journal-title":"arXiv:2411.02829"},{"key":"ref145","doi-asserted-by":"publisher","DOI":"10.1109\/ICWS62655.2024.00099"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2024.3440503"},{"key":"ref147","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-96-3538-2_13"},{"key":"ref148","doi-asserted-by":"publisher","DOI":"10.1109\/ICoSR63848.2024.00049"},{"key":"ref149","article-title":"OptiMUS: Optimization modeling using MIP solvers and large language models","author":"AhmadiTeshnizi","year":"2023","journal-title":"arXiv:2310.06116"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01202"},{"key":"ref151","doi-asserted-by":"publisher","DOI":"10.1109\/TCCN.2024.3401712"},{"key":"ref152","article-title":"Scaling laws for downstream task performance of large language models","volume-title":"Proc. ICLR Workshop Math. Empirical Understand. Found. Models","author":"Isik"},{"issue":"4","key":"ref153","first-page":"573","article-title":"Diagnosing infeasible models","volume":"62","author":"Chen","year":"2024","journal-title":"INFOR: Inf. Syst. Oper. Res."},{"key":"ref154","article-title":"ChannelGPT: A large model to generate digital twin channel for 6G environment intelligence","author":"Yu","year":"2024","journal-title":"arXiv:2410.13379"},{"key":"ref155","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2024.3384013"},{"key":"ref156","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2010.11929"},{"key":"ref157","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3608134"},{"key":"ref158","article-title":"A survey of reinforcement learning from human feedback","author":"Kaufmann","year":"2023","journal-title":"arXiv:2312.14925"},{"key":"ref159","article-title":"Let\u2019s verify step by step","volume-title":"Proc. 12th Represent.","author":"Lightman"},{"key":"ref160","article-title":"Large language model (LLM)-enabled in-context learning for wireless network optimization: A case study of power control","author":"Zhou","year":"2024","journal-title":"arXiv:2408.00214"},{"key":"ref161","article-title":"Self-refined large language model as automated reward function designer for deep reinforcement learning in robotics","author":"Song","year":"2023","journal-title":"arXiv:2309.06687"},{"key":"ref162","first-page":"3008","article-title":"Learning to summarize with human feedback","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Stiennon"},{"key":"ref163","article-title":"DeepSeekMath: Pushing the limits of mathematical reasoning in open language models","author":"Shao","year":"2024","journal-title":"arXiv:2402.03300"},{"key":"ref164","first-page":"5389","article-title":"Trajectory improvement and reward learning from comparative language feedback","volume-title":"Proc. Conf. Robot Learn.","author":"Yang"},{"key":"ref165","article-title":"Training language models to self-correct via reinforcement learning","volume-title":"Proc. 13th Int. Conf. Learn. Represent.","author":"Kumar"},{"key":"ref166","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2024.3415661"},{"key":"ref167","first-page":"11499","article-title":"Self-generated critiques","volume-title":"Proc. Conf. Nations Americas Chapter Assoc. Comput. Linguistics, Hum. Lang. Technol. (Long Papers)","author":"Yu"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.3390\/electronics13224375"},{"key":"ref169","article-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021","journal-title":"arXiv:2110.14168"},{"key":"ref170","article-title":"Improve mathematical reasoning in language models by automated process supervision","author":"Luo","year":"2024","journal-title":"arXiv:2406.06592"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2024.3395748"},{"key":"ref172","article-title":"Reward design with language models","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Kwon"},{"key":"ref173","article-title":"Eureka: Human-level reward design via coding large language models","volume-title":"Proc. 12th Int. Conf. Learn. Represent.","author":"Ma"},{"key":"ref174","first-page":"53728","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Rafailov"},{"key":"ref175","article-title":"Rewarding progress: Scaling automated process verifiers for LLM reasoning","volume-title":"Proc. 13th Int. Conf. Learn. Represent.","author":"Setlur"},{"key":"ref176","first-page":"58348","article-title":"Token-level direct preference optimization","volume-title":"Proc. Mach. Learn. Res.","author":"Zeng"},{"key":"ref177","first-page":"114289","article-title":"Cal-DPO: Calibrated direct preference optimization for language model alignment","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"37","author":"Xiao"},{"key":"ref178","article-title":"Smaug: Fixing failure modes of preference optimisation with DPO-positive","author":"Pal","year":"2024","journal-title":"arXiv:2402.13228"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.626"},{"key":"ref180","article-title":"Query-dependent prompt evaluation and optimization with offline inverse RL","volume-title":"Proc. 12th Int. Conf. Learn. Represent.","author":"Sun"},{"key":"ref181","article-title":"AutoGen: Enabling next-gen LLM applications via multi-agent conversations","volume-title":"Proc. 1st Conf. Lang. Modeling","author":"Wu"},{"key":"ref182","volume-title":"Crewai, 2023","year":"2025"},{"key":"ref183","volume-title":"Langgraph, 2023","year":"2023"},{"key":"ref184","first-page":"38154","article-title":"HuggingGPT: Solving AI tasks with ChatGPT and its friends in hugging face","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shen"},{"key":"ref185","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.1265"},{"key":"ref186","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.001.2300550"},{"key":"ref187","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2024.3460078"},{"key":"ref188","doi-asserted-by":"publisher","DOI":"10.1109\/ICC51166.2024.10622972"},{"key":"ref189","author":"Bigio","year":"2024","journal-title":"OpenAI Swarm"},{"key":"ref190","article-title":"WirelessAgent: Large language model agents for intelligent wireless networks","author":"Tong","year":"2024","journal-title":"arXiv:2409.07964"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.3233\/faia342"},{"key":"ref192","article-title":"Mastering diverse domains through world models","author":"Hafner","year":"2023","journal-title":"arXiv:2301.04104"},{"key":"ref193","doi-asserted-by":"publisher","DOI":"10.23919\/JCIN.2024.10582827"},{"key":"ref194","article-title":"Wireless multi-agent generative AI: From connected intelligence to collective intelligence","author":"Zou","year":"2023","journal-title":"arXiv:2307.02757"}],"container-title":["IEEE Communications Surveys &amp; Tutorials"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9739\/11321210\/11180008.pdf?arnumber=11180008","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,11]],"date-time":"2026-02-11T20:58:14Z","timestamp":1770843494000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11180008\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":194,"URL":"https:\/\/doi.org\/10.1109\/comst.2025.3614494","relation":{},"ISSN":["1553-877X","2373-745X"],"issn-type":[{"value":"1553-877X","type":"electronic"},{"value":"2373-745X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}