{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T18:27:46Z","timestamp":1766773666273,"version":"3.48.0"},"reference-count":71,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2024YFE0200600"],"award-info":[{"award-number":["2024YFE0200600"]}]},{"name":"Zhejiang Provincial Natural Science Foundation of China","award":["LR23F010005"],"award-info":[{"award-number":["LR23F010005"]}]},{"name":"Huawei Cooperation Project","award":["TC20240829036"],"award-info":[{"award-number":["TC20240829036"]}]},{"name":"National Key Laboratory of Wireless Communications Foundation","award":["2023KP01601"],"award-info":[{"award-number":["2023KP01601"]}]},{"name":"Big Data and Intelligent Computing Key Lab of CQUPT","award":["BDIC-2023-B-001"],"award-info":[{"award-number":["BDIC-2023-B-001"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Commun."],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1109\/tcomm.2025.3606417","type":"journal-article","created":{"date-parts":[[2025,9,4]],"date-time":"2025-09-04T18:21:43Z","timestamp":1757010103000},"page":"14594-14609","source":"Crossref","is-referenced-by-count":0,"title":["Conditional Diffusion Model With OOD Mitigation as High-Dimensional Offline Resource Allocation Planner in Clustered Ad Hoc Networks"],"prefix":"10.1109","volume":"73","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7621-9203","authenticated-orcid":false,"given":"Kechen","family":"Meng","sequence":"first","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-7667-2999","authenticated-orcid":false,"given":"Sinuo","family":"Zhang","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4297-5060","authenticated-orcid":false,"given":"Rongpeng","family":"Li","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chan","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Lei","sequence":"additional","affiliation":[{"name":"College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5479-7890","authenticated-orcid":false,"given":"Zhifeng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang Laboratory and the College of Information Science and Electronic Engineering, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-34736-3_1"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2022.3201121"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2022.3145857"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2022.3188679"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CCAA.2015.7148429"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2009.59"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-021-04301-9"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2022.3185159"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"article-title":"Continuous control with deep reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Lillicrap","key":"ref10"},{"key":"ref11","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CDS49703.2020.00051"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2022.3233592"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2019.01.003"},{"key":"ref15","first-page":"14975","article-title":"OPAL: Offline primitive discovery for accelerating offline reinforcement learning","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Ajay"},{"key":"ref16","first-page":"4754","article-title":"Deep reinforcement learning in a handful of trials using probabilistic dynamics models","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","volume":"31","author":"Chua"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"ref18","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-642-27645-3_2","volume-title":"Reinforcement Learning: State-of-the-Art","author":"Lange","year":"2012"},{"key":"ref19","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Fujimoto"},{"key":"ref20","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020","journal-title":"arXiv:2005.01643"},{"key":"ref21","first-page":"14129","article-title":"MOPO: Model-based offline policy optimization","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","author":"Yu"},{"key":"ref22","article-title":"Reward-consistent dynamics models are strongly generalizable for offline reinforcement learning","author":"Luo","year":"2023","journal-title":"arXiv:2310.05422"},{"key":"ref23","first-page":"3603","article-title":"Implicit generation and modeling with energy based models","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","volume":"32","author":"Du"},{"article-title":"Auto-encoding variational Bayes","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Kingma","key":"ref24"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-658-40442-0_9"},{"key":"ref26","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Proces. Syst. (NeurIPS)","author":"Brown"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/LCOMM.2024.3499745"},{"key":"ref29","article-title":"Deep diffusion deterministic policy gradient based performance optimization for Wi-Fi networks","author":"Liu","year":"2024","journal-title":"arXiv:2404.15684"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2024.3486728"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/3ICT.2018.8855753"},{"key":"ref32","first-page":"1179","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Proces. Syst. (NeurIPS)","author":"Kumar"},{"key":"ref33","first-page":"20876","article-title":"Offline reinforcement learning with implicit Q-learning","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Kostrikov"},{"key":"ref34","first-page":"15084","article-title":"Decision transformer: Reinforcement learning via sequence modeling","volume-title":"Proc. Adv. Neural Inf. Proces. Syst. (NeurIPS)","author":"Chen"},{"key":"ref35","first-page":"9902","article-title":"Planning with diffusion for flexible behavior synthesis","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"J\u00e4nner"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/INFCOM.1998.659669"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3074594"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICTC49870.2020.9289080"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.23919\/IFIPNetworking52078.2021.9472801"},{"key":"ref40","first-page":"2323","article-title":"Chaineddiffuser: Unifying trajectory diffusion and keypose prediction for robotic manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Xian"},{"key":"ref41","first-page":"64896","article-title":"Diffusion model is an effective planner and data synthesizer for multi-task reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","author":"He"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2024.3400011"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOMWKSHPS54753.2022.9798257"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/MWC.009.2300165"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2024.3525410"},{"key":"ref46","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Haarnoja"},{"issue":"178","key":"ref47","first-page":"1","article-title":"Monotonic value function factorisation for deep multi-agent reinforcement learning","volume":"21","author":"Rashid","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref48","first-page":"8647","article-title":"Exploring model-based planning with policy networks","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Wang"},{"key":"ref49","first-page":"465","article-title":"PILCO: A model-based and data-efficient approach to policy search","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Deisenroth"},{"key":"ref50","first-page":"1273","article-title":"Offline reinforcement learning as one big sequence modeling problem","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","author":"J\u00e4nner"},{"key":"ref51","first-page":"28954","article-title":"COMBO: Conservative offline model-based policy optimization","volume-title":"Proc. Adv. Neural Inf. Proces. Syst. (NeurIPS)","author":"Yu"},{"key":"ref52","first-page":"38815","article-title":"Recovering from out-of-sample states via inverse dynamics in offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","author":"Jiang"},{"key":"ref53","first-page":"33177","article-title":"Model-Bellman inconsistency for model-based offline reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sun"},{"key":"ref54","first-page":"12519","article-title":"When to trust your model: Model-based policy optimization","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","author":"J\u00e4nner"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3116668"},{"key":"ref56","article-title":"Understanding diffusion models: A unified perspective","author":"Luo","year":"2022","journal-title":"arXiv:2208.11970"},{"key":"ref57","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume-title":"Proc. Adv. Neural Inf. Proces. Syst. (NeurIPS)","author":"Ho"},{"key":"ref58","first-page":"29598","article-title":"Is conditional generative modeling all you need for decision-making?","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Ajay"},{"key":"ref59","first-page":"8229","article-title":"Learning Markov state abstractions for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","author":"Allen"},{"key":"ref60","first-page":"8780","article-title":"Diffusion models beat GANs on image synthesis","volume-title":"Proc. Adv. Neural Inf. Proces. Syst.","author":"Dhariwal"},{"key":"ref61","article-title":"Classifier-free diffusion guidance","author":"Ho","year":"2022","journal-title":"arXiv:2207.12598"},{"key":"ref62","first-page":"143","article-title":"DART: Noise injection for robust imitation learning","volume-title":"Proc. Conf. Robot Learn. (CoRL)","author":"Laskey"},{"key":"ref63","first-page":"11918","article-title":"Generative modeling by estimating gradients of the data distribution","volume-title":"Proc. Adv. Neural Inf. Proces. Syst. (NeurIPS)","author":"Song"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i8.20886"},{"key":"ref65","first-page":"2878","article-title":"Fighting uncertainty with gradients: Offline reinforcement learning via diffusion score matching","volume-title":"Proc. Conf. Robot Learn. (CoRL)","author":"Suh"},{"volume-title":"Elementary Classical Analysis","year":"1974","author":"Marsden","key":"ref66"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-3675-0"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3068889"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/JRPROC.1946.234568"},{"key":"ref71","first-page":"14205","article-title":"Denoising diffusion implicit models","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Song"}],"container-title":["IEEE Transactions on Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/26\/11314911\/11151241.pdf?arnumber=11151241","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T18:23:53Z","timestamp":1766773433000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11151241\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12]]},"references-count":71,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tcomm.2025.3606417","relation":{},"ISSN":["0090-6778","1558-0857"],"issn-type":[{"type":"print","value":"0090-6778"},{"type":"electronic","value":"1558-0857"}],"subject":[],"published":{"date-parts":[[2025,12]]}}}