{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T09:47:53Z","timestamp":1784454473444,"version":"3.55.0"},"reference-count":66,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Beijing Municipal Natural Science Foundation","award":["L241018"],"award-info":[{"award-number":["L241018"]}]},{"name":"National Key R&amp;D Program of China","award":["2023YFB3308200"],"award-info":[{"award-number":["2023YFB3308200"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62373026"],"award-info":[{"award-number":["62373026"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beihang World TOP University Cooperation Program"},{"name":"Academic Excellence Foundation of BUAA"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Netw. Sci. Eng."],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/tnse.2025.3616360","type":"journal-article","created":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T17:40:09Z","timestamp":1759340409000},"page":"4406-4421","source":"Crossref","is-referenced-by-count":2,"title":["Toward Diffusion-Based Deep Reinforcement Learning for Discrete Decision-Making: Methods and Evaluations"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0444-8828","authenticated-orcid":false,"given":"Zhen","family":"Chen","sequence":"first","affiliation":[{"name":"Hangzhou International Innovation Institute, Beihang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7108-3655","authenticated-orcid":false,"given":"Xiao","family":"Long","sequence":"additional","affiliation":[{"name":"Hangzhou International Innovation Institute, Beihang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1989-6102","authenticated-orcid":false,"given":"Lin","family":"Zhang","sequence":"additional","affiliation":[{"name":"Hangzhou International Innovation Institute, Beihang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0183-3835","authenticated-orcid":false,"given":"Wentong","family":"Cai","sequence":"additional","affiliation":[{"name":"College of Computing and Data Science, Nanyang Technological University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2020.3005252"},{"key":"ref2","volume-title":"Industrial Applications of Combinatorial Optimization","volume":"16","author":"Yu","year":"2013"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/JSAC.2024.3459037"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2023.3298888"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.cor.2021.105400"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1142\/s1793962325500400"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/MWC.016.2300547"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2025.3541078"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2024.3361474"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01736"},{"key":"ref11","first-page":"2256","article-title":"Deep unsupervised learning using nonequilibrium thermodynamics","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sohl-Dickstein","year":"2015"},{"key":"ref12","article-title":"Diffusion model for planning: A systematic literature review","author":"Ubukata","year":"2024"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/iros60139.2025.11247358"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"ref16","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","volume-title":"Proc. 33rd Int. Conf. Mach. Learn.","author":"Wang","year":"2016"},{"key":"ref17","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. 33rd Int. Conf. Mach. Learn.","author":"Mnih","year":"2016"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.12794\/metadc1505267"},{"key":"ref19","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Haarnoja","year":"2018"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.2023.3287655"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1016\/j.rcim.2022.102412"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.cie.2023.109053"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2024.3521885"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2025.3546100"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TNSE.2022.3201121"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.003.2200641"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.131375"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.35833\/MPCE.2024.000391"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/s10845-024-02513-0"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.jmsy.2025.01.010"},{"key":"ref31","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref32","article-title":"Score-based generative modeling through stochastic differential aligns","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Song","year":"2020"},{"key":"ref33","first-page":"11895","article-title":"Generative modeling by estimating gradients of the data distribution","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Song","year":"2019"},{"key":"ref34","first-page":"2256","article-title":"Deep unsupervised learning using nonequilibrium thermodynamics","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sohl-Dickstein","year":"2015"},{"key":"ref35","first-page":"2021","article-title":"Denoising diffusion implicit models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Song"},{"key":"ref36","first-page":"8162","article-title":"Improved denoising diffusion probabilistic models","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","volume":"139","author":"Nichol","year":"2021"},{"key":"ref37","first-page":"17981","article-title":"Structured denoising diffusion models in discrete state-spaces","volume":"34","author":"Austin","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref38","article-title":"DiffuSeq: Sequence to sequence text generation with diffusion models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gong","year":"2023"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/3626235"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TIFS.2025.3570202"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2024.3400011"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2024.3356178"},{"key":"ref43","first-page":"741","article-title":"Stochastic latent actor-critic: Deep reinforcement learning with a latent variable model","volume":"33","author":"X Lee","year":"2020","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref44","first-page":"594","article-title":"Dexpoint: Generalizable point cloud reinforcement learning for sim-to-real dexterous manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Qin","year":"2023"},{"key":"ref45","first-page":"652","article-title":"Pointnet: Deep learning on point sets for 3D classification and segmentation","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit.","author":"Qi","year":"2017"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/MCS.2025.3534477"},{"key":"ref47","first-page":"4565","article-title":"Generative adversarial imitation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ho","year":"2016"},{"key":"ref48","first-page":"2052","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto","year":"2019"},{"key":"ref49","article-title":"Learning an embedding space for transferable robot skills","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hausman","year":"2018"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-64793-3_15"},{"key":"ref51","first-page":"1352","article-title":"Reinforcement learning with deep energy-based policies","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja","year":"2017"},{"key":"ref52","article-title":"Diffusion policies as an expressive policy class for offline reinforcement learning","author":"Wang","year":"2022"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3363530"},{"key":"ref54","first-page":"67195","article-title":"Efficient diffusion policies for offline reinforcement learning","volume":"36","author":"Kang","journal-title":"Proc. Adv. Neural Inf. Process. Syst."},{"key":"ref55","article-title":"Offline reinforcement learning via high-fidelity generative behavior modeling","author":"Chen","year":"2022"},{"key":"ref56","article-title":"IDQL: Implicit Q-learning as an actor-critic method with diffusion policies","author":"Hansen-Estruch","year":"2023"},{"key":"ref57","article-title":"DiffCPS: Diffusion model based constrained policy search for offline reinforcement learning","author":"He","year":"2023"},{"key":"ref58","first-page":"22825","article-title":"Contrastive energy prediction for exact energy-guided diffusion sampling in offline reinforcement learning","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","volume":"202","author":"Lu","year":"2023"},{"key":"ref59","article-title":"Beyond conservatism: Diffusion policies in offline multiagent reinforcement learning","author":"Li","year":"2023"},{"key":"ref60","first-page":"335","article-title":"Boosting continuous control with consistency policy","volume-title":"Proc. 23rd Int. Conf. Autono. Agents Multiagent Syst.","author":"Chen","year":"2024"},{"key":"ref61","article-title":"Discrete diffusion reward guidance methods for offline reinforcement learning","author":"Coleman","year":"2023"},{"key":"ref62","article-title":"Offline reinforcement learning with discrete diffusion skills","author":"Qiao","year":"2025"},{"key":"ref63","article-title":"Jumanji: A diverse suite of scalable reinforcement learning environments in JAX","author":"Bonnet","year":"2024","journal-title":"arXiv:2306.09884"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2017.43"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1145\/3326285.3329074"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2022.06.012"}],"container-title":["IEEE Transactions on Network Science and Engineering"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6488902\/11264281\/11185339.pdf?arnumber=11185339","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T18:25:06Z","timestamp":1766773506000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11185339\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":66,"URL":"https:\/\/doi.org\/10.1109\/tnse.2025.3616360","relation":{},"ISSN":["2327-4697","2334-329X"],"issn-type":[{"value":"2327-4697","type":"electronic"},{"value":"2334-329X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}