{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T06:16:30Z","timestamp":1783059390630,"version":"3.54.6"},"reference-count":14,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Netw. Lett."],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/lnet.2026.3699505","type":"journal-article","created":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T20:09:10Z","timestamp":1780430950000},"page":"263-267","source":"Crossref","is-referenced-by-count":0,"title":["Delay-Aware Reinforcement Learning for O-RAN Control Under Stochastic Reward Feedback"],"prefix":"10.1109","volume":"8","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-3044-6080","authenticated-orcid":false,"given":"Xingqi","family":"Wu","sequence":"first","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of Michigan&#x2013;Dearborn, Dearborn, MI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0618-9345","authenticated-orcid":false,"given":"Junaid","family":"Farooq","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of Michigan&#x2013;Dearborn, Dearborn, MI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8966-7239","authenticated-orcid":false,"given":"Tao","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Systems Engineering, City University of Hong Kong, Kowloon, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7726-4926","authenticated-orcid":false,"given":"Juntao","family":"Chen","sequence":"additional","affiliation":[{"name":"Department of Computer and Information Science, Fordham University, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9004-7253","authenticated-orcid":false,"given":"Ying","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Systems Engineering, Stevens Institute of Technology, Hoboken, NJ, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.101.2001120"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/WCNC51071.2022.9771643"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOMWKSHPS57453.2023.10226154"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.23919\/IFIPNetworking62109.2024.10619886"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2023.3320933"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/MCOMSTD.0001.2200041"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2023.3297883"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/VTC2021-Fall52928.2021.9625281"},{"key":"ref9","first-page":"1453","article-title":"Online learning under delayed feedback","volume-title":"Proc. 30th Int. Conf. Mach. Learn. (ICML)","author":"Csaba"},{"key":"ref10","first-page":"1270","article-title":"Online learning with adversarial delays","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"28","author":"Quanrud"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CDC49753.2023.10384003"},{"key":"ref12","first-page":"8280","article-title":"Off-policy reinforcement learning with delayed rewards","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Han"},{"key":"ref13","article-title":"Reinforcement learning with delayed, composite, and partially anonymous reward","author":"Mondal","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref14","article-title":"RUDDER: Return decomposition for delayed rewards","volume-title":"Proc. 33rd Conf. Neural Inf. Process. Syst. (NeurIPS)","volume":"32","author":"Arjona-Medina"}],"container-title":["IEEE Networking Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/8253410\/11411875\/11547201.pdf?arnumber=11547201","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T05:23:07Z","timestamp":1783056187000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11547201\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":14,"URL":"https:\/\/doi.org\/10.1109\/lnet.2026.3699505","relation":{},"ISSN":["2576-3156"],"issn-type":[{"value":"2576-3156","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}