{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,20]],"date-time":"2026-03-20T16:20:37Z","timestamp":1774023637195,"version":"3.50.1"},"reference-count":27,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T00:00:00Z","timestamp":1734307200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T00:00:00Z","timestamp":1734307200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,16]]},"DOI":"10.1109\/cdc56724.2024.10886169","type":"proceedings-article","created":{"date-parts":[[2025,2,26]],"date-time":"2025-02-26T18:43:32Z","timestamp":1740595412000},"page":"6051-6056","source":"Crossref","is-referenced-by-count":1,"title":["Reinforcement Learning for Joint Resource and Power Allocation in D2D Communications"],"prefix":"10.1109","author":[{"given":"Ifrah","family":"Saeed","sequence":"first","affiliation":[{"name":"The University of Melbourne,Department of Electrical and Electronic Engineering,Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andrew C.","family":"Cullen","sequence":"additional","affiliation":[{"name":"The University of Melbourne,School of Computing and Information Systems,Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zainab R.","family":"Zaidi","sequence":"additional","affiliation":[{"name":"The University of Melbourne,Department of Electrical and Electronic Engineering,Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sarah","family":"Erfani","sequence":"additional","affiliation":[{"name":"The University of Melbourne,School of Computing and Information Systems,Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tansu","family":"Alpcan","sequence":"additional","affiliation":[{"name":"The University of Melbourne,Department of Electrical and Electronic Engineering,Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/OJCOMS.2022.3192040"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/MCOM.2016.7432168"},{"key":"ref3","volume-title":"Public Safety Mobile Broadband Strategic Review","year":"2022"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2020.3032991"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/PIMRC.2015.7343481"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2015.2511924"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ANZCC47194.2019.8945719"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CDC42340.2020.9303805"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CDC49753.2023.10383216"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICC42927.2021.9501055"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM46510.2021.9685485"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2020.3014926"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2022.3208572"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2021.3131534"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM38437.2019.9014095"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/LANMAN52105.2021.9478814"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CCNC46108.2020.9045215"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"ref19","first-page":"2961","article-title":"Actor-Attention-Critic for Multi-agent Reinforcement Learning","volume-title":"Proceedings of the International Conference on Machine Learning","volume":"97","author":"Iqbal"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN52387.2021.9533975"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3389400.3389404"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3067665.3067668"},{"key":"ref23","volume-title":"Study on LTE Device to Device Proximity Services (Release 12)","year":"2014"},{"key":"ref24","article-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"ref26","article-title":"Global Optimization Toolbox version: 4.8 (R2022b)","year":"2023"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.6028\/nist.ir.8372"}],"event":{"name":"2024 IEEE 63rd Conference on Decision and Control (CDC)","location":"Milan, Italy","start":{"date-parts":[[2024,12,16]]},"end":{"date-parts":[[2024,12,19]]}},"container-title":["2024 IEEE 63rd Conference on Decision and Control (CDC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10885784\/10885785\/10886169.pdf?arnumber=10886169","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,27]],"date-time":"2025-02-27T07:51:49Z","timestamp":1740642709000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10886169\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,16]]},"references-count":27,"URL":"https:\/\/doi.org\/10.1109\/cdc56724.2024.10886169","relation":{},"subject":[],"published":{"date-parts":[[2024,12,16]]}}}