{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T02:55:25Z","timestamp":1730343325300,"version":"3.28.0"},"reference-count":22,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,7,8]],"date-time":"2024-07-08T00:00:00Z","timestamp":1720396800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,7,8]],"date-time":"2024-07-08T00:00:00Z","timestamp":1720396800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,7,8]]},"DOI":"10.23919\/fusion59988.2024.10706358","type":"proceedings-article","created":{"date-parts":[[2024,10,11]],"date-time":"2024-10-11T17:19:50Z","timestamp":1728667190000},"page":"1-7","source":"Crossref","is-referenced-by-count":0,"title":["A Deep Reinforcement Learning-Based Whittle Index Policy for Multibeam Allocation"],"prefix":"10.23919","author":[{"given":"Yuhang","family":"Hao","sequence":"first","affiliation":[{"name":"Northwestern Polytechnical University,School of Automation,Xi&#x2019;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zengfu","family":"Wang","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University,School of Automation,Xi&#x2019;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Fu","sequence":"additional","affiliation":[{"name":"Royal Melbourne Institute of Technology University,School of Engineering,Melbourne,Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Quan","family":"Pan","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University,School of Automation,Xi&#x2019;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TAES.2022.3203688"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TAES.2021.3138869"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/RadarConf2043947.2020.9266550"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2020.2976587"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2019.2919778"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1049\/iet-rsn.2019.0178"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.23919\/FUSION49465.2021.9627058"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.dt.2019.10.001"},{"issue":"1","key":"ref9","first-page":"253","article-title":"A non-myopic and fast resource scheduling algorithm for multi-target tracking of space-based radar considering optimal integrated performance","volume":"13","author":"Wang","year":"2024","journal-title":"Journal of Radars"},{"key":"ref10","article-title":"Nonmyopic beam scheduling for multiple smart target tracking in phased array radar network","author":"Hao","year":"2023","journal-title":"arXiv preprint arXiv:2312.07858"},{"key":"ref11","first-page":"888","article-title":"Optimal policy for scheduling of Gauss-Markov systems","volume-title":"Proceedings of the Seventh International Conference on Information Fusion","author":"Howard"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.2307\/3214163"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2009.5400949"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-43904-4_15"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TVT.2020.2966659"},{"key":"ref16","first-page":"828839","article-title":"NeurWIN: Neural Whittle index network for restless bandits via deep RL","volume":"34","author":"Nakhleh","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref17","first-page":"1711","article-title":"When are Kalman-filter restless bandits indexable?","volume-title":"Proceedings of the 28th International Conference on Neural Information Processing Systems","volume":"1","author":"Dance"},{"key":"ref18","article-title":"Q-learning in enormous action spaces via amortized approximate maximization","author":"Van de Wiele","year":"2020","journal-title":"arXiv preprint arXiv:2001.08116"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.5772\/intechopen.80600"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2021.3087747"},{"journal-title":"Deep Learning","year":"2016","author":"Goodfellow","key":"ref21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/2623330.2623612"}],"event":{"name":"2024 27th International Conference on Information Fusion (FUSION)","start":{"date-parts":[[2024,7,8]]},"location":"Venice, Italy","end":{"date-parts":[[2024,7,11]]}},"container-title":["2024 27th International Conference on Information Fusion (FUSION)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10706250\/10706251\/10706358.pdf?arnumber=10706358","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,12]],"date-time":"2024-10-12T04:33:53Z","timestamp":1728707633000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10706358\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,8]]},"references-count":22,"URL":"https:\/\/doi.org\/10.23919\/fusion59988.2024.10706358","relation":{},"subject":[],"published":{"date-parts":[[2024,7,8]]}}}