{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T18:01:23Z","timestamp":1755799283470,"version":"3.44.0"},"reference-count":13,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,7,1]],"date-time":"2019-07-01T00:00:00Z","timestamp":1561939200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,7,1]],"date-time":"2019-07-01T00:00:00Z","timestamp":1561939200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,7]]},"DOI":"10.1109\/icmlc48188.2019.8949232","type":"proceedings-article","created":{"date-parts":[[2020,1,8]],"date-time":"2020-01-08T01:08:48Z","timestamp":1578445728000},"page":"1-6","source":"Crossref","is-referenced-by-count":2,"title":["Comments on \u201cFinite-Time Analysis of the Multiarmed Bandit Problem\u201d"],"prefix":"10.1109","author":[{"given":"Lu-Ning","family":"Zhang","sequence":"first","affiliation":[{"name":"China University of Petroleum, Beijing Campus (CUP),Department of Automation,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Zuo","sequence":"additional","affiliation":[{"name":"China University of Petroleum, Beijing Campus (CUP),Department of Automation,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian-Wei","family":"Liu","sequence":"additional","affiliation":[{"name":"China University of Petroleum, Beijing Campus (CUP),Department of Automation,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei-Min","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer Engineering and Technology Shanghai University,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nobuyasu","family":"Ito","sequence":"additional","affiliation":[{"name":"Center for Computational Science, RIKEN,Kobe,Hyogo,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","first-page":"64","article-title":"Multi-armed bandits with application to 5G small cells","volume":"23 3","author":"setareh","year":"2016","journal-title":"IEEE Wireless Communications"},{"key":"ref11","first-page":"1350","article-title":"Active Learning on Heterogeneous Information Networks: A Multi-armed Bandit Approach","author":"doris","year":"0","journal-title":"2018 IEEE International Conference on Data Mining (ICDM) IEEE"},{"key":"ref12","first-page":"374","article-title":"A neural networks committee for the contextual bandit problem","author":"allesiardo","year":"0","journal-title":"Proceedings of International Conference on Neural Information Processing Springer Cham"},{"key":"ref13","article-title":"Handbook of mathematical functions","volume":"55","author":"abramowitz","year":"1964","journal-title":"National Bureau of Standards Applied Mathematics Series"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref3","first-page":"397","article-title":"Using confidence bounds for exploitation-exploration trade-offs","volume":"3","author":"peter","year":"2002","journal-title":"Journal of Machine Learning Research"},{"key":"ref6","first-page":"8388","article-title":"Spatio-Temporal Edge Service Placement: A Bandit Learning Approach","volume":"17 12","author":"lixing","year":"2018","journal-title":"IEEE Transactions on Wireless Communications"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"484","DOI":"10.1038\/nature16961","article-title":"Mastering the game of Go with deep neural networks and tree search","volume":"529","author":"silver","year":"2016","journal-title":"Nature"},{"key":"ref8","first-page":"2094","article-title":"Deep reinforcement learning with double q-learning","author":"van hasselt","year":"0","journal-title":"THIRTIETH AAAI Conference on Artificial Intelligence"},{"key":"ref7","first-page":"282","article-title":"Bandit based monte-carlo planning","author":"levente","year":"2006","journal-title":"European conference on machine learning Springer"},{"journal-title":"Reinforcement Learning An Introduction","year":"2018","author":"sutton","key":"ref2"},{"key":"ref1","doi-asserted-by":"crossref","first-page":"235","DOI":"10.1023\/A:1013689704352","article-title":"Finite-time analysis of the multiarmed bandit problem","volume":"47","author":"peter","year":"2002","journal-title":"Machine Learning"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1214\/14-STS504"}],"event":{"name":"2019 International Conference on Machine Learning and Cybernetics (ICMLC)","start":{"date-parts":[[2019,7,7]]},"location":"Kobe, Japan","end":{"date-parts":[[2019,7,10]]}},"container-title":["2019 International Conference on Machine Learning and Cybernetics (ICMLC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8942645\/8949161\/08949232.pdf?arnumber=8949232","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,19]],"date-time":"2025-08-19T18:09:59Z","timestamp":1755626999000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8949232\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,7]]},"references-count":13,"URL":"https:\/\/doi.org\/10.1109\/icmlc48188.2019.8949232","relation":{},"subject":[],"published":{"date-parts":[[2019,7]]}}}