{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T00:49:20Z","timestamp":1780447760229,"version":"3.54.1"},"reference-count":5,"publisher":"Association for Computing Machinery (ACM)","issue":"2","license":[{"start":{"date-parts":[[2023,9,28]],"date-time":"2023-09-28T00:00:00Z","timestamp":1695859200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["SIGMETRICS Perform. Eval. Rev."],"published-print":{"date-parts":[[2023,9,28]]},"abstract":"<jats:p>For many years, reinforcement learning (RL) has proven to be very successful in solving a wide variety of learning and decision making under uncertainty (DMuU) problems, including those related to game playing and robotic control. Many different RL approaches, with varying levels of success, have been developed to address these problems.<\/jats:p>","DOI":"10.1145\/3626570.3626585","type":"journal-article","created":{"date-parts":[[2023,10,2]],"date-time":"2023-10-02T22:16:57Z","timestamp":1696285017000},"page":"39-41","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Markov Decision Process Framework for Control-Based Reinforcement Learning"],"prefix":"10.1145","volume":"51","author":[{"given":"Yingdong","family":"Lu","sequence":"first","affiliation":[{"name":"IBM Research, Yorktown Heights, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mark S.","family":"Squillante","sequence":"additional","affiliation":[{"name":"IBM Research, Yorktown Heights, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chai","family":"Wah Wu","sequence":"additional","affiliation":[{"name":"IBM Research, Yorktown Heights, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,2]]},"reference":[{"issue":"98","key":"e_1_2_1_1_1","first-page":"1","article-title":"On the theory of policy gradient methods: Optimality, approximation, and distribution shift","volume":"22","author":"Agarwal A.","year":"2021","unstructured":"A. Agarwal, et al. On the theory of policy gradient methods: Optimality, approximation, and distribution shift. JMLR, 22(98):1--76, 2021.","journal-title":"JMLR"},{"key":"e_1_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/81.168933"},{"key":"e_1_2_1_3_1","volume-title":"Preprint","author":"Lu Y.","year":"2023","unstructured":"Y. Lu, M.S. Squillante, C.W. Wu. Markov Decision Process Framework for Control-Based Reinforcement Learning. Preprint, May 2023."},{"key":"e_1_2_1_4_1","volume-title":"Neural network dynamics for model-based deep reinforcement learning with model-free fine-tuning. arXiv:1708.02596v2","author":"Nagabandi A.","year":"2017","unstructured":"A. Nagabandi, et al. Neural network dynamics for model-based deep reinforcement learning with model-free fine-tuning. arXiv:1708.02596v2, 2017."},{"key":"e_1_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10898-011-9659-4"}],"container-title":["ACM SIGMETRICS Performance Evaluation Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626570.3626585","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3626570.3626585","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:45Z","timestamp":1750178205000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626570.3626585"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,28]]},"references-count":5,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,9,28]]}},"alternative-id":["10.1145\/3626570.3626585"],"URL":"https:\/\/doi.org\/10.1145\/3626570.3626585","relation":{},"ISSN":["0163-5999"],"issn-type":[{"value":"0163-5999","type":"print"}],"subject":[],"published":{"date-parts":[[2023,9,28]]},"assertion":[{"value":"2023-10-02","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}