{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T02:50:20Z","timestamp":1777690220558,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T00:00:00Z","timestamp":1531958400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Key Research and Development Program of China","award":["2016YFB1000904"],"award-info":[{"award-number":["2016YFB1000904"]}]},{"name":"National Natural Science Foundation of China","award":["61702190, U1609220"],"award-info":[{"award-number":["61702190, U1609220"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,7,19]]},"DOI":"10.1145\/3219819.3219961","type":"proceedings-article","created":{"date-parts":[[2018,7,19]],"date-time":"2018-07-19T13:05:12Z","timestamp":1532005512000},"page":"2447-2456","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":169,"title":["Supervised Reinforcement Learning with Recurrent Neural Network for Dynamic Treatment Recommendation"],"prefix":"10.1145","author":[{"given":"Lu","family":"Wang","sequence":"first","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Zhang","sequence":"additional","affiliation":[{"name":"East China Normal University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaofeng","family":"He","sequence":"additional","affiliation":[{"name":"East China Normal University, Shnaghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongyuan","family":"Zha","sequence":"additional","affiliation":[{"name":"Georgia Tech, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2018,7,19]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1002\/sim.4512"},{"key":"e_1_3_2_2_3_1","volume-title":"Predicting Medications from Diagnostic Codes with Recurrent Neural Networks. ICLR","author":"Bajor Jacek M","year":"2017"},{"key":"e_1_3_2_2_4_1","unstructured":"Andrew G Barto . 2002. Reinforcement Learning in Motor Control. In The handbook of brain theory and neural networks.   Andrew G Barto . 2002. Reinforcement Learning in Motor Control. In The handbook of brain theory and neural networks."},{"key":"e_1_3_2_2_5_1","volume-title":"supervised actor-critic reinforcement learning. Handbook of learning and approximate dynamic programming","author":"Barto MTRAG","year":"2004"},{"key":"e_1_3_2_2_6_1","volume-title":"Biped dynamic walking using reinforcement learning. Robotics and Autonomous Systems","author":"Benbrahim Hamid","year":"1997"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","volume-title":"Statistical methods for dynamic treatment regimes","author":"Chakraborty Bibhas","DOI":"10.1007\/978-1-4614-7428-9"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1186\/s12859-016-1421-y"},{"key":"e_1_3_2_2_9_1","volume-title":"Andy Schuetz, Walter F. Stewart, and Jimeng Sun .","author":"Choi Edward","year":"2016"},{"key":"e_1_3_2_2_10_1","volume-title":"International Workshop on Machine Learning. 92--110","author":"Jeffery"},{"key":"e_1_3_2_2_11_1","volume-title":"Model-free reinforcement learning with continuous action in practice American Control Conference (ACC)","author":"Degris Thomas"},{"key":"e_1_3_2_2_12_1","unstructured":"Chelsea Finn Sergey Levine and Pieter Abbeel . 2016. Guided cost learning: Deep inverse optimal control via policy optimization ICML. 49--58.   Chelsea Finn Sergey Levine and Pieter Abbeel . 2016. Guided cost learning: Deep inverse optimal control via policy optimization ICML. 49--58."},{"key":"e_1_3_2_2_13_1","volume-title":"A Pilot SMART for Developing an Adaptive Treatment Strategy for Adolescent Depression. J Clin Child Adolesc Psychol","author":"Gunlicksstoessel M","year":"2017"},{"key":"e_1_3_2_2_14_1","volume-title":"Deep recurrent q-learning for partially observable mdps. CoRR, abs\/1507.06527","author":"Hausknecht Matthew","year":"2015"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"crossref","unstructured":"Jianying Hu Adam Perer and Fei Wang . 2016. Data driven analytics for personalized healthcare. In Healthcare Information Management Systems. 529--554.  Jianying Hu Adam Perer and Fei Wang . 2016. Data driven analytics for personalized healthcare. In Healthcare Information Management Systems. 529--554.","DOI":"10.1007\/978-3-319-20765-0_31"},{"key":"e_1_3_2_2_16_1","volume-title":"et almbox","author":"Johnson Alistair EW","year":"2016"},{"key":"e_1_3_2_2_17_1","unstructured":"Vijay R Konda and John N Tsitsiklis . 2000. Actor-critic algorithms. In NIPS. 1008--1014.   Vijay R Konda and John N Tsitsiklis . 2000. Actor-critic algorithms. In NIPS. 1008--1014."},{"key":"e_1_3_2_2_18_1","volume-title":"End-to-end training of deep visuomotor policies. JMLR","author":"Levine Sergey","year":"2016"},{"key":"e_1_3_2_2_19_1","unstructured":"Sergey Levine Zoran Popovic and Vladlen Koltun . 2011. Nonlinear inverse reinforcement learning with gaussian processes NIPS. 19--27.   Sergey Levine Zoran Popovic and Vladlen Koltun . 2011. Nonlinear inverse reinforcement learning with gaussian processes NIPS. 19--27."},{"key":"e_1_3_2_2_20_1","volume-title":"Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971","author":"Lillicrap Timothy P","year":"2015"},{"key":"e_1_3_2_2_21_1","volume-title":"The demise of early goal-directed therapy for severe sepsis and septic shock. Acta Anaesthesiologica Scandinavica","author":"Marik P. E.","year":"2015"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1017940631555"},{"key":"e_1_3_2_2_23_1","volume-title":"et almbox","author":"Mnih Volodymyr","year":"2015"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1111\/1467-9868.00389"},{"key":"e_1_3_2_2_25_1","volume-title":"Clifford","author":"Nemati Shamim","year":"2016"},{"key":"e_1_3_2_2_26_1","volume-title":"Corey Chivers, Michael Draugelis, and Barbara E Engelhardt .","author":"Prasad Niranjani","year":"2017"},{"key":"e_1_3_2_2_27_1","volume-title":"Deep Reinforcement Learning for Sepsis Treatment. arXiv preprint arXiv:1711.09602","author":"Raghu Aniruddh","year":"2017"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143936"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/0270-0255(86)90088-6"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btn141"},{"key":"e_1_3_2_2_31_1","volume-title":"Moodie","author":"Shortreed Susan M.","year":"2012"},{"key":"e_1_3_2_2_32_1","unstructured":"David Silver Guy Lever Nicolas Heess Thomas Degris Daan Wierstra and Martin Riedmiller . 2014. Deterministic policy gradient algorithms. In ICML. 387--395.   David Silver Guy Lever Nicolas Heess Thomas Degris Daan Wierstra and Martin Riedmiller . 2014. Deterministic policy gradient algorithms. In ICML. 387--395."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939866"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-010-5229-0"},{"key":"e_1_3_2_2_35_1","unstructured":"Richard S Sutton David A McAllester Satinder P Singh and Yishay Mansour . 2000. Policy gradient methods for reinforcement learning with function approximation NIPS. 1057--1063.   Richard S Sutton David A McAllester Satinder P Singh and Yishay Mansour . 2000. Policy gradient methods for reinforcement learning with function approximation NIPS. 1057--1063."},{"key":"e_1_3_2_2_36_1","volume-title":"Personalized Prescription for Comorbidity. In International Conference on Database Systems for Advanced Applications. 3-19","author":"Wang Lu","year":"2018"},{"key":"e_1_3_2_2_37_1","volume-title":"Machine learning","author":"Watkins Christopher JCH","year":"1992"},{"key":"e_1_3_2_2_38_1","volume-title":"Representation and Reinforcement Learning for Personalized Glycemic Control in Septic Patients. arXiv preprint arXiv:1712.00654","author":"Weng Wei-Hung","year":"2017"},{"key":"e_1_3_2_2_39_1","volume-title":"Solving deep memory POMDPs with recurrent policy gradients International Conference on Artificial Neural Networks","author":"Wierstra Daan"},{"key":"e_1_3_2_2_40_1","volume-title":"AMIA Summits on Translational Science Proceedings","author":"Zhang Ping","year":"2014"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098109"},{"key":"e_1_3_2_2_42_1","volume-title":"Reinforcement Learning Strategies for Clinical Trials in Nonsmall Cell Lung Cancer. Biometrics","author":"Zhao Yufan","year":"2011"},{"key":"e_1_3_2_2_43_1","volume-title":"A Physician Advisory System for Chronic Heart Failure management based on knowledge patterns. Theory and Practice of Logic Programming","author":"Zhuo Chen","year":"2016"},{"key":"e_1_3_2_2_44_1","unstructured":"Brian D Ziebart Andrew L Maas J Andrew Bagnell and Anind K Dey . 2008. Maximum Entropy Inverse Reinforcement Learning.. In AAAI. 1433--1438.   Brian D Ziebart Andrew L Maas J Andrew Bagnell and Anind K Dey . 2008. Maximum Entropy Inverse Reinforcement Learning.. In AAAI. 1433--1438."}],"event":{"name":"KDD '18: The 24th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining","location":"London United Kingdom","acronym":"KDD '18","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 24th ACM SIGKDD International Conference on Knowledge Discovery &amp; Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3219819.3219961","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3219819.3219961","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:07:22Z","timestamp":1750212442000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3219819.3219961"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,7,19]]},"references-count":44,"alternative-id":["10.1145\/3219819.3219961","10.1145\/3219819"],"URL":"https:\/\/doi.org\/10.1145\/3219819.3219961","relation":{},"subject":[],"published":{"date-parts":[[2018,7,19]]},"assertion":[{"value":"2018-07-19","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}