{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,24]],"date-time":"2026-01-24T13:47:07Z","timestamp":1769262427719,"version":"3.49.0"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,15]],"date-time":"2024-12-15T00:00:00Z","timestamp":1734220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,15]],"date-time":"2024-12-15T00:00:00Z","timestamp":1734220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,15]]},"DOI":"10.1109\/wsc63780.2024.10838808","type":"proceedings-article","created":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T18:40:24Z","timestamp":1737398424000},"page":"2595-2606","source":"Crossref","is-referenced-by-count":1,"title":["Distortion Risk Measure-Based Deep Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Jinyang","family":"Jiang","sequence":"first","affiliation":[{"name":"Wuhan Institute for Artificial Intelligence, Guanghua School of Management, Peking University,Beijing,CHINA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bernd","family":"Heidergott","sequence":"additional","affiliation":[{"name":"Vrije Universiteit Amsterdam,Amsterdam,THE NETHERLANDS"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaqiao","family":"Hu","sequence":"additional","affiliation":[{"name":"Stony Brook University,Dept. of Applied Mathematics and Statistics,Stony Brook,NY,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yijie","family":"Peng","sequence":"additional","affiliation":[{"name":"Wuhan Institute for Artificial Intelligence, Guanghua School of Management, Peking University,Beijing,CHINA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"449","article-title":"A Distributional Perspective on Reinforcement Learning","volume-title":"Proceedings of the 34th International Conference on Machine Learning","author":"Bellemare","year":"2017"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/sj.jors.2600425"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2014.2309262"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6911(01)00152-9"},{"issue":"167","key":"ref5","first-page":"1","article-title":"Risk-Constrained Reinforcement Learning with Percentile Risk Criteria","volume":"18","author":"Chow","year":"2018","journal-title":"Journal of Machine Learning Research"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1002\/0470016450"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s13385-012-0058-0"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.1090.1090"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.35.11.1367"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1287\/ijoc.2020.1016"},{"key":"ref12","first-page":"1861","article-title":"Soft Actor-Critic: Off-Policy Maximum Entropy Deep Reinforcement Learning with a Stochastic Actor","volume-title":"Proceedings of the 35th International Conference on Machine Learning","author":"Haarnoja","year":"2018"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1287\/moor.2015.0728"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1287\/opre.1080.0531"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.1080.0901"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1287\/ijoc.2022.1214"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2015.1356"},{"key":"ref19","doi-asserted-by":"crossref","first-page":"2712","DOI":"10.1109\/WSC57314.2022.10015456","article-title":"Quantile-Based Policy Optimization for Reinforcement Learning","volume-title":"2022 Winter Simulation Conference (WSC)","author":"Jiang","year":"2022"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4899-2696-8"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/s10626-017-0247-8"},{"issue":"1","key":"ref22","first-page":"1334","article-title":"End-to-End Training of Deep Visuomotor Policies","volume":"17","author":"Levine","year":"2016","journal-title":"Journal of Machine Learning Research"},{"key":"ref23","article-title":"Continuous Control with Deep Reinforcement Learning","author":"Lillicrap","year":"2015","journal-title":"arXiv preprint"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1002\/nav.20358"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-48914-3_2"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref27","first-page":"805","article-title":"An Approximate Solution Method for Large Risk-Averse Markov Decision Processes","volume-title":"Proceedings of the Twenty-Eighth Conference on Uncertainty in Artificial Intelligence","author":"Petrik","year":"2012"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1561\/2200000091"},{"key":"ref29","article-title":"Actor-Critic Algorithms for Risk-Sensitive MDPs","volume-title":"Advances in Neural Information Processing Systems","volume":"26","author":"Prashanth","year":"2013"},{"key":"ref30","first-page":"1406","article-title":"Cumulative Prospect Theory Meets Reinforcement Learning: Prediction and Control","volume-title":"Proceedings of the 33rd International Conference on Machine Learning","volume":"48","author":"Prashanth","year":"2016"},{"key":"ref31","article-title":"Evolution Strategies as a Scalable Alternative to Reinforcement Learning","author":"Salimans","year":"2017","journal-title":"arXiv preprint"},{"key":"ref32","first-page":"1889","article-title":"Trust Region Policy Optimization","volume-title":"Proceedings of the 32nd International Conference on Machine Learning","author":"Schulman","year":"2015"},{"key":"ref33","article-title":"High-Dimensional Continuous Control Using Generalized Advantage Estimation","author":"Schulman","year":"2015","journal-title":"arXiv preprint"},{"key":"ref34","article-title":"Proximal Policy Optimization Algorithms","author":"Schulman","year":"2017","journal-title":"arXiv preprint"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.13140\/RG.2.2.18893.74727"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/9.119632"},{"key":"ref37","article-title":"Policy Gradients Beyond Expectations: Conditional Value-at-Risk","author":"Tamar","year":"2014","journal-title":"arXiv preprint"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/BF00122574"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1201\/b14876"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.2307\/253675"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1023\/A:1022672621406"}],"event":{"name":"2024 Winter Simulation Conference (WSC)","location":"Orlando, FL, USA","start":{"date-parts":[[2024,12,15]]},"end":{"date-parts":[[2024,12,18]]}},"container-title":["2024 Winter Simulation Conference (WSC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10838618\/10838619\/10838808.pdf?arnumber=10838808","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T06:56:29Z","timestamp":1737442589000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10838808\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,15]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/wsc63780.2024.10838808","relation":{},"subject":[],"published":{"date-parts":[[2024,12,15]]}}}