{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,28]],"date-time":"2025-10-28T05:56:18Z","timestamp":1761630978402,"version":"3.40.3"},"publisher-location":"Cham","reference-count":44,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030946616"},{"type":"electronic","value":"9783030946623"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-030-94662-3_8","type":"book-chapter","created":{"date-parts":[[2022,1,11]],"date-time":"2022-01-11T12:03:08Z","timestamp":1641902588000},"page":"107-128","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Safe Distributional Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Jianyi","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Paul","family":"Weng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,1,11]]},"reference":[{"key":"8_CR1","unstructured":"Achiam, J., Held, D., Tamar, A., Abbeel, P.: Constrained policy optimization. In: International Conference on Machine Learning (2017)"},{"key":"8_CR2","doi-asserted-by":"crossref","unstructured":"Alshiekh, M., Bloem, R., Ehlers, R., K\u00f6nighofer, B., Niekum, S., Topcu, U.: Safe reinforcement learning via shielding. In: AAAI (2018)","DOI":"10.1609\/aaai.v32i1.11797"},{"key":"8_CR3","volume-title":"Constrained Markov Decision Processes","author":"E Altman","year":"1999","unstructured":"Altman, E.: Constrained Markov Decision Processes. CRC Press, Boca Raton (1999)"},{"key":"8_CR4","unstructured":"Barth-Maron, G., et al.: Distributed distributional deterministic policy gradients. In: International Conference on Learning Representations (2018)"},{"key":"8_CR5","unstructured":"Bellemare, M.G., Dabney, W., Munos, R.: A distributional perspective on reinforcement learning, pp. 449\u2013458 (2017)"},{"key":"8_CR6","unstructured":"Berkenkamp, F., Turchetta, M., Schoellig, A.P., Krause, A.: Safe model-based reinforcement learning with stability guarantees. In: NeurIPS (2017)"},{"key":"8_CR7","series-title":"Smart Innovation, Systems and Technologies","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1007\/978-3-319-04129-2_20","volume-title":"Recent Advances of Neural Network Models and Applications","author":"F Bertoluzzo","year":"2014","unstructured":"Bertoluzzo, F., Corazza, M.: Reinforcement learning for automated financial trading: basics and applications. In: Bassis, S., Esposito, A., Morabito, F.C. (eds.) Recent Advances of Neural Network Models and Applications. SIST, vol. 26, pp. 197\u2013213. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-04129-2_20"},{"issue":"9","key":"8_CR8","doi-asserted-by":"publisher","first-page":"2574","DOI":"10.1109\/TAC.2014.2309262","volume":"59","author":"V Borkar","year":"2014","unstructured":"Borkar, V., Jain, R.: Risk-constrained Markov decision processes. IEEE Trans. Autom. Control 59(9), 2574\u20132579 (2014)","journal-title":"IEEE Trans. Autom. Control"},{"key":"8_CR9","unstructured":"Borkar, V.S.: Learning algorithms for risk-sensitive control. In: International Symposium on Mathematical Theory of Networks and Systems (2010)"},{"key":"8_CR10","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441","volume-title":"Convex Optimization","author":"S Boyd","year":"2004","unstructured":"Boyd, S., Boyd, S.P., Vandenberghe, L.: Convex Optimization. Cambridge University Press, New York (2004)"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Br\u00e1zdil, T., Chatterjee, K., Novotn\u1ef3, P., Vahala, J.: Reinforcement learning of risk-constrained policies in Markov decision processes. In: AAAI (2020)","DOI":"10.1609\/aaai.v34i06.6531"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Cheng, R., Orosz, G., Murray, R.M., Burdick, J.W.: End-to-end safe reinforcement learning through barrier functions for safety-critical continuous control tasks. In: AAAI (2019)","DOI":"10.1609\/aaai.v33i01.33013387"},{"key":"8_CR13","unstructured":"Chow, Y., Ghavamzadeh, M.: Algorithms for CVaR optimization in MDPs. In: NeurIPS (2014)"},{"issue":"1","key":"8_CR14","first-page":"1","volume":"18","author":"Y Chow","year":"2017","unstructured":"Chow, Y., Ghavamzadeh, M., Janson, L., Pavone, M.: Risk-constrained reinforcement learning with percentile risk criteria. J. Mach. Learn. Res. 18(1), 1\u201351 (2017)","journal-title":"J. Mach. Learn. Res."},{"key":"8_CR15","unstructured":"Chow, Y., Tamar, A., Mannor, S., Pavone, M.: Risk-sensitive and robust decision-making: a CVaR optimization approach. In: NeurIPS (2015)"},{"key":"8_CR16","unstructured":"Dabney, W., Ostrovski, G., Silver, D., Munos, R.: Implicit quantile networks for distributional reinforcement learning, pp. 1096\u20131105 (2018)"},{"key":"8_CR17","unstructured":"Dalal, G., Dvijotham, K., Vecerik, M., Hester, T., Paduraru, C., Tassa, Y.: Safe exploration in continuous action spaces. ArXiv (2018)"},{"key":"8_CR18","unstructured":"Fulton, N., Platzer, A.: Safe reinforcement learning via formal methods. In: AAAI Conference on Artificial Intelligence (2018)"},{"key":"8_CR19","first-page":"1437","volume":"16","author":"J Garc\u0131a","year":"2015","unstructured":"Garc\u0131a, J., Fern\u00e1ndez, F.: A comprehensive survey on safe reinforcement learning. J. Mach. Learn. Res. 16, 1437\u20131480 (2015)","journal-title":"J. Mach. Learn. Res."},{"key":"8_CR20","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1613\/jair.1666","volume":"24","author":"P Geibel","year":"2005","unstructured":"Geibel, P., Wysotzki, F.: Risk-sensitive reinforcement learning applied to control under constraints. J. Artif. Intell. Res. 24, 81\u2013108 (2005)","journal-title":"J. Artif. Intell. Res."},{"key":"8_CR21","doi-asserted-by":"crossref","unstructured":"Isele, D., Nakhaei, A., Fujimura, K.: Safe reinforcement learning on autonomous vehicles. In: IROS, pp. 1\u20136. IEEE (2018)","DOI":"10.1109\/IROS.2018.8593420"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Jia, Y., Burden, J., Lawton, T., Habli, I.: Safe reinforcement learning for sepsis treatment. In: 2020 IEEE International Conference on Healthcare Informatics (ICHI), pp. 1\u20137. IEEE (2020)","DOI":"10.1109\/ICHI48887.2020.9374367"},{"issue":"4","key":"8_CR23","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1257\/jep.15.4.143","volume":"15","author":"R Koenker","year":"2001","unstructured":"Koenker, R., Hallock, K.F.: Quantile regression. J. Econ. Perspect. 15(4), 143\u2013156 (2001)","journal-title":"J. Econ. Perspect."},{"key":"8_CR24","unstructured":"Lim, S.H., Malik, I.: Distributional reinforcement learning for risk-sensitive policies (2021). https:\/\/openreview.net\/forum?id=19drPzGV691"},{"key":"8_CR25","doi-asserted-by":"crossref","unstructured":"Liu, Y., Ding, J., Liu, X.: IPO: interior-point policy optimization under constraints. In: AAAI Conference on Artificial Intelligence (2020)","DOI":"10.1609\/aaai.v34i04.5932"},{"key":"8_CR26","unstructured":"Miryoosefi, S., Brantley, K., Daum\u00e9 III, H., Dud\u00edk, M., Schapire, R.: Reinforcement learning with convex constraints. In: NeurIPS (2019)"},{"key":"8_CR27","doi-asserted-by":"crossref","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015)","DOI":"10.1038\/nature14236"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Pilarski, P.M., Dawson, M.R., Degris, T., Fahimi, F., Carey, J.P., Sutton, R.S.: Online human training of a myoelectric prosthesis controller via actor-critic reinforcement learning. In: 2011 IEEE International Conference on Rehabilitation Robotics, pp. 1\u20137. IEEE (2011)","DOI":"10.1109\/ICORR.2011.5975338"},{"key":"8_CR29","unstructured":"Pirotta, M., Restelli, M., Pecorino, A., Calandriello, D.: Safe policy iteration. In: International Conference on Machine Learning (2013)"},{"issue":"3","key":"8_CR30","doi-asserted-by":"publisher","first-page":"367","DOI":"10.1007\/s10994-016-5569-5","volume":"105","author":"LA Prashanth","year":"2016","unstructured":"Prashanth, L.A., Ghavamzadeh, M.: Variance-constrained actor-critic algorithms for discounted and average reward MDPs. Mach. Learn. 105(3), 367\u2013417 (2016). https:\/\/doi.org\/10.1007\/s10994-016-5569-5","journal-title":"Mach. Learn."},{"key":"8_CR31","unstructured":"Ray, A., Achiam, J., Amodei, D.: Benchmarking Safe Exploration in Deep Reinforcement Learning (2019)"},{"key":"8_CR32","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., Moritz, P.: Trust region policy optimization. In: ICML, pp. 1889\u20131897. PMLR (2015)"},{"key":"8_CR33","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. CoRR (2017). http:\/\/arxiv.org\/abs\/1707.06347"},{"issue":"7676","key":"8_CR34","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., et al.: Mastering the game of go without human knowledge. Nature 550(7676), 354\u2013359 (2017)","journal-title":"Nature"},{"key":"8_CR35","unstructured":"Singh, R., Zhang, Q., Chen, Y.: Improving robustness via risk averse distributional reinforcement learning. In: Learning for Dynamics and Control, pp. 958\u2013968. PMLR (2020)"},{"key":"8_CR36","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (2018)"},{"key":"8_CR37","unstructured":"Sutton, R.S., McAllester, D.A., Singh, S.P., Mansour, Y., et al.: Policy gradient methods for reinforcement learning with function approximation. In: NeurIPS (2000)"},{"key":"8_CR38","unstructured":"Tessler, C., Mankowitz, D.J., Mannor, S.: Reward constrained policy optimization. In: International Conference on Learning Representations (2019)"},{"key":"8_CR39","unstructured":"Turchetta, M., Berkenkamp, F., Krause, A.: Safe exploration in finite Markov decision processes with gaussian processes. In: NeurIPS (2016)"},{"key":"8_CR40","doi-asserted-by":"crossref","unstructured":"Wachi, A., Sui, Y., Yue, Y., Ono, M.: Safe exploration and optimization of constrained MDPs using gaussian processes. In: AAAI (2018)","DOI":"10.1609\/aaai.v32i1.12103"},{"key":"8_CR41","unstructured":"Xie, T., et al.: A block coordinate ascent algorithm for mean-variance optimization. In: NeurIPS (2018)"},{"key":"8_CR42","unstructured":"Yang, D., Zhao, L., Lin, Z., Qin, T., Bian, J., Liu, T.: Fully parameterized quantile function for distributional reinforcement learning. In: NeurIPS (2019)"},{"key":"8_CR43","unstructured":"Yang, T.Y., Rosca, J., Narasimhan, K., Ramadge, P.J.: Projection-based constrained policy optimization. In: ICLR (2020)"},{"key":"8_CR44","unstructured":"Yu, M., Yang, Z., Kolar, M., Wang, Z.: Convergent policy optimization for safe reinforcement learning. In: NeurIPS (2019)"}],"container-title":["Lecture Notes in Computer Science","Distributed Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-94662-3_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,22]],"date-time":"2023-01-22T15:21:50Z","timestamp":1674400910000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-94662-3_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783030946616","9783030946623"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-94662-3_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"11 January 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Distributed Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Shanghai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 December 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 December 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dai22021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.adai.ai\/dai\/2021\/2021.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"31","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"15","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"48% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}