{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T23:08:33Z","timestamp":1779318513080,"version":"3.51.4"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032214768","type":"print"},{"value":"9783032214775","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21477-5_12","type":"book-chapter","created":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T22:47:19Z","timestamp":1779317239000},"page":"175-190","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Invariance to\u00a0Quantile Selection in\u00a0Distributional Continuous Control"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-6809-1409","authenticated-orcid":false,"given":"Felix","family":"Gr\u00fcn","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1774-7330","authenticated-orcid":false,"given":"Muhammad","family":"Saif-ur-Rehman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1886-1696","authenticated-orcid":false,"given":"Tobias","family":"Glasmachers","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9876-4396","authenticated-orcid":false,"given":"Ioannis","family":"Iossifidis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,5,1]]},"reference":[{"key":"12_CR1","unstructured":"Agarwal, R., Schwarzer, M., Castro, P.S., Courville, A.C., Bellemare, M.: Deep reinforcement learning at the edge of the statistical precipice. In: Advances in Neural Information Processing Systems. vol.\u00a034, pp. 29304\u201329320. Curran Associates, Inc. (2021). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2021\/hash\/f514cec81cb148559cf475e7426eed5e-Abstract.html"},{"key":"12_CR2","doi-asserted-by":"crossref","unstructured":"Akiba, T., Sano, S., Yanase, T., Ohta, T., Koyama, M.: OptUNA: a next-generation hyperparameter optimization framework. In: Proceedings of the 25rd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining (2019)","DOI":"10.1145\/3292500.3330701"},{"key":"12_CR3","doi-asserted-by":"publisher","unstructured":"Barth-Maron, G., et al.: Distributed Distributional Deterministic Policy Gradients (2018). https:\/\/doi.org\/10.48550\/arXiv.1804.08617","DOI":"10.48550\/arXiv.1804.08617"},{"key":"12_CR4","unstructured":"Bellemare, M.G., Dabney, W., Munos, R.: A distributional perspective on reinforcement learning. In: Proceedings of the 34th International Conference on Machine Learning, pp. 449\u2013458. PMLR (2017), https:\/\/proceedings.mlr.press\/v70\/bellemare17a.html, iSSN: 2640-3498"},{"key":"12_CR5","doi-asserted-by":"publisher","unstructured":"Cao, J., et al.: Gamma and VEGA hedging using deep distributional reinforcement learning. Front. Artif. Intell. 6 (2023).https:\/\/doi.org\/10.3389\/frai.2023.1129370, publisher: Frontiers","DOI":"10.3389\/frai.2023.1129370"},{"key":"12_CR6","unstructured":"Coumans, E., Bai, Y.: PyBullet, a Python module for physics simulation for games, robotics and machine learning (2016). http:\/\/pybullet.org"},{"key":"12_CR7","unstructured":"Dabney, W., Ostrovski, G., Silver, D., Munos, R.: Implicit quantile networks for distributional reinforcement learning. In: Proceedings of the 35th International Conference on Machine Learning, pp. 1096\u20131105. PMLR (2018). https:\/\/proceedings.mlr.press\/v80\/dabney18a.html, iSSN: 2640-3498"},{"key":"12_CR8","doi-asserted-by":"publisher","unstructured":"Dabney, W., Rowland, M., Bellemare, M., Munos, R.: Distributional reinforcement learning with quantile regression. In: Proceedings of the AAAI Conference on Artificial Intelligence 32(1) (2018). https:\/\/doi.org\/10.1609\/aaai.v32i1.11791","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"12_CR9","doi-asserted-by":"publisher","unstructured":"Duan, J., Guan, Y., Li, S.E., Ren, Y., Cheng, B.: Distributional soft actor-critic: off-policy reinforcement learning for addressing value estimation errors. IEEE Trans. Neural Networks Learn. Syst. 1\u201315 (2021). https:\/\/doi.org\/10.1109\/TNNLS.2021.3082568","DOI":"10.1109\/TNNLS.2021.3082568"},{"key":"12_CR10","unstructured":"Fujimoto, S., Hoof, H., Meger, D.: Addressing function approximation error in actor-critic methods. In: Proceedings of the 35th International Conference on Machine Learning, pp. 1587\u20131596. PMLR (2018). https:\/\/proceedings.mlr.press\/v80\/fujimoto18a.html, iSSN: 2640-3498"},{"key":"12_CR11","unstructured":"Haarnoja, T., et al.: Soft actor-critic algorithms and applications. arXiv:1812.05905 [cs, stat] (2019). http:\/\/arxiv.org\/abs\/1812.05905"},{"key":"12_CR12","doi-asserted-by":"publisher","unstructured":"Hessel, M., et al.: Rainbow: combining improvements in deep reinforcement learning. In: Proceedings of the AAAI Conference on Artificial Intelligence 32(1) (2018). https:\/\/doi.org\/10.1609\/aaai.v32i1.11796, number: 1","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"12_CR13","doi-asserted-by":"publisher","unstructured":"Huber, P.J.: Robust estimation of a location parameter. Ann. Math. Stat. 492\u2013518 (1964). https:\/\/doi.org\/10.1007\/978-1-4612-4380-9_35, book Title: Breakthroughs in Statistics ISBN: 9780387940397 9781461243809 Place: New York, NY Publisher: Springer New York","DOI":"10.1007\/978-1-4612-4380-9_35"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Koenker, R., Bassett\u00a0Jr, G.: Regression quantiles. Econometrica: journal of the Econometric Society, pp. 33\u201350 (1978)","DOI":"10.2307\/1913643"},{"key":"12_CR15","unstructured":"Kuznetsov, A., Shvechikov, P., Grishin, A., Vetrov, D.: Controlling overestimation bias with truncated mixture of continuous distributional quantile critics. In: Proceedings of the 37th International Conference on Machine Learning, pp. 5556\u20135566. PMLR (2020). https:\/\/proceedings.mlr.press\/v119\/kuznetsov20a.html"},{"key":"12_CR16","unstructured":"Li, S., Bing, S., Yang, S.: Distributional advantage actor-critic. arXiv:1806.06914 [cs, stat] (2018). http:\/\/arxiv.org\/abs\/1806.06914"},{"key":"12_CR17","unstructured":"Lillicrap, T.P., et al.: Continuous control with deep reinforcement learning. arXiv:1509.02971 [cs, stat] (2019). http:\/\/arxiv.org\/abs\/1509.02971"},{"key":"12_CR18","doi-asserted-by":"publisher","unstructured":"Lyle, C., Bellemare, M.G., Castro, P.S.: A comparative analysis of expected and distributional reinforcement learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 33, pp. 4504\u20134511 (2019). https:\/\/doi.org\/10.1609\/aaai.v33i01.33014504","DOI":"10.1609\/aaai.v33i01.33014504"},{"key":"12_CR19","unstructured":"Ma, X., Xia, L., Zhou, Z., Yang, J., Zhao, Q.: DSAC: distributional soft actor critic for risk-sensitive reinforcement learning. arXiv:2004.14547 [cs] (2020). http:\/\/arxiv.org\/abs\/2004.14547"},{"key":"12_CR20","doi-asserted-by":"publisher","unstructured":"Malekzadeh, P., Plataniotis, K.N., Poulos, Z., Wang, Z.: A robust quantile huber loss with interpretable parameter adjustment in distributional reinforcement learning (2024).https:\/\/doi.org\/10.48550\/arXiv.2401.02325, arXiv:2401.02325 [cs]","DOI":"10.48550\/arXiv.2401.02325"},{"issue":"7540","key":"12_CR21","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015). https:\/\/doi.org\/10.1038\/nature14236","journal-title":"Nature"},{"key":"12_CR22","unstructured":"Nam, D.W., Kim, Y., Park, C.Y.: GMAC: a distributional perspective on actor-critic framework. In: Proceedings of the 38th International Conference on Machine Learning, pp. 7927\u20137936. PMLR (2021). https:\/\/proceedings.mlr.press\/v139\/nam21a.html, iSSN: 2640-3498"},{"key":"12_CR23","doi-asserted-by":"publisher","unstructured":"Nguyen-Tang, T., Gupta, S., Venkatesh, S.: Distributional reinforcement learning via moment matching. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, no. 10, pp. 9144\u20139152 (2021). https:\/\/doi.org\/10.1609\/aaai.v35i10.17104","DOI":"10.1609\/aaai.v35i10.17104"},{"key":"12_CR24","unstructured":"Raffin, A., Hill, A., Gleave, A., Kanervisto, A., Ernestus, M., Dormann, N.: Stable-Baselines3: reliable reinforcement learning implementations. J. Mach. Learn. Res. 22(268), 1\u20138 (2021). http:\/\/jmlr.org\/papers\/v22\/20-1364.html"},{"key":"12_CR25","unstructured":"Silver, D., Lever, G., Heess, N., Degris, T., Wierstra, D., Riedmiller, M.: Deterministic policy gradient algorithms. In: Proceedings of the 31st International Conference on Machine Learning, pp. 387\u2013395. PMLR (2014). https:\/\/proceedings.mlr.press\/v32\/silver14.html"},{"key":"12_CR26","unstructured":"Sun, K., et al.: Interpreting distributional reinforcement learning: regularization and optimization perspectives (2022). http:\/\/arxiv.org\/abs\/2110.03155"},{"key":"12_CR27","doi-asserted-by":"publisher","unstructured":"Todorov, E., Erez, T., Tassa, Y.: MuJoCo: a physics engine for model-based control. In: 2012 IEEE\/RSJ International Conference on Intelligent Robots and Systems, pp. 5026\u20135033. IEEE (2012).https:\/\/doi.org\/10.1109\/IROS.2012.6386109","DOI":"10.1109\/IROS.2012.6386109"},{"key":"12_CR28","doi-asserted-by":"publisher","unstructured":"Witten, I.H.: An adaptive optimal controller for discrete-time Markov environments. Inf. Control 34(4), 286\u2013295 (1977). https:\/\/doi.org\/10.1016\/S0019-9958(77)90354-0","DOI":"10.1016\/S0019-9958(77)90354-0"},{"key":"12_CR29","unstructured":"Yang, D., Zhao, L., Lin, Z., Qin, T., Bian, J., Liu, T.Y.: Fully parameterized quantile function for distributional reinforcement learning. In: Advances in Neural Information Processing Systems, vol.\u00a032. Curran Associates, Inc. (2019). https:\/\/proceedings.neurips.cc\/paper\/2019\/hash\/f471223d1a1614b58a7dc45c9d01df19-Abstract.html"}],"container-title":["Lecture Notes in Computer Science","Machine Learning, Optimization, and Data Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21477-5_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T22:47:24Z","timestamp":1779317244000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21477-5_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032214768","9783032214775"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21477-5_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors declare that the research was conducted without a potential conflict of interest.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"LOD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Artificial Intelligence Symposium","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Castiglione della Pescaia","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mod2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/lod2025.icas.events","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}