{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T01:14:10Z","timestamp":1782350050132,"version":"3.54.5"},"reference-count":25,"publisher":"Springer Science and Business Media LLC","issue":"2-3","license":[{"start":{"date-parts":[[2002,11,1]],"date-time":"2002-11-01T00:00:00Z","timestamp":1036108800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2002,11,1]],"date-time":"2002-11-01T00:00:00Z","timestamp":1036108800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Learning"],"published-print":{"date-parts":[[2002,11]]},"DOI":"10.1023\/a:1017940631555","type":"journal-article","created":{"date-parts":[[2002,12,30]],"date-time":"2002-12-30T09:36:44Z","timestamp":1041241004000},"page":"267-290","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":155,"title":["Risk-Sensitive Reinforcement Learning"],"prefix":"10.1007","volume":"49","author":[{"given":"Oliver","family":"Mihatsch","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ralph","family":"Neuneier","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","reference":[{"key":"395112_CR1","volume-title":"H ? -optimal control and related minimax design problems: A dynamic game approach","author":"T. S. Basar","year":"1995","unstructured":"Basar, T. S., & Bernhard, P. (1995). H\n?\n-optimal control and related minimax design problems: A dynamic game approach (2nd edn.). Boston: Birkh\u00e4user.","edition":"2nd edn."},{"key":"395112_CR2","doi-asserted-by":"crossref","DOI":"10.1515\/9781400874651","volume-title":"Applied dynamic programming","author":"R. E. Bellman","year":"1962","unstructured":"Bellman, R. E., & Dreyfus, S. E. (1962). Applied dynamic programming. Princeton: Princeton University Press."},{"key":"395112_CR3","volume-title":"Dynamic programming and optimal control (Vol. 2.)","author":"D. P. Bertsekas","year":"1995","unstructured":"Bertsekas, D. P. (1995). Dynamic programming and optimal control (Vol. 2.). Belmont, MA: Athena Scientific."},{"key":"395112_CR4","volume-title":"Neuro-dynamic programming","author":"D. P. Bertsekas","year":"1996","unstructured":"Bertsekas, D. P., & Tsitsiklis, J. N. (1996). Neuro-dynamic programming. Belmont, MA: Athena Scientific."},{"key":"395112_CR5","unstructured":"Coraluppi, S. (1997). Optimal control of Markov decision processes for performance and Robustness. Ph.D. Thesis, University of Maryland."},{"key":"395112_CR6","doi-asserted-by":"crossref","first-page":"21","DOI":"10.1007\/978-1-4612-1784-8_2","volume-title":"Stochastic analysis, control, optimization and applications","author":"S. P. Coraluppi","year":"1999","unstructured":"Coraluppi, S. P., & Marcus, S. I. (1999). Risk-sensitive, minimax and mixed risk-neutral\/minimax control of Markov decision processes. In W. M. McEarney, G. G. Yin, & Q. Zhang (Eds.), Stochastic analysis, control, optimization and applications (pp. 21\u201340). Boston: Birkh\u00e4user."},{"key":"395112_CR7","volume-title":"Modern portfolio theory and investment analysis","author":"E. J. Elton","year":"1995","unstructured":"Elton, E. J., & Gruber, M. J. (1995). Modern portfolio theory and investment analysis. New York: John Wiley & Sons."},{"key":"395112_CR8","first-page":"261","volume-title":"Machine Learning: Proceedings of the Twelfth International Conference","author":"G. J. Gordon","year":"1995","unstructured":"Gordon, G. J. (1995). Stable function approximation in dynamic programming. In A. Prieditis, & S. J. Russel (Eds.), Machine Learning: Proceedings of the Twelfth International Conference (pp. 261\u2013268). San Francisco: Morgan Kaufmann Publishers."},{"key":"395112_CR9","first-page":"105","volume-title":"Machine Learning: Proceedings of the Eleventh International Conference","author":"M. Heger","year":"1994","unstructured":"Heger, M. (1994a). Consideration of risk and reinforcement learning. In W. W. Cohen, & H. Hirsh (Eds.), Machine Learning: Proceedings of the Eleventh International Conference (pp. 105\u2013111). San Francisco: Morgan Kaufmann Publishers."},{"key":"395112_CR10","series-title":"Technical Report","volume-title":"Risk and reinforcement learning: Concepts and dynamic programming","author":"M. Heger","year":"1994","unstructured":"Heger, M. (1994b). Risk and reinforcement learning: Concepts and dynamic programming. Technical Report, Zentrum f\u00fcr Kognitionswissenschaften, Universit\u00e4t Bremen, Germany."},{"issue":"7","key":"395112_CR11","doi-asserted-by":"crossref","first-page":"356","DOI":"10.1287\/mnsc.18.7.356","volume":"18","author":"R. A. Howard","year":"1972","unstructured":"Howard, R. A., & Matheson, J. E. (1972). Risk-sensitive Markov decision processes. Management Science, 18:7, 356\u2013369.","journal-title":"Management Science"},{"key":"395112_CR12","doi-asserted-by":"crossref","unstructured":"Koenig, S., & Simmons, R. G. (1994). Risk-sensitive planning with probabilistic decision graphs. In Proceedings of the Fourth International Conference on Principles of Knowledge Representation and Reasoning (KR) (pp. 363-373).","DOI":"10.1016\/B978-1-4832-1452-8.50129-9"},{"key":"395112_CR13","first-page":"310","volume-title":"Machine Learning: Proceedings of the Thirteenth International Conference","author":"M. L. Littman","year":"1996","unstructured":"Littman, M. L., & Szepesvri, C. (1996). A generalized reinforcement-learning model: Convergence and applications. In L. Saitta (Ed.), Machine Learning: Proceedings of the Thirteenth International Conference (pp. 310\u2013318). San Francisco: Morgan Kaufman Publishers."},{"issue":"2","key":"395112_CR14","doi-asserted-by":"crossref","first-page":"197","DOI":"10.1109\/49.824797","volume":"18","author":"P. Marbach","year":"2000","unstructured":"Marbach, P., Mihatsch, O., & Tsitsiklis. J. N. (2000). Call admission control and routing in integrated services networks using neuro-dynamic programming. IEEE Journal on Selected Areas in Communications, 18:2, 197\u2013208.","journal-title":"IEEE Journal on Selected Areas in Communications"},{"key":"395112_CR15","volume-title":"Advances in neural information processing systems (Vol. 10)","author":"R. Neuneier","year":"1998","unstructured":"Neuneier, R. (1998). Enhancing Q-learning for optimal asset allocation. In M. I. Jordan, M. J. Kearns, & S. A. Solla (Eds.), Advances in neural information processing systems (Vol. 10). Cambridge, MA: The MIT Press."},{"key":"395112_CR16","unstructured":"Neuneier, R., & Mihatsch, O. (2000). Risk-averse asset allocation using reinforcement learning. In: Proceedings of the Seventh International Conference on Forecasting Financial Markets: Advances for Exchange Rates, Interest Rates and Asset Management."},{"key":"395112_CR17","doi-asserted-by":"crossref","first-page":"122","DOI":"10.2307\/1913738","volume":"32","author":"J. W. Pratt","year":"1964","unstructured":"Pratt, J. W. (1964). Risk aversion in the small and in the large. Econometrica, 32, 122\u2013136.","journal-title":"Econometrica"},{"key":"395112_CR18","doi-asserted-by":"crossref","DOI":"10.1002\/9780470316887","volume-title":"Markov decision processes","author":"M. L. Puterman","year":"1994","unstructured":"Puterman, M. L. (1994). Markov decision processes. New York: John Wiley & Sons."},{"key":"395112_CR19","first-page":"974","volume-title":"Advances in neural information processing systems","author":"S. Singh","year":"1997","unstructured":"Singh, S., & Bertsekas, D. (1997). Reinforcement learning for dynamic channel allocation in cellular telephone systems. In M. C. Mozer, M. I. Jordan, and T. Petsche (Eds.), Advances in neural information processing systems (Vol. 9, pp. 974\u2013980). Cambridge, MA: The MIT Press."},{"key":"395112_CR20","first-page":"9","volume":"3","author":"R. S. Sutton","year":"1988","unstructured":"Sutton, R. S. (1988). Learning to predict by the methods of temporal differences. Machine Learning, 3, 9\u201344.","journal-title":"Machine Learning"},{"issue":"5","key":"395112_CR21","doi-asserted-by":"crossref","first-page":"674","DOI":"10.1109\/9.580874","volume":"42","author":"J. N. Tsitsiklis","year":"1997","unstructured":"Tsitsiklis, J. N., & Van Roy, B. (1997). An analysis of temporal-difference learning with function approximation. IEEE Transactions on Automatic Control, 42:5, 674\u2013690.","journal-title":"IEEE Transactions on Automatic Control"},{"issue":"10","key":"395112_CR22","doi-asserted-by":"crossref","first-page":"1840","DOI":"10.1109\/9.793723","volume":"44","author":"J. N. Tsitsiklis","year":"1999","unstructured":"Tsitsiklis, J. N., & Van Roy, B. (1999). Optimal stopping of Markov processes: Hilbert space theory, approximation algorithms, and an application to pricing financial derivatives. IEEE Transactions on Automatic Control, 44:10, 1840\u20131851.","journal-title":"IEEE Transactions on Automatic Control"},{"key":"395112_CR23","unstructured":"von Neumann, J., & Morgenstern, O. (1953). Theory of games and economic behavior (3rd edn.). Princeton University Press."},{"key":"395112_CR24","volume-title":"Learning from delayed rewards","author":"C. J. C. H. Watkins","year":"1989","unstructured":"Watkins, C. J. C. H. (1989). Learning from delayed rewards. Ph.D. Thesis, University of Cambridge, England."},{"key":"395112_CR25","first-page":"1024","volume-title":"Advances in neural information processing systems","author":"W. Zhang","year":"1996","unstructured":"Zhang, W., & Dietterich, T. G. (1996). High-performance job-shop scheduling with a time-delay TD(?) network. In D. S. Touretzky, M. C. Mozer, & M. E. Hasselmo (Eds.), Advances in neural information processing systems (Vol. 8, pp. 1024\u20131030). Cambridge, MA: The MIT Press."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1023\/A:1017940631555.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1023\/A:1017940631555\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1023\/A:1017940631555.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,10]],"date-time":"2025-07-10T11:30:44Z","timestamp":1752147044000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1023\/A:1017940631555"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2002,11]]},"references-count":25,"journal-issue":{"issue":"2-3","published-print":{"date-parts":[[2002,11]]}},"alternative-id":["395112"],"URL":"https:\/\/doi.org\/10.1023\/a:1017940631555","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2002,11]]},"assertion":[{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}