{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2022,4,2]],"date-time":"2022-04-02T05:34:44Z","timestamp":1648877684895},"reference-count":17,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2012,11,6]],"date-time":"2012-11-06T00:00:00Z","timestamp":1352160000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Artif Life Robotics"],"published-print":{"date-parts":[[2012,12]]},"DOI":"10.1007\/s10015-012-0058-9","type":"journal-article","created":{"date-parts":[[2012,11,5]],"date-time":"2012-11-05T13:47:36Z","timestamp":1352123256000},"page":"293-299","source":"Crossref","is-referenced-by-count":0,"title":["Reinforcement learning approach to multi-stage decision making problems with changes in action sets"],"prefix":"10.1007","volume":"17","author":[{"given":"Takuya","family":"Etoh","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hirotaka","family":"Takano","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junichi","family":"Murata","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2012,11,6]]},"reference":[{"issue":"4","key":"58_CR1","doi-asserted-by":"crossref","first-page":"B-141","DOI":"10.1287\/mnsc.17.4.B141","volume":"17","author":"RE Bellman","year":"1970","unstructured":"Bellman RE, Zadeh LA (1970) Decision-making in a fuzzy environment. Manag Sci 17(4):B-141\u2013B-164","journal-title":"Manag Sci"},{"key":"58_CR2","unstructured":"Bertsekas DP (2007) Dynamic programming and optimal control, vol 1. Athena Scientigic, Belmont"},{"issue":"5","key":"58_CR3","doi-asserted-by":"crossref","first-page":"317","DOI":"10.1287\/mnsc.12.5.317","volume":"12","author":"RA Howard","year":"1966","unstructured":"Howard RA (1966) Dynamic programming. Manag Sci 12(5):317\u2013348","journal-title":"Manag Sci"},{"key":"58_CR4","doi-asserted-by":"crossref","unstructured":"Wang F-Y, Zhang H, Liu D (2009) Adaptive dynamic programming: an introduction, IEEE Comput Intell Mag 39\u201347","DOI":"10.1109\/MCI.2009.932261"},{"key":"58_CR5","doi-asserted-by":"crossref","unstructured":"Si J, Barto AG, Powell WB, Wunsch D (2004) Handbook of Learning and Approximate Dynamic Programming Wiley-IEEE Press, New York","DOI":"10.1109\/9780470544785"},{"key":"58_CR6","unstructured":"Momoh JA, Zhang Y (2005) Unit Commitment Using Adaptive Dynamic Programming"},{"key":"58_CR7","first-page":"343","volume":"13","author":"Andrew G Barto","year":"2003","unstructured":"Barto Andrew G, Mahadevan Sridhar (2003) Recent advances in hierarchical reinforcement learning. Discrete Event Dyn Syst Theory Appl 13:343\u2013379","journal-title":"Discrete Event Dyn Syst Theory Appl"},{"key":"58_CR8","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1177\/1059712308100236","volume":"17","author":"K Merrick","year":"2009","unstructured":"Merrick K, Maher ML (2009) Motivated learning from interesting events: adaptive, multitask learning agents for complex environments. Int Soc Adapt Behav 17:7\u201327","journal-title":"Int Soc Adapt Behav"},{"key":"58_CR9","doi-asserted-by":"crossref","unstructured":"Bedford T, Cooke R (2001) Probabilistic risk analysis: foundations and methods, Cambridge University Press, Cambridge","DOI":"10.1017\/CBO9780511813597"},{"issue":"1","key":"58_CR10","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1111\/j.1539-6924.1981.tb01350.x","volume":"1","author":"S Kaplan","year":"1981","unstructured":"Kaplan S, Garrick J (1981) On the quantitative definition of risk. Risk Anal 1(1):11\u201327","journal-title":"Risk Anal"},{"issue":"2","key":"58_CR11","doi-asserted-by":"crossref","first-page":"263","DOI":"10.2307\/1914185","volume":"47","author":"D Kahneman","year":"1979","unstructured":"Kahneman D, Tversky A (1979) An analysis of decision under risk. Econometrica 47(2):263\u2013292","journal-title":"Econometrica"},{"issue":"2","key":"58_CR12","doi-asserted-by":"crossref","first-page":"371","DOI":"10.1093\/rfs\/14.2.371","volume":"14","author":"S Basak","year":"2001","unstructured":"Basak S, Shapiro A (2001) Value-at-risk-based risk management: optimal policies and asset prices. Rev Financ Stud Summer 14(2):371\u2013405","journal-title":"Rev Financ Stud Summer"},{"key":"58_CR13","doi-asserted-by":"crossref","first-page":"81","DOI":"10.1613\/jair.1666","volume":"24","author":"P Geibel","year":"2005","unstructured":"Geibel P, Wysotzki F (2005) Risk-sensitive reinforcement learning applied to control under constraints. J Artif Intell Res 24:81\u2013108","journal-title":"J Artif Intell Res"},{"key":"58_CR14","first-page":"244","volume":"2000","author":"M Sato","year":"2000","unstructured":"Sato M, Kobayashi S (2000) Variance-penalized reinforcement learning for risk-averse asset allocation. Proc IDEAL 2000:244\u2013249","journal-title":"Proc IDEAL"},{"key":"58_CR15","unstructured":"Shibuya T (2010) A study on reinforcement learning in unstationary dynamic environments. Proc SSI 2010 3B1\u20133B2 (in Japanese)"},{"key":"58_CR16","doi-asserted-by":"crossref","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning\u2014an introduction, The MIT Press, Cambridge","DOI":"10.1109\/TNN.1998.712192"},{"key":"58_CR17","volume-title":"Dynamic programming and markov processes","author":"RA Howard","year":"1960","unstructured":"Howard RA (1960) Dynamic programming and markov processes. The MIT Press, Cambridge"}],"container-title":["Artificial Life and Robotics"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10015-012-0058-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10015-012-0058-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10015-012-0058-9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,7,5]],"date-time":"2019-07-05T03:52:08Z","timestamp":1562298728000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10015-012-0058-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,11,6]]},"references-count":17,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2012,12]]}},"alternative-id":["58"],"URL":"https:\/\/doi.org\/10.1007\/s10015-012-0058-9","relation":{},"ISSN":["1433-5298","1614-7456"],"issn-type":[{"value":"1433-5298","type":"print"},{"value":"1614-7456","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012,11,6]]}}}