{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,5]],"date-time":"2024-09-05T19:44:17Z","timestamp":1725565457366},"publisher-location":"Berlin, Heidelberg","reference-count":12,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540233398"},{"type":"electronic","value":"9783540278351"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2004]]},"DOI":"10.1007\/978-3-540-27835-1_7","type":"book-chapter","created":{"date-parts":[[2010,9,15]],"date-time":"2010-09-15T17:07:42Z","timestamp":1284570462000},"page":"80-94","source":"Crossref","is-referenced-by-count":1,"title":["Biologically Inspired Reinforcement Learning: Reward-Based Decomposition for Multi-goal Environments"],"prefix":"10.1007","author":[{"given":"Weidong","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Richard","family":"Coggins","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"7_CR1","doi-asserted-by":"crossref","first-page":"227","DOI":"10.1613\/jair.639","volume":"13","author":"T.G. Dietterich","year":"2000","unstructured":"Dietterich, T.G.: Hierarchical reinforcement learning with the MAXQ value function decomposition. Journal of Artificial Intelligence Research\u00a013, 227\u2013303 (2000)","journal-title":"Journal of Artificial Intelligence Research"},{"issue":"1","key":"7_CR2","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"R.S. Sutton","year":"1999","unstructured":"Sutton, R.S., Precup, D., Singh, S.: Between MDPs and semi-MDPs: A framework for temporal abstraction in reinforcement learning. Artificial Intelligence\u00a0112(1), 181\u2013211 (1999)","journal-title":"Artificial Intelligence"},{"key":"7_CR3","unstructured":"Hengst, B.: Discovering hierarchy in reinforcement learning with HEXQ. In: Hoffmann, C.S., Achim (eds.) Nineteenth International Conference on Machine Learning, Sydney Australia, pp. 243\u2013250 (2002)"},{"key":"7_CR4","first-page":"1082","volume-title":"Advances in Neural Information Processing Systems","author":"C.R. Shelton","year":"2001","unstructured":"Shelton, C.R.: Balancing multiple sources of reward in reinforcement learning. In: Advances in Neural Information Processing Systems, vol.\u00a013, pp. 1082\u20131088. MIT Press, Cambridge (2001)"},{"key":"7_CR5","volume-title":"The brain and emotion","author":"E.T. Rolls","year":"1999","unstructured":"Rolls, E.T.: The brain and emotion. Oxford University Press, Oxford (1999)"},{"key":"7_CR6","doi-asserted-by":"publisher","first-page":"272","DOI":"10.1093\/cercor\/10.3.272","volume":"10","author":"W.L. Schultz","year":"2000","unstructured":"Schultz, W.L., Tremblay, L., Hollerman, J.R.: Reward processing in primate orbitofrontal cortex and basla ganglia. Cerebral Cortex\u00a010, 272\u2013283 (2000)","journal-title":"Cerebral Cortex"},{"key":"7_CR7","doi-asserted-by":"crossref","unstructured":"Zhou, W., Coggins, R.: A biologically inspired hierarchical reinforcement learning system. Cybernetics and Systems (2004) (to appear)","DOI":"10.1080\/01969720590887270"},{"key":"7_CR8","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1016\/0959-4388(92)90011-9","volume":"2","author":"J.E. LeDoux","year":"1992","unstructured":"LeDoux, J.E.: Brain mechanisms of emotion and emotional learning. Current Opinion in Neurobiology\u00a02, 191\u2013197 (1992)","journal-title":"Current Opinion in Neurobiology"},{"key":"7_CR9","first-page":"215","volume-title":"Models of information processing in the basal ganglia","author":"G. Barto","year":"1995","unstructured":"Barto, G.: Adaptive critics and the basal ganglia. In: Davis, J.L., Houk Beiser, J.C. (eds.) Models of information processing in the basal ganglia, pp. 215\u2013232. MIT Press, Cambridge (1995)"},{"key":"7_CR10","first-page":"9","volume":"3","author":"R.S. Sutton","year":"1988","unstructured":"Sutton, R.S.: Learning to predict by the methods of temporal differences. Machine Learning\u00a03, 9\u201344 (1988)","journal-title":"Machine Learning"},{"key":"7_CR11","first-page":"393","volume-title":"Advances in Neural Information Processing Systems","author":"S.J. Bradtke","year":"1995","unstructured":"Bradtke, S.J., Duff, M.O.: Reinforcement learning methods for continuous-time markov decision problems. In: Advances in Neural Information Processing Systems, vol.\u00a07, pp. 393\u2013500. MIT Press, Cambridge (1995)"},{"issue":"5","key":"7_CR12","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1080\/019697201750257766","volume":"32","author":"J. Gadanho","year":"2001","unstructured":"Gadanho, J., Hallam, S.: Emotion-triggered learning in autonomous robot control. Cybernetics and Systems\u00a032(5), 531\u2013559 (2001)","journal-title":"Cybernetics and Systems"}],"container-title":["Lecture Notes in Computer Science","Biologically Inspired Approaches to Advanced Information Technology"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-27835-1_7.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,2]],"date-time":"2021-05-02T23:30:50Z","timestamp":1619998250000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-27835-1_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2004]]},"ISBN":["9783540233398","9783540278351"],"references-count":12,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-27835-1_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2004]]}}}