{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T02:37:20Z","timestamp":1774924640685,"version":"3.50.1"},"publisher-location":"New York, NY","reference-count":72,"publisher":"Springer New York","isbn-type":[{"value":"9781461473206","type":"electronic"}],"license":[{"start":{"date-parts":[[2014,1,1]],"date-time":"2014-01-01T00:00:00Z","timestamp":1388534400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2014]]},"DOI":"10.1007\/978-1-4614-7320-6_674-1","type":"book-chapter","created":{"date-parts":[[2014,10,2]],"date-time":"2014-10-02T02:13:53Z","timestamp":1412216033000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Reward-Based Learning, Model-Based and Model-Free"],"prefix":"10.1007","author":[{"given":"Quentin J. M.","family":"Huys","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anthony","family":"Cruickshank","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peggy","family":"Seri\u00e8s","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,2,21]]},"reference":[{"issue":"3","key":"674-1_CR1","doi-asserted-by":"publisher","first-page":"590","DOI":"10.1037\/0735-7044.108.3.590","volume":"108","author":"B Balleine","year":"1994","unstructured":"Balleine B, Dickinson A (1994) Role of cholecystokinin in the motivational control of instrumental action in rats. Behav Neurosci 108(3):590\u2013605","journal-title":"Behav Neurosci"},{"issue":"5","key":"674-1_CR2","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TSMC.1983.6313077","volume":"13","author":"A Barto","year":"1983","unstructured":"Barto A, Sutton R, Anderson C (1983) Neuronlike elements that can solve difficult learning control problems. IEEE Trans Syst Man Cybern 13(5):834\u2013846","journal-title":"IEEE Trans Syst Man Cybern"},{"issue":"1","key":"674-1_CR3","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1016\/j.neuron.2005.05.020","volume":"47","author":"HM Bayer","year":"2005","unstructured":"Bayer HM, Glimcher PW (2005) Midbrain dopamine neurons encode a quantitative reward prediction error signal. Neuron 47(1):129\u2013141","journal-title":"Neuron"},{"issue":"3","key":"674-1_CR4","doi-asserted-by":"publisher","first-page":"1428","DOI":"10.1152\/jn.01140.2006","volume":"98","author":"HM Bayer","year":"2007","unstructured":"Bayer HM, Lau B, Glimcher PW (2007) Statistics of midbrain dopamine neuron spike trains in the awake primate. JNeurophysiol 98(3):1428\u20131439","journal-title":"JNeurophysiol"},{"key":"674-1_CR5","volume-title":"Dynamic programming","author":"RE Bellman","year":"1957","unstructured":"Bellman RE (1957) Dynamic programming. Princeton University Press, Princeton"},{"key":"674-1_CR6","volume-title":"Neuro-dynamic programming","author":"DP Bertsekas","year":"1996","unstructured":"Bertsekas DP, Tsitsiklis JN (1996) Neuro-dynamic programming. Athena Scientific, Belmon"},{"key":"674-1_CR7","unstructured":"Boutilier C, Dearden R, Goldszmidt M (1995) Exploiting structure in policy construction. In: Proceedings of the IJCAI Montreal, Quebec, Canada August 20\u201325,1995, vol 14, pp 1104\u20131113"},{"key":"674-1_CR8","volume-title":"Learning and behavior: a contemporary synthesis","author":"ME Bouton","year":"2006","unstructured":"Bouton ME (2006) Learning and behavior: a contemporary synthesis. Sinauer, Sunderland"},{"issue":"1\u20132","key":"674-1_CR9","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1016\/S0004-3702(01)00129-1","volume":"134","author":"M Campbell","year":"2002","unstructured":"Campbell M, Hoane A et al (2002) Deep blue. Artif Intell 134(1\u20132):57\u201383","journal-title":"Artif Intell"},{"issue":"4","key":"674-1_CR10","doi-asserted-by":"publisher","first-page":"553","DOI":"10.1037\/0735-7044.116.4.553","volume":"116","author":"RN Cardinal","year":"2002","unstructured":"Cardinal RN, Parkinson JA, Lachenal G, Halkerston KM, Rudarakanchana N, Hall J, Morrison CH, Howes SR, Robbins TW, Everitt BJ (2002) Effects of selective excitotoxic lesions of the nucleus accumbens core, anterior cingulate cortex, and central nucleus of the amygdala on autoshaping performance in rats. Behav Neurosci 116(4):553\u2013567","journal-title":"Behav Neurosci"},{"issue":"4","key":"674-1_CR11","doi-asserted-by":"publisher","first-page":"962","DOI":"10.1523\/JNEUROSCI.4507-04.2005","volume":"25","author":"LH Corbit","year":"2005","unstructured":"Corbit LH, Balleine BW (2005a) Double dissociation of basolateral and central amygdala lesions on the general and outcome-specific forms of Pavlovian-instrumental transfer. J Neurosci 25(4):962\u2013970","journal-title":"J Neurosci"},{"key":"674-1_CR12","unstructured":"Balleine BW, Corbit LH (2005b) Double dissociation of nucleus accumbens core and shell on the general and ouctome-specific forms of Pavlovian-instrumental transfer. Program No. 71.16. 2005 Neuroscience Meeting Planner. Washington, DC: Society for Neuroscience, 2005. Online"},{"issue":"5867","key":"674-1_CR13","doi-asserted-by":"publisher","first-page":"1264","DOI":"10.1126\/science.1150605","volume":"319","author":"K D\u2019Ardenne","year":"2008","unstructured":"D\u2019Ardenne K, McClure SM, Nystrom LE, Cohen JD (2008) Bold responses reflecting dopaminergic signals in the human ventral tegmental area. Science 319(5867):1264\u20131267","journal-title":"Science"},{"issue":"12","key":"674-1_CR14","doi-asserted-by":"publisher","first-page":"1704","DOI":"10.1038\/nn1560","volume":"8","author":"ND Daw","year":"2005","unstructured":"Daw ND, Niv Y, Dayan P (2005) Uncertainty-based competition between prefrontal and dorsolateral striatal systems for behavioral control. Nat Neurosci 8(12):1704\u20131711","journal-title":"Nat Neurosci"},{"issue":"6","key":"674-1_CR15","doi-asserted-by":"publisher","first-page":"1204","DOI":"10.1016\/j.neuron.2011.02.027","volume":"69","author":"ND Daw","year":"2011","unstructured":"Daw ND, Gershman SJ, Seymour B, Dayan P, Dolan RJ (2011) Model-based influences on humans\u2019 choices and striatal prediction errors. Neuron 69(6):1204\u20131215","journal-title":"Neuron"},{"issue":"8","key":"674-1_CR16","doi-asserted-by":"publisher","first-page":"1020","DOI":"10.1038\/nn1923","volume":"10","author":"JJ Day","year":"2007","unstructured":"Day JJ, Roitman MF, Wightman RM, Carelli RM (2007) Associative learning mediates dynamic shifts in dopamine signaling in the nucleus accumbens. Nat Neurosci 10(8):1020\u20131028","journal-title":"Nat Neurosci"},{"key":"674-1_CR17","doi-asserted-by":"crossref","unstructured":"Dayan P, Berridge KC (2013) Pavlovian values. Cogn Affect Behav Neurosci. 2014 Mar 20. [Epub ahead of print] doi: 10.3758\/s13415-014-0277-8","DOI":"10.3758\/s13415-014-0277-8"},{"issue":"8","key":"674-1_CR18","doi-asserted-by":"publisher","first-page":"1153","DOI":"10.1016\/j.neunet.2006.03.002","volume":"19","author":"P Dayan","year":"2006","unstructured":"Dayan P, Niv Y, Seymour B, Daw ND (2006) The misbehavior of value and the discipline of the will. Neural Netw 19(8):1153\u20131160","journal-title":"Neural Netw"},{"key":"674-1_CR19","first-page":"203","volume-title":"Mechanisms of learning and motivation","author":"A Dickinson","year":"1979","unstructured":"Dickinson A, Dearing MF (1979) Appetitive-aversive interactions and inhibitory processes. In: Dickinson A, Boakes RA (eds) Mechanisms of learning and motivation. Erlbaum, Hillsdale, pp 203\u2013231"},{"issue":"3","key":"674-1_CR20","doi-asserted-by":"publisher","first-page":"468","DOI":"10.1037\/0735-7044.114.3.468","volume":"114","author":"A Dickinson","year":"2000","unstructured":"Dickinson A, Smith J, Mirenowicz J (2000) Dissociation of Pavlovian and instrumental incentive learning under dopamine antagonists. Behav Neurosci 114(3):468\u2013483","journal-title":"Behav Neurosci"},{"key":"674-1_CR21","doi-asserted-by":"crossref","unstructured":"Dietterich TG (1999) Hierarchical reinforcement learning with the maxq value function decomposition. CoRR, cs.LG\/9905014","DOI":"10.1613\/jair.639"},{"issue":"37","key":"674-1_CR22","doi-asserted-by":"publisher","first-page":"15462","DOI":"10.1073\/pnas.1014457108","volume":"108","author":"K Enomoto","year":"2011","unstructured":"Enomoto K, Matsumoto N, Nakai S, Satoh T, Sato TK, Ueda Y, Inokawa H, Haruno M, Kimura M (2011) Dopamine neurons learn to encode the long-term value of multiple future rewards. Proc Natl Acad Sci U S A 108(37):15462\u201315467","journal-title":"Proc Natl Acad Sci U S A"},{"issue":"7328","key":"674-1_CR23","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1038\/nature09588","volume":"469","author":"SB Flagel","year":"2011","unstructured":"Flagel SB, Clark JJ, Robinson TE, Mayo L, Czuj A, Willuhn I, Akers CA, Clinton SM, Phillips PEM, Akil H (2011) A selective role for dopamine in stimulus-reward learning. Nature 469(7328):53\u201357","journal-title":"Nature"},{"issue":"5703","key":"674-1_CR24","doi-asserted-by":"publisher","first-page":"1940","DOI":"10.1126\/science.1102941","volume":"306","author":"MJ Frank","year":"2004","unstructured":"Frank MJ, Seeberger LC, O\u2019Reilly RC (2004) By carrot or by stick: cognitive reinforcement learning in Parkinsonism. Science 306(5703):1940\u20131943","journal-title":"Science"},{"issue":"7","key":"674-1_CR25","doi-asserted-by":"publisher","first-page":"718","DOI":"10.1176\/appi.ajp.2011.10071062","volume":"168","author":"CM Gillan","year":"2011","unstructured":"Gillan CM, Papmeyer M, Morein-Zamir S, Sahakian BJ, Fineberg NA, Robbins TW, de Wit S (2011) Disruption in the balance between goal-directed behavior and habit learning in obsessive-compulsive disorder. Am J Psychiatry 168(7):718\u2013726","journal-title":"Am J Psychiatry"},{"key":"674-1_CR26","doi-asserted-by":"crossref","unstructured":"Gillan CM, Morein-Zamir S, Urcelay GP, Sule A, Voon V, Apergis-Schoute AM, Fineberg NA, Sahakian BJ, Robbins TW (2014) Enhanced avoidance habits in obsessive-compulsive disorder. Biol Psychiatry 75:631\u2013638","DOI":"10.1016\/j.biopsych.2013.02.002"},{"issue":"4","key":"674-1_CR27","doi-asserted-by":"publisher","first-page":"585","DOI":"10.1016\/j.neuron.2010.04.016","volume":"66","author":"J Gl\u00e4scher","year":"2010","unstructured":"Gl\u00e4scher J, Daw N, Dayan P, O\u2019Doherty JP (2010) States versus rewards: dissociable neural prediction error signals underlying model-based and model-free reinforcement learning. Neuron 66(4):585\u2013595","journal-title":"Neuron"},{"issue":"21","key":"674-1_CR28","doi-asserted-by":"publisher","first-page":"7867","DOI":"10.1523\/JNEUROSCI.6376-10.2011","volume":"31","author":"M Guitart-Masip","year":"2011","unstructured":"Guitart-Masip M, Fuentemilla L, Bach DR, Huys QJM, Dayan P, Dolan RJ, Duzel E (2011) Action dominates valence in anticipatory representations in the human striatum and dopaminergic midbrain. J Neurosci 31(21):7867\u20137875","journal-title":"J Neurosci"},{"issue":"32","key":"674-1_CR29","doi-asserted-by":"publisher","first-page":"8360","DOI":"10.1523\/JNEUROSCI.1010-06.2006","volume":"26","author":"AN Hampton","year":"2006","unstructured":"Hampton AN, Bossaerts P, O\u2019Doherty JP (2006) The role of the ventromedial pre-frontal cortex in abstract state-based inference during decision making in humans. J Neurosci 26(32):8360\u20138367, 6","journal-title":"J Neurosci"},{"key":"674-1_CR30","volume-title":"Principles of behavior","author":"C Hull","year":"1943","unstructured":"Hull C (1943) Principles of behavior. Appleton, New York"},{"key":"674-1_CR31","unstructured":"Huys QJM (2007) Reinforcers and control. Towards a computational etiology of depression. PhD thesis, Gatsby Computational Neuroscience Unit, UCL, University of London"},{"key":"674-1_CR32","unstructured":"Huys QJM, Tobler PT, Hasler G, Flagel S. The role of learning-related dopamine signals in addiction vulnerability. Prog Neurobiol (In Press)"},{"issue":"4","key":"674-1_CR33","doi-asserted-by":"publisher","first-page":"e1002028","DOI":"10.1371\/journal.pcbi.1002028","volume":"7","author":"QJM Huys","year":"2011","unstructured":"Huys QJM, Cools R, G\u00f6lzer M, Friedel E, Heinz A, Dolan RJ, Dayan P (2011) Disentangling the roles of approach, activation and valence in instrumental and pavlovian responding. PLoS Comput Biol 7(4):e1002028","journal-title":"PLoS Comput Biol"},{"issue":"3","key":"674-1_CR34","doi-asserted-by":"publisher","first-page":"e1002410","DOI":"10.1371\/journal.pcbi.1002410","volume":"8","author":"QJM Huys","year":"2012","unstructured":"Huys QJM, Eshel N, O\u2019Nions E, Sheridan L, Dayan P, Roiser JP (2012) Bonsai trees in your head: how the Pavlovian system sculpts goal-directed choices by pruning decision trees. PLoS Comput Biol 8(3):e1002410","journal-title":"PLoS Comput Biol"},{"issue":"45","key":"674-1_CR35","doi-asserted-by":"publisher","first-page":"12176","DOI":"10.1523\/JNEUROSCI.3761-07.2007","volume":"27","author":"A Johnson","year":"2007","unstructured":"Johnson A, Redish AD (2007) Neural ensembles in ca3 transiently encode paths forward of the animal at a decision point. J Neurosci 27(45):12176\u201312189","journal-title":"J Neurosci"},{"issue":"1","key":"674-1_CR36","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1016\/S0004-3702(98)00023-X","volume":"101","author":"LP Kaelbling","year":"1998","unstructured":"Kaelbling LP, Littman ML, Cassandra AR (1998) Planning and acting in partially observable stochastic domains. Artif intell 101(1):99\u2013134","journal-title":"Artif intell"},{"key":"674-1_CR37","volume-title":"Punishment and aversive behavior","author":"LJ Kamin","year":"1969","unstructured":"Kamin LJ (1969) Predictability, surprise, attention and conditioning. In: Campbell BA, Church RM (eds) Punishment and aversive behavior. Appleton, New York"},{"issue":"2\u20133","key":"674-1_CR38","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1023\/A:1017984413808","volume":"49","author":"M Kearns","year":"2002","unstructured":"Kearns M, Singh S (2002) Near-optimal reinforcement learning in polynomial time. Mach Learn 49(2\u20133):209\u2013232","journal-title":"Mach Learn"},{"issue":"5","key":"674-1_CR39","doi-asserted-by":"publisher","first-page":"e1002055","DOI":"10.1371\/journal.pcbi.1002055","volume":"7","author":"M Keramati","year":"2011","unstructured":"Keramati M, Dezfouli A, Piray P (2011) Speed\/accuracy trade-off between the habitual and the goal-directed processes. PLoS Comput Biol 7(5):e1002055","journal-title":"PLoS Comput Biol"},{"issue":"4","key":"674-1_CR40","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1093\/cercor\/13.4.400","volume":"13","author":"S Killcross","year":"2003","unstructured":"Killcross S, Coutureau E (2003) Coordination of actions and habits in the medial prefrontal cortex of rats. Cereb Cortex 13(4):400\u2013408","journal-title":"Cereb Cortex"},{"issue":"4","key":"674-1_CR41","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/0004-3702(75)90019-3","volume":"6","author":"D Knuth","year":"1975","unstructured":"Knuth D, Moore R (1975) An analysis of alpha-beta pruning. Artif Intell 6(4):293\u2013326","journal-title":"Artif Intell"},{"key":"674-1_CR42","doi-asserted-by":"crossref","unstructured":"Kocsis L, Szepesv\u00e0ri C (2006) Bandit based Monte-Carlo planning. In: Proceedings of the Machine learning: ECML 2006, Berlin, Germany, Springer, pp 282\u2013293","DOI":"10.1007\/11871842_29"},{"issue":"2","key":"674-1_CR43","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1038\/nn.2723","volume":"14","author":"TV Maia","year":"2011","unstructured":"Maia TV, Frank MJ (2011) From reinforcement learning models to psychiatric and neurological disorders. Nat Neurosci 14(2):154\u2013162","journal-title":"Nat Neurosci"},{"key":"674-1_CR44","doi-asserted-by":"publisher","first-page":"423","DOI":"10.1016\/S0166-2236(03)00177-2","volume":"26","author":"SM McClure","year":"2003","unstructured":"McClure SM, Daw ND, Montague PR (2003) A computational substrate for incentive salience. Trends Neurosci 26:423\u2013428","journal-title":"Trends Neurosci"},{"issue":"7","key":"674-1_CR45","doi-asserted-by":"publisher","first-page":"2700","DOI":"10.1523\/JNEUROSCI.5499-10.2011","volume":"31","author":"MA McDannald","year":"2011","unstructured":"McDannald MA, Lucantonio F, Burke KA, Niv Y, Schoenbaum G (2011) Ventral striatum and orbitofrontal cortex are both required for model-based, but not model-free, reinforcement learning. J Neurosci 31(7):2700\u20132705","journal-title":"J Neurosci"},{"issue":"5","key":"674-1_CR46","doi-asserted-by":"crossref","first-page":"1936","DOI":"10.1523\/JNEUROSCI.16-05-01936.1996","volume":"16","author":"PR Montague","year":"1996","unstructured":"Montague PR, Dayan P, Sejnowski TJ (1996) A framework for mesencephalic dopamine systems based on predictive hebbian learning. J Neurosci 16(5):1936\u20131947","journal-title":"J Neurosci"},{"issue":"8","key":"674-1_CR47","doi-asserted-by":"publisher","first-page":"1057","DOI":"10.1038\/nn1743","volume":"9","author":"G Morris","year":"2006","unstructured":"Morris G, Nevet A, Arkadir D, Vaadia E, Bergman H (2006) Midbrain dopamine neurons encode decisions for future action. Nat Neurosci 9(8):1057\u20131063","journal-title":"Nat Neurosci"},{"issue":"14","key":"674-1_CR48","doi-asserted-by":"publisher","first-page":"3805","DOI":"10.1523\/JNEUROSCI.4305-05.2006","volume":"26","author":"A Nelson","year":"2006","unstructured":"Nelson A, Killcross S (2006) Amphetamine exposure enhances habit formation. J Neurosci 26(14):3805\u20133812","journal-title":"J Neurosci"},{"issue":"7447","key":"674-1_CR49","doi-asserted-by":"publisher","first-page":"74","DOI":"10.1038\/nature12112","volume":"497","author":"BE Pfeiffer","year":"2013","unstructured":"Pfeiffer BE, Foster DJ (2013) Hippocampal place-cell sequences depict future paths to remembered goals. Nature 497(7447):74\u201379","journal-title":"Nature"},{"key":"674-1_CR50","series-title":"Wiley series in probability and statistics","volume-title":"Markov decision processes: discrete stochastic dynamic programming","author":"ML Puterman","year":"2005","unstructured":"Puterman ML (2005) Markov decision processes: discrete stochastic dynamic programming, Wiley series in probability and statistics. Wiley-Interscience, New York"},{"issue":"4","key":"674-1_CR51","doi-asserted-by":"crossref","first-page":"415","DOI":"10.1017\/S0140525X0800472X","volume":"31","author":"AD Redish","year":"2008","unstructured":"Redish AD, Jensen S, Johnson A (2008) A unified framework for addiction: vulnerabilities in the decision process. Behav Brain Sci 31(4):415\u2013437; discussion 437\u2013487","journal-title":"Behav Brain Sci"},{"issue":"1","key":"674-1_CR52","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1016\/j.tics.2011.11.009","volume":"16","author":"TW Robbins","year":"2012","unstructured":"Robbins TW, Gillan CM, Smith DG, de Wit S, Ersche KD (2012) Neurocognitive endophenotypes of impulsivity and compulsivity: towards dimensional psychiatry. Trends Cogn Sci 16(1):81\u201391","journal-title":"Trends Cogn Sci"},{"issue":"4","key":"674-1_CR53","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1016\/j.cub.2013.01.016","volume":"23","author":"MJF Robinson","year":"2013","unstructured":"Robinson MJF, Berridge KC (2013) Instant transformation of learned repulsion into motivational \u201cwanting\u201d. Curr Biol 23(4):282\u2013289","journal-title":"Curr Biol"},{"issue":"12","key":"674-1_CR54","doi-asserted-by":"publisher","first-page":"1615","DOI":"10.1038\/nn2013","volume":"10","author":"MR Roesch","year":"2007","unstructured":"Roesch MR, Calu DJ, Schoenbaum G (2007) Dopamine neurons encode the better option in rats deciding between differently delayed or sized rewards. Nat Neurosci 10(12):1615\u20131624","journal-title":"Nat Neurosci"},{"issue":"12","key":"674-1_CR55","doi-asserted-by":"crossref","first-page":"885","DOI":"10.1038\/nrn2753","volume":"10","author":"G Schoenbaum","year":"2009","unstructured":"Schoenbaum G, Roesch MR, Stalnaker TA, Takahashi YK (2009) A new perspective on the role of the orbitofrontal cortex in adaptive behaviour. Nat Rev Neurosci 10(12):885\u2013892","journal-title":"Nat Rev Neurosci"},{"issue":"3","key":"674-1_CR56","doi-asserted-by":"crossref","first-page":"607","DOI":"10.1152\/jn.1990.63.3.607","volume":"63","author":"W Schultz","year":"1990","unstructured":"Schultz W, Romo R (1990) Dopamine neurons of the monkey midbrain: contingencies of responses to stimuli eliciting immediate behavioral reactions. J Neurophysiol 63(3):607\u2013624","journal-title":"J Neurophysiol"},{"issue":"5306","key":"674-1_CR57","doi-asserted-by":"publisher","first-page":"1593","DOI":"10.1126\/science.275.5306.1593","volume":"275","author":"W Schultz","year":"1997","unstructured":"Schultz W, Dayan P, Montague PR (1997) A neural substrate of prediction and reward. Science 275(5306):1593\u20131599","journal-title":"Science"},{"key":"674-1_CR58","unstructured":"Sebold M, Deserno L, Nebe S, Schad D, Garbusow M, H\u00e4gele C, Keller J, J\u00fcnger E, Kathmann N, Smolka M, Rapp MA, Schlagenhauf F, Heinz A, Huys QJM. Model-based and model-free decisions in alcohol dependence. Neuropsychobiology (In Press)"},{"issue":"2","key":"674-1_CR59","doi-asserted-by":"publisher","first-page":"361","DOI":"10.1016\/j.neuron.2013.05.038","volume":"79","author":"KS Smith","year":"2013","unstructured":"Smith KS, Graybiel AM (2013) A dual operator view of habitual behavior reflecting cortical and striatal dynamics. Neuron 79(2):361\u2013374","journal-title":"Neuron"},{"issue":"7","key":"674-1_CR60","doi-asserted-by":"publisher","first-page":"966","DOI":"10.1038\/nn.3413","volume":"16","author":"EE Steinberg","year":"2013","unstructured":"Steinberg EE, Keiflin R, Boivin JR, Witten IB, Deisseroth K, Janak PH (2013) A causal link between prediction errors, dopamine neurons and learning. Nat Neurosci 16(7):966\u2013973","journal-title":"Nat Neurosci"},{"key":"674-1_CR61","doi-asserted-by":"crossref","unstructured":"Sutton R (1990) Integrated architectures for learning, planning, and reacting based on approximating dynamic programming. In: Proceedings of the seventh international conference on machine learning, Austin, Texas, USA, vol 216, p 224","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"key":"674-1_CR62","series-title":"Computation and machine learning","volume-title":"Reinforcement learning: an introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton RS, Barto AG (1998) Reinforcement learning: an introduction, Computation and machine learning. The MIT Press, Cambridge, MA"},{"issue":"1","key":"674-1_CR63","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"RS Sutton","year":"1999","unstructured":"Sutton RS, Precup D, Singh S et al (1999) Between mdps and semi-mdps: a framework for temporal abstraction in reinforcement learning. Artif Intell 112(1):181\u2013211","journal-title":"Artif Intell"},{"issue":"5715","key":"674-1_CR64","doi-asserted-by":"publisher","first-page":"1642","DOI":"10.1126\/science.1105370","volume":"307","author":"PN Tobler","year":"2005","unstructured":"Tobler PN, Fiorillo CD, Schultz W (2005) Adaptive coding of reward value by dopamine neurons. Science 307(5715):1642\u20131645","journal-title":"Science"},{"issue":"4","key":"674-1_CR65","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1037\/h0061626","volume":"55","author":"EC Tolman","year":"1948","unstructured":"Tolman EC (1948) Cognitive maps in rats and men. Psychol Rev 55(4):189\u2013208","journal-title":"Psychol Rev"},{"issue":"15","key":"674-1_CR66","doi-asserted-by":"publisher","first-page":"4019","DOI":"10.1523\/JNEUROSCI.0564-07.2007","volume":"27","author":"VV Valentin","year":"2007","unstructured":"Valentin VV, Dickinson A, O\u2019Doherty JP (2007) Determining the neural substrates of goal-directed learning in the human brain. J Neurosci 27(15):4019\u20134026","journal-title":"J Neurosci"},{"issue":"6842","key":"674-1_CR67","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1038\/35083500","volume":"412","author":"P Waelti","year":"2001","unstructured":"Waelti P, Dickinson A, Schultz W (2001) Dopamine responses comply with basic assumptions of formal learning theory. Nature 412(6842):43\u201348","journal-title":"Nature"},{"issue":"3","key":"674-1_CR68","first-page":"279","volume":"8","author":"C Watkins","year":"1992","unstructured":"Watkins C, Dayan P (1992) Q-learning. Mach Learn 8(3):279\u2013292","journal-title":"Mach Learn"},{"issue":"3","key":"674-1_CR69","doi-asserted-by":"publisher","first-page":"418","DOI":"10.1016\/j.neuron.2012.03.042","volume":"75","author":"K Wunderlich","year":"2012","unstructured":"Wunderlich K, Smittenaar P, Dolan RJ (2012) Dopamine enhances model-based over model-free choice behavior. Neuron 75(3):418\u2013424","journal-title":"Neuron"},{"issue":"1","key":"674-1_CR70","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1111\/j.1460-9568.2004.03095.x","volume":"19","author":"HH Yin","year":"2004","unstructured":"Yin HH, Knowlton BJ, Balleine BW (2004) Lesions of dorsolateral striatum preserve outcome expectancy but disrupt habit formation in instrumental learning. Eur J Neurosci 19(1):181\u2013189","journal-title":"Eur J Neurosci"},{"issue":"2","key":"674-1_CR71","doi-asserted-by":"publisher","first-page":"513","DOI":"10.1111\/j.1460-9568.2005.04218.x","volume":"22","author":"HH Yin","year":"2005","unstructured":"Yin HH, Ostlund SB, Knowlton BJ, Balleine BW (2005) The role of the dorsomedial striatum in instrumental conditioning. Eur J Neurosci 22(2):513\u2013523","journal-title":"Eur J Neurosci"},{"issue":"5920","key":"674-1_CR72","doi-asserted-by":"publisher","first-page":"1496","DOI":"10.1126\/science.1167342","volume":"323","author":"KA Zaghloul","year":"2009","unstructured":"Zaghloul KA, Blanco JA, Weidemann CT, McGill K, Jaggi JL, Baltuch GH, Kahana MJ (2009) Human substantia nigra neurons encode unexpected financial rewards. Science 323(5920):1496\u20131499","journal-title":"Science"}],"container-title":["Encyclopedia of Computational Neuroscience"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-1-4614-7320-6_674-1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,20]],"date-time":"2023-02-20T02:16:26Z","timestamp":1676859386000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-1-4614-7320-6_674-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014]]},"ISBN":["9781461473206"],"references-count":72,"URL":"https:\/\/doi.org\/10.1007\/978-1-4614-7320-6_674-1","relation":{},"subject":[],"published":{"date-parts":[[2014]]},"assertion":[{"value":"28 January 2014, 12:28:14","order":1,"name":"received","label":"Received","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"28 January 2014, 12:28:14","order":2,"name":"accepted","label":"Accepted","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"21 February 2014","order":3,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}