{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T08:42:52Z","timestamp":1742978572724,"version":"3.40.3"},"publisher-location":"Berlin, Heidelberg","reference-count":34,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642398018"},{"type":"electronic","value":"9783642398025"}],"license":[{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2013,1,1]],"date-time":"2013-01-01T00:00:00Z","timestamp":1356998400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-39802-5_17","type":"book-chapter","created":{"date-parts":[[2013,7,1]],"date-time":"2013-07-01T13:37:04Z","timestamp":1372685824000},"page":"191-203","source":"Crossref","is-referenced-by-count":6,"title":["Learning Epistemic Actions in Model-Free Memory-Free Reinforcement Learning: Experiments with a Neuro-robotic Model"],"prefix":"10.1007","author":[{"given":"Dimitri","family":"Ognibene","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nicola Catenacci","family":"Volpi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giovanni","family":"Pezzulo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gianluca","family":"Baldassare","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"9","key":"17_CR1","doi-asserted-by":"publisher","first-page":"1214","DOI":"10.1038\/nn1954","volume":"10","author":"T.E.J. Behrens","year":"2007","unstructured":"Behrens, T.E.J., Woolrich, M.W., Walton, M.E., Rushworth, M.F.S.: Learning the value of information in an uncertain world. Nat. Neurosci.\u00a010(9), 1214\u20131221 (2007)","journal-title":"Nat. Neurosci."},{"issue":"7210","key":"17_CR2","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1038\/nature07200","volume":"455","author":"A. Kepecs","year":"2008","unstructured":"Kepecs, A., Uchida, N., Zariwala, H.A., Mainen, Z.F.: Neural correlates, computation and behavioural impact of decision confidence. Nature\u00a0455(7210), 227\u2013231 (2008)","journal-title":"Nature"},{"key":"17_CR3","doi-asserted-by":"publisher","first-page":"92","DOI":"10.3389\/fpsyg.2013.00092","volume":"4","author":"G. Pezzulo","year":"2013","unstructured":"Pezzulo, G., Rigoli, F., Chersi, F.: The mixed instrumental controller: using value of information to combine habitual choice and mental simulation. Front Psychol.\u00a04, 92 (2013)","journal-title":"Front Psychol."},{"key":"17_CR4","unstructured":"Roy, N., Thrun, S.: Coastal navigation with mobile robots. In: Advances in Neural Information Processing Systems, vol.\u00a012 (2000)"},{"key":"17_CR5","unstructured":"Cassandra, A., Kaelbling, L., Kurien, J.: Acting under uncertainty: discrete bayesian models for mobile-robotnavigation. In: Proc. of IROS 1996 (1996)"},{"key":"17_CR6","unstructured":"Kwok, C., Fox, D.: Reinforcement learning for sensing strategies. In: Proc. of IROS 2004 (2004)"},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Hsiao, K., Kaelbling, L., Lozano-Perez, T.: Task-driven tactile exploration. In: Proc. of Robotics: Science and Systems (RSS) (2010)","DOI":"10.15607\/RSS.2010.VI.029"},{"key":"17_CR8","doi-asserted-by":"crossref","unstructured":"Lepora, N., Martinez, U., Prescott, T.: Active touch for robust perception under position uncertainty. In: IEEE Proceedings of ICRA (2013)","DOI":"10.1109\/ICRA.2013.6630996"},{"issue":"2","key":"17_CR9","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1109\/JSEN.2011.2148114","volume":"12","author":"J. Sullivan","year":"2012","unstructured":"Sullivan, J., Mitchinson, B., Pearson, M.J., Evans, M., Lepora, N.F., Fox, C.W., Melhuish, C., Prescott, T.J.: Tactile discrimination using active whisker sensors. IEEE Sensors Journal\u00a012(2), 350\u2013362 (2012)","journal-title":"IEEE Sensors Journal"},{"key":"17_CR10","unstructured":"Moore, R.: 9 a formal theory of knowledge and action. In: Hobbs, J., Moore, R. (eds.) Formal Theories of the Commonsense World. Intellect Books (1985)"},{"key":"17_CR11","unstructured":"Herzig, A., Lang, J., Marquis, P.: Action representation and partially observable planning in epistemic logic. In: Proc. of IJCAI 2003 (2003)"},{"issue":"4","key":"17_CR12","doi-asserted-by":"publisher","first-page":"513","DOI":"10.1207\/s15516709cog1804_1","volume":"18","author":"D. Kirsh","year":"1994","unstructured":"Kirsh, D., Maglio, P.: On distinguishing epistemic from pragmatic action. Cognitive Science\u00a018(4), 513\u2013549 (1994)","journal-title":"Cognitive Science"},{"key":"17_CR13","doi-asserted-by":"crossref","unstructured":"Kirsh, D.: Thinking with external representations. AI & Society (February 2010)","DOI":"10.1007\/s00146-010-0272-8"},{"key":"17_CR14","unstructured":"Cassandra, A.R.: Exact and Approximate Algorithms for Partially Observable Markov Decision Processes. PhD thesis, Brown University (1998)"},{"key":"17_CR15","unstructured":"Melo, F.S., Ribeiro, I.M.: Transition entropy in partially observable markov decision processes. In: Proc. of the 9th IAS, pp. 282\u2013289 (2006)"},{"issue":"2","key":"17_CR16","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1109\/34.982896","volume":"24","author":"J. Denzler","year":"2002","unstructured":"Denzler, J., Brown, C.: Information theoretic sensor data selection for active object recognition and state estimation. IEEE Trans. on Pattern Analysis and Machine Intelligence\u00a024(2), 145\u2013157 (2002)","journal-title":"IEEE Trans. on Pattern Analysis and Machine Intelligence"},{"issue":"1-2","key":"17_CR17","doi-asserted-by":"publisher","first-page":"271","DOI":"10.1016\/0004-3702(94)00012-P","volume":"73","author":"S. Whitehead","year":"1995","unstructured":"Whitehead, S., Lin, L.: Reinforcement learning of non-markov decision processes. Artificial Intelligence\u00a073(1-2), 271\u2013306 (1995)","journal-title":"Artificial Intelligence"},{"key":"17_CR18","doi-asserted-by":"crossref","unstructured":"Vlassis, N., Toussaint, M.: Model-free reinforcement learning as mixture learning. In: Proc. of the 26th Ann. Int. Conf. on Machine Learning, pp. 1081\u20131088. ACM (2009)","DOI":"10.1145\/1553374.1553512"},{"key":"17_CR19","volume-title":"Reinforcement Learning: An Introduction","author":"R. Sutton","year":"1998","unstructured":"Sutton, R., Barto, A.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (1998)"},{"issue":"1-4","key":"17_CR20","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1016\/S0925-2312(01)00598-7","volume":"42","author":"S. Nolfi","year":"2002","unstructured":"Nolfi, S.: Power and the limits of reactive agents. Neurocomputing\u00a042(1-4), 119\u2013145 (2002)","journal-title":"Neurocomputing"},{"key":"17_CR21","unstructured":"Aberdeen, D., Baxter, J.: Scalable internal-state policy-gradient methods for pomdps. In: Proc. of Int. Conf. Machine Learning, pp. 3\u201310 (2002)"},{"issue":"1","key":"17_CR22","first-page":"45","volume":"7","author":"S.D. Whitehead","year":"1991","unstructured":"Whitehead, S.D., Ballard, D.H.: Learning to perceive and act by trial and error. Machine Learning\u00a07(1), 45\u201383 (1991)","journal-title":"Machine Learning"},{"key":"17_CR23","doi-asserted-by":"crossref","unstructured":"Koenig, S., Simmons, R.G.: The effect of representation and knowledge on goal-directed exploration with reinforcement-learning algorithms. Mach. Learn. (1996)","DOI":"10.1007\/BF00114729"},{"key":"17_CR24","unstructured":"Ognibene, D.: Ecological Adaptive Perception from a Neuro-Robotic perspective: theory, architecture and experiments. PhD thesis, University of Genoa (May 2009)"},{"key":"17_CR25","doi-asserted-by":"crossref","unstructured":"Ognibene, D., Pezzulo, G., Baldassarre, G.: Learning to look in different environments: An active-vision model which learns and readapts visual routines. In: Proc. of the 11th Conf. on Simulation of Adaptive Behaviour (2010)","DOI":"10.1007\/978-3-642-15193-4_19"},{"issue":"2","key":"17_CR26","first-page":"171","volume":"1","author":"C. Balkenius","year":"2000","unstructured":"Balkenius, C.: Attention, habituation and conditioning: Toward a computational model. Cognitive Science Quarterly\u00a01(2), 171\u2013204 (2000)","journal-title":"Cognitive Science Quarterly"},{"key":"17_CR27","doi-asserted-by":"crossref","unstructured":"Sutton, R.S.: Integrated architectures for learning, planning, and reacting based on approximating dynamic programming. In: Proc. ICML, pp. 216\u2013224 (1990)","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"issue":"3731","key":"17_CR28","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1126\/science.153.3731.25","volume":"153","author":"Berlyne","year":"1966","unstructured":"Berlyne: Curiosity and exploration. Science\u00a0153(3731), 9\u201396 (1966)","journal-title":"Science"},{"key":"17_CR29","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-32375-1","volume-title":"Intrinsically Motivated Learning in Natural and Artificial Systems","author":"G. Baldassarre","year":"2013","unstructured":"Baldassarre, G., Mirolli, M.: Intrinsically Motivated Learning in Natural and Artificial Systems. Springer, Berlin (2013)"},{"key":"17_CR30","doi-asserted-by":"crossref","unstructured":"Tishby, N., Polani, D.: Information theory of decisions and actions. In: Perception-Action Cycle, pp. 601\u2013636. Springer (2011)","DOI":"10.1007\/978-1-4419-1452-1_19"},{"key":"17_CR31","unstructured":"Ng, A.Y., Harada, D., Russell, S.: Policy invariance under reward transformations: Theory and application to reward shaping. In: Proc.of the ICML, pp. 278\u2013287 (1999)"},{"key":"17_CR32","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1177\/1059712303114001","volume":"11","author":"R.D. Beer","year":"2003","unstructured":"Beer, R.D.: The dynamics of active categorical perception in an evolved model agent. Adapt. Behav.\u00a011, 209\u2013243 (2003)","journal-title":"Adapt. Behav."},{"key":"17_CR33","doi-asserted-by":"crossref","unstructured":"Friston, K., Adams, R.A., Perrinet, L., Breakspear, M.: Perceptions as hypotheses: saccades as experiments. Frontiers in Psychology\u00a03 (2012)","DOI":"10.3389\/fpsyg.2012.00151"},{"key":"17_CR34","doi-asserted-by":"crossref","unstructured":"Ortega, P.A., Braun, D.A.: Thermodynamics as a theory of decision-making with information-processing costs. Proceedings of the Royal Society A: Mathematical, Physical and Engineering Science\u00a0469(2153) (2013)","DOI":"10.1098\/rspa.2012.0683"}],"container-title":["Lecture Notes in Computer Science","Biomimetic and Biohybrid Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-39802-5_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,7]],"date-time":"2023-02-07T16:00:19Z","timestamp":1675785619000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-642-39802-5_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642398018","9783642398025"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-39802-5_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2013]]}}}