{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,6]],"date-time":"2024-09-06T23:11:54Z","timestamp":1725664314865},"publisher-location":"Berlin, Heidelberg","reference-count":14,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540584094"},{"type":"electronic","value":"9783540487807"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[1994]]},"DOI":"10.1007\/3-540-58409-9_1","type":"book-chapter","created":{"date-parts":[[2012,2,26]],"date-time":"2012-02-26T10:55:46Z","timestamp":1330253746000},"page":"1-9","source":"Crossref","is-referenced-by-count":14,"title":["Fuzzy reinforcement Learning and dynamic programming"],"prefix":"10.1007","author":[{"given":"Hamid R.","family":"Berenji","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2005,6,2]]},"reference":[{"key":"1_CR1","unstructured":"A. G. Barto, S. Bradtke, and S. Singh. Learning to act using real-time dynamic programming. Submitted to AI Journal special issue on Computational Theories of Interaction and Agency, 1993."},{"key":"1_CR2","doi-asserted-by":"crossref","first-page":"834","DOI":"10.1109\/TSMC.1983.6313077","volume":"13","author":"A. G. Barto","year":"1983","unstructured":"A. G. Barto, R. S. Sutton, and C. W. Anderson. Neuronlike adaptive elements that can solve difficult learning control problems. IEEE Transactions on Systems, Man, and Cybernetics, 13:834\u2013846, 1983.","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics"},{"key":"1_CR3","volume-title":"Dynamic Programming","author":"R. Bellman","year":"1957","unstructured":"R. Bellman. Dynamic Programming. Princeton University Press, Princeton, NJ, 1957."},{"issue":"4","key":"1_CR4","doi-asserted-by":"crossref","first-page":"B","DOI":"10.1287\/mnsc.17.4.B141","volume":"17","author":"R.E. Bellman","year":"1970","unstructured":"R.E. Bellman and L.A. Zadeh. Decision-making in a fuzzy environment. Management Science, 17(4):B\u2013141:B-164, 1970.","journal-title":"Management Science"},{"key":"1_CR5","doi-asserted-by":"crossref","unstructured":"H.R. Berenji and P. Khedkar. Learning and tuning fuzzy logic controllers through reinforcements. IEEE Transactions on Neural Networks, 3(5), 1992.","DOI":"10.1109\/72.159061"},{"key":"1_CR6","unstructured":"H.R. Berenji, Y. Jani R.N Lea, P. Khedkar, A. Malkani, and J. Hoblit. Space shuttle attitude control by fuzzy logic and reinforcement learning. In Second IEEE International conference on Fuzzy Systems, San Francisco, CA, March 1993."},{"key":"1_CR7","unstructured":"L.J. Lin. Programming robots using reinforcement learning and teaching. In Proceedings of the Ninth National Conference on Artificial Intelligence, 1991."},{"key":"1_CR8","unstructured":"A. Moore and C. Atkeson. Prioritized sweeping: Reinforcement learning with less data and less real time. Machine Learning, page to appear."},{"key":"1_CR9","first-page":"9","volume":"3","author":"R.S. Sutton","year":"1988","unstructured":"R.S. Sutton. Learning to predict by the methods of temporal differences. Machine Learning, 3:9\u201344, 1988.","journal-title":"Machine Learning"},{"key":"1_CR10","doi-asserted-by":"crossref","unstructured":"R.S. Sutton. Integrated architectures for learning, planning, and reacting based on approximating dynamic programming. In Proceedings of the Seventh International Conference on Machine Learning, 1990.","DOI":"10.1016\/B978-1-55860-141-3.50030-4"},{"key":"1_CR11","first-page":"257","volume":"8","author":"G. Tesauro","year":"1992","unstructured":"G. Tesauro. Practical issues in temporal difference learning. Machine Learning, (8):257\u2013277, 1992.","journal-title":"Machine Learning"},{"issue":"2","key":"1_CR12","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1162\/neco.1994.6.2.215","volume":"6","author":"G. Tesauro","year":"1994","unstructured":"G. Tesauro. Td-gammon, a self-teaching backgammon program, achieves master-level play. Neural Computation, 6(2):215\u2013219, 1994.","journal-title":"Neural Computation"},{"key":"1_CR13","first-page":"279","volume":"8","author":"C. Watkins","year":"1992","unstructured":"C. Watkins and P. Dayan. Q-learning. Machine Learning, (8):279\u2013292, 1992.","journal-title":"Machine Learning"},{"key":"1_CR14","unstructured":"C.J.C.H. Watkins. Learning with Delayed Rewards. PhD thesis, Cambridge University, Psychology Department, 1989."}],"container-title":["Lecture Notes in Computer Science","Fuzzy Logic in Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/3-540-58409-9_1.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,17]],"date-time":"2020-11-17T16:20:12Z","timestamp":1605630012000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/3-540-58409-9_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[1994]]},"ISBN":["9783540584094","9783540487807"],"references-count":14,"URL":"https:\/\/doi.org\/10.1007\/3-540-58409-9_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[1994]]}}}