{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,6]],"date-time":"2024-09-06T23:05:14Z","timestamp":1725663914063},"publisher-location":"Berlin, Heidelberg","reference-count":13,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540567981"},{"type":"electronic","value":"9783540477419"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[1993]]},"DOI":"10.1007\/3-540-56798-4_157","type":"book-chapter","created":{"date-parts":[[2012,2,26]],"date-time":"2012-02-26T11:34:51Z","timestamp":1330256091000},"page":"261-266","source":"Crossref","is-referenced-by-count":0,"title":["B-Learning: A reinforcement learning algorithm, comparison with dynamic programming"],"prefix":"10.1007","author":[{"given":"Thibault","family":"Langlols","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"St\u00e9phane","family":"Canu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2005,6,1]]},"reference":[{"key":"42_CR1","doi-asserted-by":"crossref","unstructured":"Charles W. Anderson. Learning to control an inverted pendulum using neural networks. IEEE Control Magazine., pages 31\u201337, april 1989.","DOI":"10.1109\/37.24809"},{"key":"42_CR2","volume-title":"Technical Report 91-57","author":"A. G. Barto","year":"1991","unstructured":"Andrew G. Barto, Steven J. Bradtke, and Satinder P. Singh. Real-time learning and control using asynchronous dynamic programming. Technical Report 91-57, University of Massachusetts, Dept of Computer Science, Amherst MA 01003, August 1991."},{"issue":"5","key":"42_CR3","doi-asserted-by":"crossref","first-page":"834","DOI":"10.1109\/TSMC.1983.6313077","volume":"SMC-13","author":"A. G. Barto","year":"1983","unstructured":"Andrew G. Barto, Richard S. Sutton, and Charles W. Anderson. Neuronlike adaptive elements that can solve difficult learning problems. IEEE Transactions on Systems, Man and Cybernetics, SMC-13(5):834\u2013846, September October 1983.","journal-title":"IEEE Transactions on Systems, Man and Cybernetics"},{"key":"42_CR4","unstructured":"J. C. Hoskins and D. M. Himelblau. Process control via incremental neural networks and reinforcement learning. In Chicago Meeting of American Institute of Chemical Engineers, Chicago Illinois, November 1990."},{"key":"42_CR5","unstructured":"Wayne C. Jouse and John G. Williams. The control of nuclear reactor start-up using drive reinforcement theory. In Cihan H. Dagli, Soundar R. T. Kumara, and Yung C. Shin, editors, Intelligent Engineering Systems Through Artificial Neural Networks, pages 537\u2013544, St. Louis, Missouri, USA, November 1991."},{"key":"42_CR6","unstructured":"R. Kora, P. Lesueur, and P. Villon. An adaptive optimal control algorithm for water treatment plants. In to be published, 1992."},{"key":"42_CR7","unstructured":"Thibault Langlois and St\u00e9phane Canu. B-learning: a reinforcement learning variant for the control of a plant. In Intelligent Engineering Systems Through Artificial Neural Networks (ANNIE'92). ASME Press, 1992."},{"key":"42_CR8","unstructured":"Thibault Langlois and St\u00e9phane Canu. Control of time-delay systems using reinforcement learning. In Artificial Neural Networks, 2. Elsevier Science Publishers, 1992."},{"key":"42_CR9","unstructured":"Long-Ji Lin. Programming robots using reinforcement learning and teaching. In NinthNational Conference on Artificial Intelligence, pages 781\u2013786, 1991."},{"issue":"3\u20134","key":"42_CR10","first-page":"923","volume":"8","author":"L. Lin","year":"1992","unstructured":"Long-Ji Lin. Self-improving reactive agents based on reinforcement learning, planning and teaching. Machine-Learning, 8(3\u20134):923\u2013321, 1992.","journal-title":"Machine-Learning"},{"key":"42_CR11","first-page":"9","volume":"3","author":"R. S. Sutton","year":"1988","unstructured":"Richard S. Sutton. Learning to predict by the method of temporal differences. Machine learning, 3:9\u201344, 1988.","journal-title":"Machine learning"},{"key":"42_CR12","unstructured":"Richard S. Sutton. Integrated modeling control based on reinforcement learning and dynamic programming, hi Richard P. Lippman, John E. Moody, and David S. Touretzky, editors, Advances in Neural Information Processing Systems, volume 3, pages 471\u2013478. Morgan Kaufmann, 1990."},{"key":"42_CR13","unstructured":"Christopher J. C. H. Watkins. Learning with Delayed Rewards. PhD thesis, Cambridge University Psychology Department, 1989."}],"container-title":["Lecture Notes in Computer Science","New Trends in Neural Computation"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/3-540-56798-4_157.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,4,28]],"date-time":"2021-04-28T00:55:24Z","timestamp":1619571324000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/3-540-56798-4_157"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[1993]]},"ISBN":["9783540567981","9783540477419"],"references-count":13,"URL":"https:\/\/doi.org\/10.1007\/3-540-56798-4_157","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[1993]]}}}