{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,23]],"date-time":"2026-01-23T05:16:11Z","timestamp":1769145371916,"version":"3.49.0"},"reference-count":14,"publisher":"Hindawi Limited","issue":"8","license":[{"start":{"date-parts":[[2015,9,1]],"date-time":"2015-09-01T00:00:00Z","timestamp":1441065600000},"content-version":"tdm","delay-in-days":8278,"URL":"http:\/\/doi.wiley.com\/10.1002\/tdm_license_1.1"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Int. J. Intell. Syst."],"published-print":{"date-parts":[[1993]]},"DOI":"10.1002\/int.4550080805","type":"journal-article","created":{"date-parts":[[2007,7,8]],"date-time":"2007-07-08T19:39:28Z","timestamp":1183923568000},"page":"875-894","source":"Crossref","is-referenced-by-count":14,"title":["Reinforcement learning: Architectures and algorithms"],"prefix":"10.1155","volume":"8","author":[{"given":"Mieczyslaw M.","family":"Kokar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Spiridon A.","family":"Reveliotis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"98","reference":[{"key":"10.1002\/int.4550080805-BIB1","author":"Stephanou","year":"1988","unstructured":", and , \u201cIntelligent control: From perception to action,\u201d In Proceedings of the IEEE International Symposium on Intelligent Control, 1988."},{"key":"10.1002\/int.4550080805-BIB2","author":"Sutton","year":"1990","unstructured":"\u201cIntegrated architectures for learning, planning, and reacting based on approximating dynaming programming,\u201d In Proceedings of the 7th International Conference on Machine Learning, 1990, pp. 216\u2013224."},{"key":"10.1002\/int.4550080805-BIB3","author":"Kaelbling","year":"1990","unstructured":"Learning in Embedded Systems, Technical Report TR-90-04, Teleos Research, CA, 1990."},{"key":"10.1002\/int.4550080805-BIB4","doi-asserted-by":"crossref","first-page":"323","DOI":"10.1109\/TSMC.1974.5408453","volume":"4","author":"Narendra","year":"1974","journal-title":"IEEE Trans. Syst. Man Cybern."},{"key":"10.1002\/int.4550080805-BIB5","author":"Watkins","year":"1989","unstructured":"\u201cLearning with delayed rewards,\u201d Ph.D. Thesis, Cambridge University, Psychology Department, 1989."},{"key":"10.1002\/int.4550080805-BIB6","first-page":"279","volume":"8","author":"Watkins","year":"1992","journal-title":"Mach. Learn."},{"key":"10.1002\/int.4550080805-BIB7","doi-asserted-by":"crossref","first-page":"834","DOI":"10.1109\/TSMC.1983.6313077","volume":"13","author":"Barto","year":"1983","journal-title":"IEEE Trans. Syst. Man Cyber."},{"key":"10.1002\/int.4550080805-BIB8","author":"Williams","year":"1987","unstructured":"\u201cA class of gradient-estimating algorithms for reinforcement learning in neural networks,\u201d In Proceedins of the IEEE First Annual International Conference on Neural Networks, 1987, pp. 601\u2013608."},{"key":"10.1002\/int.4550080805-BIB9","first-page":"229","volume":"8","author":"Williams","year":"1992","journal-title":"Mach. Learn."},{"key":"10.1002\/int.4550080805-BIB10","author":"Sutton","year":"1984","unstructured":"\u201cTemporal credit assignment in reinforcement learning,\u201d Ph.D. Thesis, University of Massachusetts, Amherst, Amherst, MA, 1984."},{"key":"10.1002\/int.4550080805-BIB11","author":"Kokar","year":"1991","unstructured":"and , \u201cIntegrating qualitative and quantitative methods for model validation and monitoring,\u201d In Proceedings of the 1991 IEEE International Symposium on Intelligent Control, 1991, pp. 286\u2013291."},{"key":"10.1002\/int.4550080805-BIB12","author":"Whitehead","year":"1990","unstructured":"and , \u201cActive perception and reinforcement learning,\u201d In Proceedings of the 7th International Conference on Machine Learning, 1990, pp. 179\u2013188."},{"key":"10.1002\/int.4550080805-BIB13","first-page":"2714","volume-title":"Systems and Control Encyclopedia: Theory, Technology and Applications","author":"Narendra","year":"1987","unstructured":"\u201cLarge stochastic systems: Learning automata,\u201d In Systems and Control Encyclopedia: Theory, Technology and Applications, (Ed.) Pergamon, New York, 1987, pp. 2714\u20132719."},{"key":"10.1002\/int.4550080805-BIB14","author":"Franklin","year":"1988","unstructured":"\u201cRefinement of robot motor skills through reinforcement learning,\u201d In Proceedings of the 27th Conference on Decision and Control, 1988, pp. 1096\u20131101."}],"container-title":["International Journal of Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.wiley.com\/onlinelibrary\/tdm\/v1\/articles\/10.1002%2Fint.4550080805","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/onlinelibrary.wiley.com\/doi\/full\/10.1002\/int.4550080805","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,9]],"date-time":"2023-02-09T15:33:53Z","timestamp":1675956833000},"score":1,"resource":{"primary":{"URL":"https:\/\/onlinelibrary.wiley.com\/doi\/10.1002\/int.4550080805"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[1993]]},"references-count":14,"journal-issue":{"issue":"8","published-print":{"date-parts":[[1993]]}},"URL":"https:\/\/doi.org\/10.1002\/int.4550080805","relation":{},"ISSN":["0884-8173","1098-111X"],"issn-type":[{"value":"0884-8173","type":"print"},{"value":"1098-111X","type":"electronic"}],"subject":[],"published":{"date-parts":[[1993]]}}}