{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T07:01:29Z","timestamp":1758265289185},"reference-count":23,"publisher":"Springer Science and Business Media LLC","issue":"3-4","license":[{"start":{"date-parts":[[2015,5,2]],"date-time":"2015-05-02T00:00:00Z","timestamp":1430524800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Intell Robot Syst"],"published-print":{"date-parts":[[2016,3]]},"DOI":"10.1007\/s10846-015-0222-2","type":"journal-article","created":{"date-parts":[[2015,5,1]],"date-time":"2015-05-01T08:49:33Z","timestamp":1430470173000},"page":"407-422","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Multiple Model Q-Learning for Stochastic Asynchronous Rewards"],"prefix":"10.1007","volume":"81","author":[{"given":"Jeffrey S.","family":"Campbell","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sidney N.","family":"Givigi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Howard M.","family":"Schwartz","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,5,2]]},"reference":[{"issue":"2","key":"222_CR1","doi-asserted-by":"crossref","first-page":"128","DOI":"10.1049\/iet-its.2009.0070","volume":"4","author":"I Arel","year":"2010","unstructured":"Arel, I., Liu, C., Urbanik, T., Kohls, A.G.: Reinforce-ment learning-based multi-agent system for network traffic signal control. Intell. Trans. Syst. IET 4(2), 128\u2013135 (2010)","journal-title":"Intell. Trans. Syst. IET"},{"issue":"2-4","key":"222_CR2","doi-asserted-by":"crossref","first-page":"161","DOI":"10.1007\/s10846-005-5137-x","volume":"43","author":"D Borrajo","year":"2005","unstructured":"Borrajo, D., Parker, L.E., et al.: A reinforcement learning algorithm in cooperative multi-robot domains. J. Intell. Robot. Syst. 43(2-4), 161\u2013174 (2005)","journal-title":"J. Intell. Robot. Syst."},{"issue":"3","key":"222_CR3","first-page":"236","volume":"35","author":"AS Campbell","year":"2007","unstructured":"Campbell, A.S., Schwartz, H.M.: Multiple model control improvements: hypothesis testing and modified model arrangement. Control Intell. Syst. 35(3), 236\u2013243 (2007)","journal-title":"Control Intell. Syst."},{"key":"222_CR4","doi-asserted-by":"crossref","unstructured":"Campbell, J.S., Givigi, S.N., Schwartz, H.M.: Multiple model Q-learning for stochastic reinforcement delays. In: Proceedings of the 2014 IEEE international conference on systems, man, and cybernetics. SMC (2014)","DOI":"10.1109\/SMC.2014.6974146"},{"issue":"2","key":"222_CR5","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1109\/MRA.2008.921541","volume":"15","author":"C Chen","year":"2008","unstructured":"Chen, C., Li, H.-X., Dong, D.: Hybrid control for robot navigation - a hierarchical q-learning algorithm. IEEE Robot. Autom. Mag. 15(2), 37\u201347 (2008)","journal-title":"IEEE Robot. Autom. Mag."},{"issue":"1","key":"222_CR6","doi-asserted-by":"crossref","first-page":"92","DOI":"10.1109\/TSMCC.2005.860578","volume":"36","author":"VLR Chinthalapati","year":"2006","unstructured":"Chinthalapati, V.L.R., Yadati, N., Karumanchi, R.: Learning dynamic prices in multiseller electronic retail markets with price sensitive customers, stochastic demands, and inventory replenishments. IEEE Trans. Syst., Man, Cybern., Part C: Appl. Rev. 36(1), 92\u2013106 (2006)","journal-title":"IEEE Trans. Syst., Man, Cybern., Part C: Appl. Rev."},{"issue":"10","key":"222_CR7","doi-asserted-by":"crossref","first-page":"1242","DOI":"10.1109\/TMC.2008.26","volume":"7","author":"S Gonzalez-Valenzuela","year":"2008","unstructured":"Gonzalez-Valenzuela, S., Vuong, S.T., Leung, V.C.M.: A mobile-directory approach to service discovery in wire- less ad hoc networks. IEEE Trans. Mob. Comput. 7(10), 1242\u20131256 (2008)","journal-title":"IEEE Trans. Mob. Comput."},{"issue":"6","key":"222_CR8","doi-asserted-by":"crossref","first-page":"1185","DOI":"10.1162\/neco.1994.6.6.1185","volume":"6","author":"T Jaakkola","year":"1994","unstructured":"Jaakkola, T., Jordan, M.I., Singh, S.P.: On the convergence of stochastic iterative dynamic programming algorithms. Neural Comput. 6(6), 1185\u20131201 (1994)","journal-title":"Neural Comput."},{"issue":"2","key":"222_CR9","doi-asserted-by":"crossref","first-page":"217","DOI":"10.1007\/s10846-010-9422-y","volume":"60","author":"U Kartoun","year":"2010","unstructured":"Kartoun, U., Stern, H., Edan, Y.: A human-robot collaborative reinforcement learning algorithm. J. Intell. Robot. Syst. 60(2), 217\u2013239 (2010)","journal-title":"J. Intell. Robot. Syst."},{"issue":"4","key":"222_CR10","doi-asserted-by":"crossref","first-page":"568","DOI":"10.1109\/TAC.2003.809799","volume":"48","author":"KV Katsikopoulos","year":"2003","unstructured":"Katsikopoulos, K.V., Engelbrecht, S.E.: Markov decision processes with delays and asynchronous cost collection. IEEE Trans. Autom. Control 48(4), 568\u2013574 (2003)","journal-title":"IEEE Trans. Autom. Control"},{"key":"222_CR11","first-page":"579","volume-title":"Reinforcement Learning, volume 12 of Adaptation, Learning, and Optimization","author":"J Kober","year":"2012","unstructured":"Kober, J., Peters, J.: Reinforcement learning in robotics: A survey. In: Wiering, M., Otterlo, M. (eds.) Reinforcement Learning, volume 12 of Adaptation, Learning, and Optimization, pp. 579\u2013610. Springer, Berlin Heidelberg (2012)"},{"issue":"5","key":"222_CR12","doi-asserted-by":"crossref","first-page":"547","DOI":"10.1109\/TSMCC.2010.2044174","volume":"40","author":"M Rahimiyan","year":"2010","unstructured":"Rahimiyan, M., Mashhadi, H.R.: An adaptive q -learning algorithm developed for agent-based computational modeling of electricity market. IEEE Trans. Syst., Man, Cybern., Part C: Appl. Rev. 40(5), 547\u2013556 (2010)","journal-title":"IEEE Trans. Syst., Man, Cybern., Part C: Appl. Rev."},{"issue":"1","key":"222_CR13","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1023\/A:1007968115863","volume":"21","author":"CHC Ribeiro","year":"1998","unstructured":"Ribeiro, C.H.C.: Embedding a priori knowledge in reinforcement learning. J. Intell. Robot. Syst. 21(1), 51\u201371 (1998)","journal-title":"J. Intell. Robot. Syst."},{"issue":"1-2","key":"222_CR14","doi-asserted-by":"crossref","first-page":"513","DOI":"10.1007\/s10846-013-9959-7","volume":"74","author":"OK Sahingoz","year":"2014","unstructured":"Sahingoz, O.K.: Networking models in flying ad-hoc networks (FANETs): Concepts and challenges. J. Intell. Robot. Syst. 74(1-2), 513\u2013527 (2014)","journal-title":"J. Intell. Robot. Syst."},{"issue":"1","key":"222_CR15","first-page":"9","volume":"3","author":"RS Sutton","year":"1988","unstructured":"Sutton, R.S.: Learning to predict by the methods of temporal differences. Mach. Learn. 3(1), 9\u201344 (1988)","journal-title":"Mach. Learn."},{"key":"222_CR16","doi-asserted-by":"crossref","unstructured":"Sutton, R.S., Andrew, G.B.: Reinforcement learning: Introduction (1998)","DOI":"10.1109\/TNN.1998.712192"},{"key":"222_CR17","doi-asserted-by":"crossref","unstructured":"Szita, I., Lo\u030brincz, A.: Optimistic initialization and greediness lead to polynomial time learning in factored mdps. In: Proceedings of the 26th annual international conference on machine learning, pp. 1001\u20131008. ACM (2009)","DOI":"10.1145\/1553374.1553502"},{"issue":"7","key":"222_CR18","doi-asserted-by":"crossref","first-page":"1744","DOI":"10.1109\/TPAMI.2012.252","volume":"35","author":"O Teboul","year":"2013","unstructured":"Teboul, O., Kokkinos, I., Simon, L., Koutsourakis, P., Paragios, N.: Parsing facades with shape grammars and reinforcement learning. IEEE Trans. Pattern Anal. Mach. Intell. 35(7), 1744\u20131756 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"3","key":"222_CR19","first-page":"185","volume":"16","author":"JN Tsitsiklis","year":"1994","unstructured":"Tsitsiklis, J.N.: Asynchronous stochastic approximation and Q-learning. Mach. Learn. 16(3), 185\u2013202 (1994)","journal-title":"Mach. Learn."},{"issue":"1","key":"222_CR20","doi-asserted-by":"crossref","first-page":"83","DOI":"10.1007\/s10458-008-9056-7","volume":"18","author":"TJ Walsh","year":"2009","unstructured":"Walsh, T.J., Nouri, A., Li, L., Littman, M.L.: Learning and planning in environments with delayed feedback. Auton. Agents Multi-Agent Syst. 18(1), 83\u2013105 (2009)","journal-title":"Auton. Agents Multi-Agent Syst."},{"issue":"1","key":"222_CR21","doi-asserted-by":"crossref","first-page":"17","DOI":"10.1109\/TCIAIG.2009.2037972","volume":"2","author":"H Wang","year":"2010","unstructured":"Wang, H., Gao, Y., Chen, X.: Rl-dot: A reinforcement learning npc team for playing domination games. IEEE Trans. Comput. Intell. AI Games 2(1), 17\u201326 (2010)","journal-title":"IEEE Trans. Comput. Intell. AI Games"},{"key":"222_CR22","doi-asserted-by":"crossref","unstructured":"Watkins, C.J.CH., Dayan, P.: Q-learning. Machine Learning (1992)","DOI":"10.1007\/BF00992698"},{"key":"222_CR23","unstructured":"Cornish, C.J., Watkins, H.: Learning from delayed rewards. PhD thesis, University of Cambridge (1989)"}],"container-title":["Journal of Intelligent &amp; Robotic Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10846-015-0222-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10846-015-0222-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10846-015-0222-2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,24]],"date-time":"2019-08-24T11:34:40Z","timestamp":1566646480000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10846-015-0222-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,5,2]]},"references-count":23,"journal-issue":{"issue":"3-4","published-print":{"date-parts":[[2016,3]]}},"alternative-id":["222"],"URL":"https:\/\/doi.org\/10.1007\/s10846-015-0222-2","relation":{},"ISSN":["0921-0296","1573-0409"],"issn-type":[{"value":"0921-0296","type":"print"},{"value":"1573-0409","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,5,2]]}}}