{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,10,4]],"date-time":"2023-10-04T23:55:02Z","timestamp":1696463702931},"reference-count":65,"publisher":"Informa UK Limited","issue":"6","content-domain":{"domain":["www.tandfonline.com"],"crossmark-restriction":true},"short-container-title":["Journal of Experimental &amp; Theoretical Artificial Intelligence"],"published-print":{"date-parts":[[2016,11]]},"DOI":"10.1080\/0952813x.2015.1042923","type":"journal-article","created":{"date-parts":[[2015,6,23]],"date-time":"2015-06-23T21:16:49Z","timestamp":1435094209000},"page":"913-954","update-policy":"http:\/\/dx.doi.org\/10.1080\/tandf_crossmark_01","source":"Crossref","is-referenced-by-count":1,"title":["Model-free learning on robot kinematic chains using a nested multi-agent topology"],"prefix":"10.1080","volume":"28","author":[{"given":"John N.","family":"Karigiannis","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Costas S.","family":"Tzafestas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"301","published-online":{"date-parts":[[2015,6,23]]},"reference":[{"key":"CIT0001","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"CIT0002","doi-asserted-by":"publisher","DOI":"10.1109\/70.928561"},{"key":"CIT0003","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2008.10.024"},{"key":"CIT0004","unstructured":"Ben-Israel, A. & Greville, T. N. E. (2003). Generalized inverses: Theory and applications (2nd ed). New York, NY: Springer."},{"key":"CIT0005","unstructured":"Bertsekas, D. P. & Tsitsiklis, J. N. (1996). Neuro-dynamic programming. Belmont, MA: Athena Scientific."},{"key":"CIT0006","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30301-5_60"},{"key":"CIT0007","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33486-3_15"},{"key":"CIT0008","unstructured":"Brown, G. W. (1951). Iterative solution of games by fictitious play. In T. C.\u00a0Koopmans (Ed.), Activity analysis of production and allocation, chap. XXIV (pp. 374\u2013376). New York: Wiley."},{"key":"CIT0009","doi-asserted-by":"publisher","DOI":"10.1109\/MRA.2010.936947"},{"key":"CIT0010","unstructured":"Claus, C. & Boutilier, C. (1998). The dynamics of reinforcement learning in cooperative multiagent systems. In Proceedings of the fifteenth national conference on artificial intelligence, American Association for Artificial Intelligence, Madison, WI (pp. 746\u2013752)"},{"key":"CIT0011","unstructured":"Dayan, P. & Abbott, L. F. (2001). Theoretical neuroscience, computational and mathematical modeling of neural systems. Cambridge, MA: MIT Press."},{"key":"CIT0012","doi-asserted-by":"publisher","DOI":"10.1177\/027836499701600506"},{"key":"CIT0013","unstructured":"Doya, K. (1996). Temporal difference learning in continuous time and space. In Advances in neural information processing systems (Vol.8, pp. 1073\u20131079). Cambridge, MA: MIT Press."},{"key":"CIT0015","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-013-9242-0"},{"key":"CIT0016","volume-title":"CORE Lecture Series","author":"Fudenberg D.","year":"1992"},{"key":"CIT0017","unstructured":"Glohon, M. M. & Sen, S. (2004). Learning to cooperate in multi-agent systems by combining Q-learning and evolutionary strategy. In Proceedings of the world conference on lateral computing, Dec."},{"key":"CIT0018","doi-asserted-by":"publisher","DOI":"10.1109\/FUZZY.1997.622790"},{"key":"CIT0019","unstructured":"Grefenstette, J. & Schultz, A. (1994). An evolutionary approach to learning in robots. In Proceedings of the machine learning workshop on robot learning, eleventh international conference on machine learning. Berlin: Springer."},{"key":"CIT0020","doi-asserted-by":"crossref","first-page":"1521","DOI":"10.1163\/156855307782148550","volume":"21","author":"Guenter F.","year":"2007","journal-title":"Advanced Robotics"},{"key":"CIT0021","unstructured":"Guestrin, C., Lagoudakis, M. & Parr, R. (2002). Coordinated reinforcement learning. In Proceedings of the 19th international conference on machine learning (ICML'2002)."},{"key":"CIT0022","doi-asserted-by":"publisher","DOI":"10.1109\/JRA.1987.1087111"},{"key":"CIT0023","doi-asserted-by":"publisher","DOI":"10.1007\/BF02481156"},{"key":"CIT0024","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1613\/jair.301","volume":"4","author":"Kaelbling L. P.","year":"1996","journal-title":"Journal of Artificial Intelligence Research"},{"key":"CIT0025","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.1996.506570"},{"key":"CIT0026","doi-asserted-by":"crossref","unstructured":"Karigiannis, J. N., Rekatsinas, T. & Tzafestas, C. (2010). Fuzzy rule based neurodynamic programming for mobile robot skill acquisition on the basis of a nested multi-agent architecture. In Proceedings of the IEEE\/RAS international conference on robotics and biomimetics (RO-BIO' 2010), Tianjin, China (pp. 312\u2013319)","DOI":"10.1109\/ROBIO.2010.5723346"},{"key":"CIT0027","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4419-1452-1_16"},{"key":"CIT0028","doi-asserted-by":"publisher","DOI":"10.1109\/70.508439"},{"key":"CIT0029","unstructured":"Kober, J. & Peters, J. (2009). Policy search for motor primitives in robotics. In Advances in Neural Information Processing Systems (Vol. 21, pp. 849\u2013856). Cambridge, MA: MIT Press."},{"key":"CIT0030","unstructured":"Kok, J. R. & Vlassis, N. (2004). Sparse tabular multiagent Q-learning. In Proceedings of the annual machine learning conference of Belgium and the Netherlands, Brussels."},{"key":"CIT0031","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2003.11.006"},{"key":"CIT0032","doi-asserted-by":"publisher","DOI":"10.3390\/robotics2030122"},{"key":"CIT0033","unstructured":"Lauera, M. & Riedmiller, M. (2004). Reinforcement learning for stochastic cooper-ative multi-agent systems. In Third international joint conference on Autonomous Agents and Multiagent Systems (AAMAS'04), Vol.3."},{"key":"CIT0034","unstructured":"LaValle, S. M. & Kuffner, J. J. (2001). Rapidly-exploring random trees: Progress and prospects. In B. R.\u00a0Donald, K. M.\u00a0Lynch, & D.\u00a0Rus (Eds.), Algorithmic and computational robotics: New direction (pp. 293\u2013308). Wellesley, MA: A K Peters."},{"key":"CIT0035","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4614-4803-7"},{"key":"CIT0036","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-49720-2_6"},{"key":"CIT0037","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2006.886949"},{"key":"CIT0038","unstructured":"Matsui, T., Omata, T. & Kaniyoshi, Y. (1992). Multi-agent architecture for controlling a multi-finger robot. In Proceedings of the 1992 IEEE\/RSJ international conference on intelligent robots and systems (IROS'92), Raleigh, NC."},{"key":"CIT0064","doi-asserted-by":"publisher","DOI":"10.1007\/s004220050351"},{"key":"CIT0039","unstructured":"Myerson, R. B. (1997). Game theory: Analysis of conflict. Cambridge, MA: Harvard University Press."},{"key":"CIT0040","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611970081"},{"key":"CIT0041","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2009.01.015"},{"key":"CIT0042","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2011.5980200"},{"key":"CIT0043","doi-asserted-by":"crossref","unstructured":"Rozo, L., Calinon, S., Caldwell, D. G., Jimenez, P. & Torras, C. (2013). Learning collaborative impedance-based robot behaviors. In Proceedings of the AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v27i1.8543"},{"key":"CIT0044","volume-title":"On-line Q-learning using connectionist systems\u00a0","author":"Rummery G. A.","year":"1994"},{"key":"CIT0045","doi-asserted-by":"publisher","DOI":"10.1007\/PL00014414"},{"key":"CIT0046","doi-asserted-by":"publisher","DOI":"10.1016\/S1364-6613(99)01327-3"},{"key":"CIT0047","doi-asserted-by":"publisher","DOI":"10.1016\/S0079-6123(06)65027-9"},{"key":"CIT0048","unstructured":"Schmill, M., Anderson, M. L., Fults, S., Josyula, D., Oates, T., Perlis, D. \u2026 Wrights, D. (2010). The metacognitive loop and reasoning about anomalies. In M.\u00a0Cox & A.\u00a0Raja (Eds.), The Metareasoning, Thinking about Thinking (pp. 183\u2013198). Cambridge, MA: MIT Press, chap. 12."},{"key":"CIT0049","doi-asserted-by":"publisher","DOI":"10.1109\/ICONIP.2002.1202859"},{"key":"CIT0050","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2003.1223980"},{"key":"CIT0051","unstructured":"Shibata, K. & Okabe, Y. (1994). A robot that learns an evaluation function for acquiring of appropriate motions. In World congress on neural networks, San Diego, June, International Neural Network Society Annual Meeting (Vol.2, pp. 29\u201334)"},{"key":"CIT0052","volume-title":"Smoothing-evaluation method in delayed reinforcement learning","author":"Shibata K.","year":"1995"},{"key":"CIT0053","unstructured":"Shibata, K., Sugisaka, M. & Ito, K. (2001). Fast and stable learning in direct-vision-based reinforcement learning. In Proceedings of the 6th international symposium on artificial life and robotics (AROB) (pp. 562\u2013565)"},{"key":"CIT0054","unstructured":"Shoham, Y. & Tennenholtz, M. (1992). On the synthesis of useful social laws for artificial agent societies. In Proceedings of the 1992 AAAI conference (AAAI'92) (pp. 276\u2013281)"},{"key":"CIT0055","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-84628-642-1"},{"key":"CIT0056","volume-title":"Making reinforcement learning work on real robots\u00a0","author":"Smart W. D.","year":"2002"},{"key":"CIT0057","doi-asserted-by":"publisher","DOI":"10.1109\/ICHR.2010.5686320"},{"key":"CIT0058","unstructured":"Sutton, R. S. (1996). Generalization in reinforcement learning: successful example using sparse coarse coding. In Advances in neural information processing systems (Vol.8, pp. 1038\u20131044). Cambridge, MA: MIT Press."},{"key":"CIT0059","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-3618-5"},{"key":"CIT0060","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.1985.6313399"},{"key":"CIT0061","unstructured":"Takahashi, T., Tanaka, T., Nishida, K. & Kurita, T. (2001). Self-organization of place cells and reward-based navigation for a mobile robot. In Proceedings of the international conference on neural information processing (ICONIP'2001)."},{"key":"CIT0062","volume-title":"Learning from delayed rewards\u00a0","author":"Watkins C.","year":"1989"},{"key":"CIT0014","doi-asserted-by":"publisher","DOI":"10.1115\/1.3426611"},{"key":"CIT0063","doi-asserted-by":"publisher","DOI":"10.1109\/70.976030"},{"key":"CIT0065","unstructured":"Zhang, C. & Lesser, V. (2013). Coordinating multi-agent reinforcement learning with limited communication. In Proceedings of the 2013 international conference on autonomous agents and multiagent systems (AAMAS'13), Minnesota USA."}],"container-title":["Journal of Experimental &amp; Theoretical Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/www.tandfonline.com\/doi\/pdf\/10.1080\/0952813X.2015.1042923","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,11]],"date-time":"2023-08-11T18:44:19Z","timestamp":1691779459000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.tandfonline.com\/doi\/full\/10.1080\/0952813X.2015.1042923"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,6,23]]},"references-count":65,"journal-issue":{"issue":"6","published-online":{"date-parts":[[2015,7,17]]},"published-print":{"date-parts":[[2016,11]]}},"alternative-id":["10.1080\/0952813X.2015.1042923"],"URL":"https:\/\/doi.org\/10.1080\/0952813x.2015.1042923","relation":{},"ISSN":["0952-813X","1362-3079"],"issn-type":[{"value":"0952-813X","type":"print"},{"value":"1362-3079","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,6,23]]},"assertion":[{"value":"The publishing and review policy for this title is described in its Aims & Scope.","order":1,"name":"peerreview_statement","label":"Peer Review Statement"},{"value":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=teta20","URL":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=teta20","order":2,"name":"aims_and_scope_url","label":"Aim & Scope"}]}}