{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T04:48:17Z","timestamp":1747975697976,"version":"3.37.3"},"reference-count":46,"publisher":"Informa UK Limited","issue":"3","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61991415"],"award-info":[{"award-number":["61991415"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006579","name":"Ministry of Industry and Information Technology","doi-asserted-by":"publisher","award":["MC-201920-X01"],"award-info":[{"award-number":["MC-201920-X01"]}],"id":[{"id":"10.13039\/501100006579","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["www.tandfonline.com"],"crossmark-restriction":true},"short-container-title":["Connection Science"],"published-print":{"date-parts":[[2021,7,3]]},"DOI":"10.1080\/09540091.2020.1832961","type":"journal-article","created":{"date-parts":[[2020,11,3]],"date-time":"2020-11-03T18:15:20Z","timestamp":1604427320000},"page":"407-426","update-policy":"https:\/\/doi.org\/10.1080\/tandf_crossmark_01","source":"Crossref","is-referenced-by-count":7,"title":["Learning adversarial policy in multiple scenes environment via multi-agent reinforcement learning"],"prefix":"10.1080","volume":"33","author":[{"given":"Yang","family":"Li","sequence":"first","affiliation":[{"name":"School of Computer Engineering and Science, Shanghai University, Shanghai, People's Republic of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinzhi","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer Engineering and Science, Shanghai University, Shanghai, People's Republic of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer Engineering and Science, Shanghai University, Shanghai, People's Republic of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Engineering and Science, Shanghai University, Shanghai, People's Republic of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianshu","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computer Engineering and Science, Shanghai University, Shanghai, People's Republic of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiangfeng","family":"Luo","sequence":"additional","affiliation":[{"name":"School of Computer Engineering and Science, Shanghai University, Shanghai, People's Republic of China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaorong","family":"Xie","sequence":"additional","affiliation":[{"name":"School of Computer Engineering and Science, Shanghai University, Shanghai, People's Republic of China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"301","published-online":{"date-parts":[[2020,11,3]]},"reference":[{"key":"CIT0001","unstructured":"Bacchiani, G., Molinari, D. & Patander, M. (2019). Microscopic traffic simulation by cooperative multi-agent deep reinforcement learning. InProceedings of the 18th international conference on autonomous agents and multiagent systems(pp. 1547\u20131555)."},{"key":"CIT0002","doi-asserted-by":"publisher","DOI":"10.1080\/09540091.2017.1350938"},{"key":"CIT0003","doi-asserted-by":"publisher","DOI":"10.1080\/09540091.2014.885268"},{"key":"CIT0004","unstructured":"Billings, D., Papp, D., Schaeffer, J. & Szafron, D. (1998). Opponent modeling in poker. InAAAI\/IAAI(Vol. 493, p. 499)."},{"key":"CIT0005","doi-asserted-by":"publisher","DOI":"10.17265\/2159-5313\/2016.09.003"},{"key":"CIT0006","doi-asserted-by":"publisher","DOI":"10.1080\/09540091.2014.885282"},{"key":"CIT0007","doi-asserted-by":"publisher","DOI":"10.1080\/09540091.2017.1405382"},{"key":"CIT0008","doi-asserted-by":"crossref","unstructured":"Foerster, J. N., Farquhar, G., Afouras, T., Nardelli, N. & Whiteson, S. (2018). Counterfactual multi-agent policy gradients. InThirty-second AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"CIT0009","unstructured":"Foerster, J., Nardelli, N., Farquhar, G., Afouras, T., Torr, P. H., Kohli, P. & Whiteson, S. (2017). Stabilising experience replay for deep multi-agent reinforcement learning. InProceedings of the 34th international conference on machine learning(Vol. 70, pp. 1146\u20131155). JMLR. org."},{"key":"CIT0010","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-71682-4_5"},{"key":"CIT0011","unstructured":"Hernandez-Leal, P., Kaisers, M., Baarslag, T. & de Cote, E. M. (2017). A survey of learning in multiagent environments: Dealing with stationarity. arXiv preprint arXiv:1707.09183."},{"key":"CIT0012","doi-asserted-by":"publisher","DOI":"10.1080\/09540091.2014.885294"},{"key":"CIT0013","unstructured":"Hu, J. & Wellman, M. P. (1998). Multiagent reinforcement learning: Theoretical framework and an algorithm. InICML(Vol. 98, pp. 242\u2013250). Citeseer."},{"key":"CIT0014","doi-asserted-by":"publisher","DOI":"10.1126\/science.aau6249"},{"key":"CIT0015","unstructured":"Juliani, A., Berges, V. P., Vckay, E., Gao, Y., Henry, H., Mattar, M. & Lange, D. (2018). Unity: A general platform for intelligent agents. arXiv preprint arXiv:1809.02627."},{"key":"CIT0016","doi-asserted-by":"publisher","DOI":"10.1080\/09540091.2019.1605498"},{"key":"CIT0017","unstructured":"Lauer, M. & Riedmiller, M. (2000). An algorithm for distributed reinforcement learning in cooperative multi-agent systems. InProceedings of the seventeenth international conference on machine learning. Citeseer."},{"key":"CIT0018","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"CIT0019","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6211"},{"key":"CIT0020","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8461113"},{"key":"CIT0021","unstructured":"Lowe, R., Wu, Y., Tamar, A., Harb, J., Abbeel, O. P. & Mordatch, I. (2017). Multi-agent actor-critic for mixed cooperative-competitive environments. InAdvances in neural information processing systems(pp. 6379\u20136390)."},{"key":"CIT0022","unstructured":"Mao, H., Zhang, Z., Xiao, Z. & Gong, Z. (2019). Modelling the dynamic joint policy of teammates with attention multi-agent DDPG. InProceedings of the 18th international conference on autonomous agents and multiagent systems(pp. 1108\u20131116). International Foundation for Autonomous Agents and Multiagent Systems."},{"key":"CIT0023","unstructured":"Matignon, L., Jeanpierre, L. & Mouaddib, A. I. (2012). Coordinated multi-robot exploration under communication constraints using decentralized markov decision processes. InTwenty-sixth AAAI conference on artificial intelligence."},{"key":"CIT0024","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2007.4399095"},{"key":"CIT0025","doi-asserted-by":"publisher","DOI":"10.1017\/S0269888912000057"},{"key":"CIT0026","doi-asserted-by":"publisher","DOI":"10.17265\/2159-5313\/2016.09.003"},{"key":"CIT0027","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2013.2275670"},{"key":"CIT0028","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.36.1.48"},{"key":"CIT0029","doi-asserted-by":"publisher","DOI":"10.1080\/09540091.2019.1604627"},{"key":"CIT0030","unstructured":"Omidshafiei, S., Pazis, J., Amato, C., How, J. P. & Vian, J. (2017). Deep decentralized multi-task multi-agent reinforcement learning under partial observability. InProceedings of the 34th international conference on machine learning(Vol. 70, pp. 2681\u20132690). JMLR. org."},{"key":"CIT0031","unstructured":"Peng, P., Yuan, Q., Wen, Y., Yang, Y., Tang, Z., Long, H. & Wang, J. (2017). Multiagent bidirectionally-coordinated nets for learning to play starcraft combat games. arXiv preprint arXiv:1703.10069 2."},{"key":"CIT0032","doi-asserted-by":"publisher","DOI":"10.1007\/s10514-014-9409-9"},{"key":"CIT0033","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6214"},{"key":"CIT0034","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2019.2922493"},{"key":"CIT0035","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A. & Klimov, O. (2017). Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347."},{"key":"CIT0036","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2974695"},{"key":"CIT0037","unstructured":"Singh, S., Kearns, M. & Mansour, Y. (2000). Nash convergence of gradient dynamics in general-sum games. InProceedings of the sixteenth conference on uncertainty in artificial intelligence(pp. 541\u2013548). Morgan Kaufmann Publishers Inc."},{"volume-title":"Reinforcement learning: An introduction","year":"2018","author":"Sutton R. S.","key":"CIT0038"},{"key":"CIT0039","unstructured":"Tao, N., Baxter, J. & Weaver, L. (2001). A multi-agent, policy-gradient approach to network routing. InProceedings of the 18th international conference on machine learning. Citeseer."},{"key":"CIT0040","unstructured":"Tesauro, G. (2004). Extending Q-learning to general adaptive multi-agent systems. InAdvances in neural information processing systems(pp. 871\u2013878)."},{"key":"CIT0041","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2974648"},{"key":"CIT0042","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"CIT0043","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"CIT0044","unstructured":"Yu, C., Wang, X., Hao, J. & Feng, Z. (2019). Reinforcement learning for cooperative overtaking. InProceedings of the 18th international conference on autonomous agents and multiagent systems(pp. 341\u2013349)."},{"key":"CIT0045","doi-asserted-by":"crossref","unstructured":"Zhang, C. & Lesser, V. (2010). Multi-agent learning with policy prediction. InTwenty-fourth AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v24i1.7639"},{"key":"CIT0046","doi-asserted-by":"publisher","DOI":"10.1109\/ICTAI.2019.00206"}],"container-title":["Connection Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.tandfonline.com\/doi\/pdf\/10.1080\/09540091.2020.1832961","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,26]],"date-time":"2022-11-26T11:56:09Z","timestamp":1669463769000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.tandfonline.com\/doi\/full\/10.1080\/09540091.2020.1832961"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,11,3]]},"references-count":46,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2021,7,3]]}},"alternative-id":["10.1080\/09540091.2020.1832961"],"URL":"https:\/\/doi.org\/10.1080\/09540091.2020.1832961","relation":{},"ISSN":["0954-0091","1360-0494"],"issn-type":[{"type":"print","value":"0954-0091"},{"type":"electronic","value":"1360-0494"}],"subject":[],"published":{"date-parts":[[2020,11,3]]},"assertion":[{"value":"The publishing and review policy for this title is described in its Aims & Scope.","order":1,"name":"peerreview_statement","label":"Peer Review Statement"},{"value":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=ccos20","URL":"http:\/\/www.tandfonline.com\/action\/journalInformation?show=aimsScope&journalCode=ccos20","order":2,"name":"aims_and_scope_url","label":"Aim & Scope"},{"value":"2020-08-06","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2020-09-20","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2020-11-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}