{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,8,4]],"date-time":"2024-08-04T00:23:43Z","timestamp":1722731023003},"reference-count":30,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"8","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2024,8,1]]},"DOI":"10.1587\/transinf.2023edp7180","type":"journal-article","created":{"date-parts":[[2024,7,31]],"date-time":"2024-07-31T22:17:16Z","timestamp":1722464236000},"page":"1040-1049","source":"Crossref","is-referenced-by-count":0,"title":["Agent Allocation-Action Learning with Dynamic Heterogeneous Graph in Multi-Task Games"],"prefix":"10.1587","volume":"E107.D","author":[{"given":"Xianglong","family":"LI","sequence":"first","affiliation":[{"name":"Academy of Military Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan","family":"LI","sequence":"additional","affiliation":[{"name":"Academy of Military Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jieyuan","family":"ZHANG","sequence":"additional","affiliation":[{"name":"Academy of Military Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinhai","family":"XU","sequence":"additional","affiliation":[{"name":"Academy of Military Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Donghong","family":"LIU","sequence":"additional","affiliation":[{"name":"Academy of Military Sciences"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] Z. Kakish, K. Elamvazhuthi, and S. Berman, \u201cUsing reinforcement learning to herd a robotic swarm to a target distribution,\u201d International Symposium on Distributed Autonomous Robotic Systems, pp.401-414, 2022. 10.1007\/978-3-030-92790-5_31","DOI":"10.1007\/978-3-030-92790-5_31"},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] X. Tan, L. Zhou, H. Wang, Y. Sun, H. Zhao, B.-C. Seet, J. Wei, and V.C.M. Leung, \u201cCooperative multi-agent reinforcement-learning-based distributed dynamic spectrum access in cognitive radio networks,\u201d IEEE Internet Things J., vol.9, no.19, pp.19477-19488, 2022. 10.1109\/jiot.2022.3168296","DOI":"10.1109\/JIOT.2022.3168296"},{"key":"3","unstructured":"[3] Y. Liu, Y. Li, X. Xu, Y. Dou, and D. Liu, \u201cHeterogeneous skill learning for multi-agent tasks,\u201d Advances in Neural Information Processing Systems, vol.35, pp.37011-37023, 2022."},{"key":"4","doi-asserted-by":"publisher","unstructured":"[4] Y. Wang, Y. Wu, Y. Tang, Q. Li, and H. He, \u201cCooperative energy management and eco-driving of plug-in hybrid electric vehicle via multi-agent reinforcement learning,\u201d Applied Energy, vol.332, p.120563, 2023. 10.1016\/j.apenergy.2022.120563","DOI":"10.1016\/j.apenergy.2022.120563"},{"key":"5","unstructured":"[5] S. Iqbal, R. Costales, and F. Sha, \u201cAlma: Hierarchical learning for composite multi-agent tasks,\u201d arXiv preprint arXiv:2205.14205, 2022."},{"key":"6","unstructured":"[6] S. Proper and P. Tadepalli, \u201cSolving multiagent assignment markov decision processes,\u201d Proc. 8th International Conference on Autonomous Agents and Multiagent Systems, vol.1, pp.681-688, 2009."},{"key":"7","unstructured":"[7] T. Wang, T. Gupta, A. Mahajan, B. Peng, S. Whiteson, and C. Zhang, \u201cRode: Learning roles to decompose multi-agent tasks,\u201d arXiv preprint arXiv:2010.01523, 2020."},{"key":"8","unstructured":"[8] J. Yang, I. Borovikov, and H. Zha, \u201cHierarchical cooperative multi-agent reinforcement learning with skill discovery,\u201d arXiv preprint arXiv:1912.03558, 2019."},{"key":"9","unstructured":"[9] B. Liu, Q. Liu, P. Stone, A. Garg, Y. Zhu, and A. Anandkumar,\u201cCoach-player multi-agent reinforcement learning for dynamic team composition,\u201d International Conference on Machine Learning, pp.6860-6870, 2021."},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] L. Yuan, C. Wang, J. Wang, F. Zhang, F. Chen, C. Guan, Z. Zhang, C. Zhang, and Y. Yu, \u201cMulti-agent concentrative coordination with decentralized task representation,\u201d IJCAI, 2022. 10.24963\/ijcai.2022\/85","DOI":"10.24963\/ijcai.2022\/85"},{"key":"11","doi-asserted-by":"publisher","unstructured":"[11] B.P. Gerkey and M.J. Matari\u0107, \u201cA formal analysis and taxonomy of task allocation in multi-robot systems,\u201d The International Journal of Robotics Research, vol.23, no.9, pp.939-954, 2004. 10.1177\/0278364904045564","DOI":"10.1177\/0278364904045564"},{"key":"12","unstructured":"[12] N. Carion, N. Usunier, G. Synnaeve, and A. Lazaric, \u201cA structured prediction approach for generalization in cooperative multi-agent reinforcement learning,\u201d Advances in Neural Information Processing Systems, vol.32, 2019."},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] X. Li, Y. Li, J. Zhang, X. Xu, and D. Liu,\u201cA hierarchical multi-agent allocation-action learning framework for multi-subtask games,\u201d Complex &amp; Intelligent Systems, pp.1-11, 2023.","DOI":"10.1007\/s40747-023-01255-5"},{"key":"14","doi-asserted-by":"publisher","unstructured":"[14] X. Wang, D. Bo, C. Shi, S. Fan, Y. Ye, and P.S. Yu, \u201cA survey on heterogeneous graph embedding: methods, techniques, applications and sources,\u201d IEEE Trans. Big Data, vol.9, no.2, pp.415-436, 2023. 10.1109\/tbdata.2022.3177455","DOI":"10.1109\/TBDATA.2022.3177455"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] M. Chen, C. Huang, L. Xia, W. Wei, Y. Xu, and R. Luo, \u201cHeterogeneous graph contrastive learning for recommendation,\u201d Proc. Sixteenth ACM International Conference on Web Search and Data Mining, pp.544-552, 2023. 10.1145\/3539597.3570484","DOI":"10.1145\/3539597.3570484"},{"key":"16","doi-asserted-by":"publisher","unstructured":"[16] L. Gao, H. Wang, Z. Zhang, H. Zhuang, and B. Zhou, \u201cHetinf: Social influence prediction with heterogeneous graph neural network,\u201d Frontiers in Physics, vol.9, p.787185, 2022. 10.3389\/fphy.2021.787185","DOI":"10.3389\/fphy.2021.787185"},{"key":"17","doi-asserted-by":"publisher","unstructured":"[17] Z. Li, Y. Zhao, Y. Zhang, and Z. Zhang, \u201cMulti-relational graph attention networks for knowledge graph completion,\u201d Knowledge-Based Systems, vol.251, p.109262, 2022. 10.1016\/j.knosys.2022.109262","DOI":"10.1016\/j.knosys.2022.109262"},{"key":"18","doi-asserted-by":"publisher","unstructured":"[18] H.-C. Yi, Z.-H. You, D.-S. Huang, and C.K. Kwoh, \u201cGraph representation learning in bioinformatics: trends, methods and applications,\u201d Briefings in Bioinformatics, vol.23, no.1, p.bbab340, 2022. 10.1093\/bib\/bbab340","DOI":"10.1093\/bib\/bbab340"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] C. Zhang, D. Song, C. Huang, A. Swami, and N.V. Chawla, \u201cHeterogeneous graph neural network,\u201d Proc. 25th ACM SIGKDD International Conference on Knowledge Discovery &amp; Data Mining, pp.793-803, 2019.","DOI":"10.1145\/3292500.3330961"},{"key":"20","doi-asserted-by":"publisher","unstructured":"[20] X. Yang, M. Yan, S. Pan, X. Ye, and D. Fan, \u201cSimple and efficient heterogeneous graph neural network,\u201d Proc. AAAI Conference on Artificial Intelligence, vol.37, no.9, pp.10816-10824, 2023. 10.1609\/aaai.v37i9.26283","DOI":"10.1609\/aaai.v37i9.26283"},{"key":"21","unstructured":"[21] K. Son, D. Kim, W.J. Kang, D.E. Hostallero, and Y. Yi, \u201cQtran: Learning to factorize with transformation for cooperative multi-agent reinforcement learning,\u201d International Conference on Machine Learning, pp.5887-5896, 2019."},{"key":"22","unstructured":"[22] P. Sunehag, G. Lever, A. Gruslys, W.M. Czarnecki, V. Zambaldi, M. Jaderberg, M. Lanctot, N. Sonnerat, J.Z. Leibo, K. Tuyls, et al., \u201cValue-decomposition networks for cooperative multi-agent learning,\u201d arXiv preprint arXiv:1706.05296, 2017."},{"key":"23","unstructured":"[23] T. Rashid, M. Samvelyan, C. Schroeder, G. Farquhar, J. Foerster, and S. Whiteson, \u201cQmix: Monotonic value function factorisation for deep multi-agent reinforcement learning,\u201d International Conference on Machine Learning, pp.4295-4304, 2018."},{"key":"24","doi-asserted-by":"publisher","unstructured":"[24] M. Iovino, E. Scukins, J. Styrud, P. \u00d6gren, and C. Smith, \u201cA survey of behavior trees in robotics and ai,\u201d Robotics and Autonomous Systems, vol.154, p.104096, 2022. 10.1016\/j.robot.2022.104096","DOI":"10.1016\/j.robot.2022.104096"},{"key":"25","unstructured":"[25] M. Karta\u0161ev, J. Saler, and P. \u00d6gren, \u201cImproving the performance of backward chained behavior trees using reinforcement learning,\u201d arXiv preprint arXiv:2112.13744, 2021."},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] L. Li, L. Wang, Y. Li, and J. Sheng, \u201cMixed deep reinforcement learning-behavior tree for intelligent agents design,\u201d ICAART, vol.1, pp.113-124, 2021. 10.5220\/0010316901130124","DOI":"10.5220\/0010316901130124"},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] F. Rovida, B. Grossmann, and V. Kr\u00fcger, \u201cExtended behavior trees for quick definition of flexible robotic tasks,\u201d 2017 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp.6793-6800, IEEE, 2017. 10.1109\/iros.2017.8206598","DOI":"10.1109\/IROS.2017.8206598"},{"key":"28","unstructured":"[28] P. Velickovic, G. Cucurull, A. Casanova, A. Romero, P. Lio, Y. Bengio, et al., \u201cGraph attention networks,\u201d stat, vol.1050, no.20, pp.10-48550, 2017."},{"key":"29","doi-asserted-by":"publisher","unstructured":"[29] K. Kurach, A. Raichuk, P. Sta\u0144czyk, M. Zaj\u0105c, O. Bachem, L.Espeholt, C. Riquelme, D. Vincent, M. Michalski, O. Bousquet, and S. Gelly, \u201cGoogle research football: A novel reinforcement learning environment,\u201d Proc. AAAI Conference on Artificial Intelligence, vol.34, no.4, pp.4501-4510, 2020. 10.1609\/aaai.v34i04.5878","DOI":"10.1609\/aaai.v34i04.5878"},{"key":"30","unstructured":"[30] B. Liu, Q. Liu, P. Stone, A. Garg, Y. Zhu, and A. Anandkumar,\u201cCoach-player multi-agent reinforcement learning for dynamic team composition,\u201d International Conference on Machine Learning, pp.6860-6870, 2021."}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E107.D\/8\/E107.D_2023EDP7180\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,3]],"date-time":"2024-08-03T04:18:34Z","timestamp":1722658714000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E107.D\/8\/E107.D_2023EDP7180\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,1]]},"references-count":30,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2024]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2023edp7180","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2024,8,1]]},"article-number":"2023EDP7180"}}