{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T23:16:25Z","timestamp":1743030985095,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":54,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789811992964"},{"type":"electronic","value":"9789811992971"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-981-19-9297-1_1","type":"book-chapter","created":{"date-parts":[[2023,1,19]],"date-time":"2023-01-19T10:33:02Z","timestamp":1674124382000},"page":"3-16","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Heterogeneous Multi-unit Control with\u00a0Curriculum Learning for\u00a0Multi-agent Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Jiali","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kai","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rupeng","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaoqiu","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ying","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,1,20]]},"reference":[{"key":"1_CR1","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: Proceedings of the 26th Annual International Conference On Machine Learning, pp. 41\u201348 (2009)","DOI":"10.1145\/1553374.1553380"},{"key":"1_CR2","doi-asserted-by":"crossref","unstructured":"Bravo, M., Alvarado, M.: On the pragmatic similarity between agent communication protocols: modeling and measuring. In: On the Move to Meaningful Internet Systems: OTM, pp. 128\u2013137 (2008)","DOI":"10.1007\/978-3-540-88875-8_32"},{"key":"1_CR3","doi-asserted-by":"crossref","unstructured":"Bravo, M., Reyes-Ortiz, J.A., Rodr\u00edguez, J., Silva-L\u00f3pez, B.: Multi-agent communication heterogeneity. In: 2015 International Conference on Computational Science and Computational Intelligence (CSCI), pp. 583\u2013588. IEEE (2015)","DOI":"10.1109\/CSCI.2015.167"},{"key":"1_CR4","unstructured":"Burda, Y., Edwards, H., Storkey, A., Klimov, O.: Exploration by random network distillation. arXiv preprint arXiv:1810.12894 (2018)"},{"key":"1_CR5","unstructured":"Calvo, J.A., Dusparic, I.: Heterogeneous multi-agent deep reinforcement learning for traffic lights control. In: AICS, pp. 2\u201313 (2018)"},{"issue":"4","key":"1_CR6","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1109\/MCI.2006.329691","volume":"1","author":"M Dorigo","year":"2006","unstructured":"Dorigo, M., Birattari, M., Stutzle, T.: Ant colony optimization. IEEE Comput. Intell. Mag. 1(4), 28\u201339 (2006)","journal-title":"IEEE Comput. Intell. Mag."},{"key":"1_CR7","doi-asserted-by":"crossref","unstructured":"Foerster, J., Farquhar, G., Afouras, T., Nardelli, N., Whiteson, S.: Counterfactual multi-agent policy gradients. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.11794"},{"key":"1_CR8","unstructured":"Fournier, P., Sigaud, O., Chetouani, M., Oudeyer, P.Y.: Accuracy-based curriculum learning in deep reinforcement learning. arXiv preprint arXiv:1806.09614 (2018)"},{"key":"1_CR9","unstructured":"Henaff, M., Bruna, J., LeCun, Y.: Deep convolutional networks on graph-structured data. arXiv preprint arXiv:1506.05163 (2015)"},{"issue":"12","key":"1_CR10","doi-asserted-by":"publisher","first-page":"2719","DOI":"10.1109\/TCYB.2015.2487318","volume":"46","author":"W Hu","year":"2015","unstructured":"Hu, W., Tan, Y.: Prototype generation using multiobjective particle swarm optimization for nearest neighbor classification. IEEE Trans. Cybern. 46(12), 2719\u20132731 (2015)","journal-title":"IEEE Trans. Cybern."},{"key":"1_CR11","doi-asserted-by":"crossref","unstructured":"Ivanovic, B., Harrison, J., Sharma, A., Chen, M., Pavone, M.: BARC: backward reachability curriculum for robotic reinforcement learning. In: 2019 International Conference on Robotics and Automation (ICRA), pp. 15\u201321. IEEE (2019)","DOI":"10.1109\/ICRA.2019.8794206"},{"key":"1_CR12","unstructured":"Jabri, A., Hsu, K., Gupta, A., Eysenbach, B., Levine, S., Finn, C.: Unsupervised curricula for visual meta-reinforcement learning. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"issue":"3\u20134","key":"1_CR13","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1561\/2200000058","volume":"10","author":"P Jain","year":"2017","unstructured":"Jain, P., Kar, P., et al.: Non-convex optimization for machine learning. Found. Trends Mach. Learn. 10(3\u20134), 142\u2013363 (2017)","journal-title":"Found. Trends Mach. Learn."},{"key":"1_CR14","unstructured":"Jiang, J., Lu, Z.: Offline decentralized multi-agent reinforcement learning. arXiv preprint arXiv:2108.01832 (2021)"},{"key":"1_CR15","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1007\/978-3-540-32274-0_8","volume-title":"Adaptive Agents and Multi-Agent Systems II","author":"S Kapetanakis","year":"2005","unstructured":"Kapetanakis, S., Kudenko, D.: Reinforcement learning of coordination in heterogeneous cooperative multi-agent systems. In: Kudenko, D., Kazakov, D., Alonso, E. (eds.) AAMAS 2003-2004. LNCS (LNAI), vol. 3394, pp. 119\u2013131. Springer, Heidelberg (2005). https:\/\/doi.org\/10.1007\/978-3-540-32274-0_8"},{"key":"1_CR16","doi-asserted-by":"crossref","unstructured":"Kennedy, J., Eberhart, R.: Particle swarm optimization. In: Proceedings of ICNN 1995-International Conference on Neural Networks, vol. 4, pp. 1942\u20131948. IEEE (1995)","DOI":"10.1109\/ICNN.1995.488968"},{"key":"1_CR17","unstructured":"K\u00f6ster, R.: Model-free conventions in multi-agent reinforcement learning with heterogeneous preferences. arXiv preprint arXiv:2010.09054 (2020)"},{"key":"1_CR18","unstructured":"Lair, N., Colas, C., Portelas, R., Dussoux, J.M., Dominey, P.F., Oudeyer, P.Y.: Language grounding through social interactions and curiosity-driven multi-goal learning. arXiv preprint arXiv:1911.03219 (2019)"},{"key":"1_CR19","unstructured":"Li, H., He, H.: Multi-agent trust region policy optimization. arXiv preprint arXiv:2010.07916 (2020)"},{"issue":"6","key":"1_CR20","doi-asserted-by":"publisher","first-page":"627","DOI":"10.1080\/00207720902755762","volume":"40","author":"CL Liu","year":"2009","unstructured":"Liu, C.L., Tian, Y.P.: Formation control of multi-agent systems with heterogeneous communication delays. Int. J. Syst. Sci. 40(6), 627\u2013636 (2009)","journal-title":"Int. J. Syst. Sci."},{"key":"1_CR21","doi-asserted-by":"crossref","unstructured":"Liu, L., Zheng, S., Tan, Y.: S-metric based multi-objective fireworks algorithm. In: 2015 IEEE Congress on Evolutionary Computation (CEC), pp. 1257\u20131264. IEEE (2015)","DOI":"10.1109\/CEC.2015.7257033"},{"key":"1_CR22","unstructured":"Lowe, R., Wu, Y.I., Tamar, A., Harb, J., Pieter Abbeel, O., Mordatch, I.: Multi-agent actor-critic for mixed cooperative-competitive environments. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"1_CR23","doi-asserted-by":"crossref","unstructured":"Meneghetti, D.D.R., Bianchi, R.A.C.: Towards heterogeneous multi-agent reinforcement learning with graph neural networks. arXiv preprint arXiv:2009.13161 (2020)","DOI":"10.5753\/eniac.2020.12161"},{"key":"1_CR24","doi-asserted-by":"crossref","unstructured":"Meneghetti, D.D.R., da Costa Bianchi, R.A.: Specializing inter-agent communication in heterogeneous multi-agent reinforcement learning using agent class information. arXiv:abs\/2012.07617 (2020)","DOI":"10.5753\/eniac.2020.12161"},{"issue":"1","key":"1_CR25","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1109\/JRPROC.1961.287775","volume":"49","author":"M Minsky","year":"1961","unstructured":"Minsky, M.: Steps toward artificial intelligence. Proc. IRE 49(1), 8\u201330 (1961)","journal-title":"Proc. IRE"},{"issue":"1","key":"1_CR26","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1073\/pnas.36.1.48","volume":"36","author":"JF Nash Jr","year":"1950","unstructured":"Nash, J.F., Jr.: Equilibrium points in n-person games. Proc. Natl. Acad. Sci. 36(1), 48\u201349 (1950)","journal-title":"Proc. Natl. Acad. Sci."},{"key":"1_CR27","doi-asserted-by":"publisher","first-page":"289","DOI":"10.1613\/jair.2447","volume":"32","author":"FA Oliehoek","year":"2008","unstructured":"Oliehoek, F.A., Spaan, M.T., Vlassis, N.: Optimal and approximate q-value functions for decentralized Pomdps. J. Artif. Intell. Res. 32, 289\u2013353 (2008)","journal-title":"J. Artif. Intell. Res."},{"key":"1_CR28","doi-asserted-by":"crossref","unstructured":"Portelas, R., Colas, C., Weng, L., Hofmann, K., Oudeyer, P.Y.: Automatic curriculum learning for deep RL: a short survey. arXiv preprint arXiv:2003.04664 (2020)","DOI":"10.24963\/ijcai.2020\/671"},{"key":"1_CR29","unstructured":"Price, B., Boutilier, C.: Reinforcement learning with imitation in heterogeneous multi-agent systems"},{"key":"1_CR30","unstructured":"Racaniere, S., Lampinen, A.K., Santoro, A., Reichert, D.P., Firoiu, V., Lillicrap, T.P.: Automated curricula through setter-solver interactions. arXiv preprint arXiv:1909.12892 (2019)"},{"key":"1_CR31","first-page":"10199","volume":"33","author":"T Rashid","year":"2020","unstructured":"Rashid, T., Farquhar, G., Peng, B., Whiteson, S.: Weighted QMIX: expanding monotonic value function factorisation for deep multi-agent reinforcement learning. Adv. Neural. Inf. Process. Syst. 33, 10199\u201310210 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1_CR32","unstructured":"Rashid, T., Samvelyan, M., Schroeder, C., Farquhar, G., Foerster, J., Whiteson, S.: QMIX: monotonic value function factorisation for deep multi-agent reinforcement learning. In: International Conference On Machine Learning, pp. 4295\u20134304. PMLR (2018)"},{"key":"1_CR33","unstructured":"Samvelyan, M., et al.: The StarCraft Multi-Agent Challenge. CoRR abs\/1902.04043 (2019)"},{"key":"1_CR34","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., Moritz, P.: Trust region policy optimization. In: International Conference on Machine Learning, pp. 1889\u20131897. PMLR (2015)"},{"key":"1_CR35","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"1_CR36","unstructured":"Son, K., Kim, D., Kang, W.J., Hostallero, D.E., Yi, Y.: QTRAN: learning to factorize with transformation for cooperative multi-agent reinforcement learning. In: International Conference on Machine Learning, pp. 5887\u20135896. PMLR (2019)"},{"key":"1_CR37","unstructured":"Su, K., Zhou, S., Gan, C., Wang, X., Lu, Z.: Ma2QL: a minimalist approach to fully decentralized multi-agent reinforcement learning. arXiv preprint arXiv:2209.08244 (2022)"},{"key":"1_CR38","unstructured":"Sunehag, P., et al.: Value-decomposition networks for cooperative multi-agent learning. arXiv preprint arXiv:1706.05296 (2017)"},{"key":"1_CR39","doi-asserted-by":"crossref","unstructured":"Tan, M.: Multi-agent reinforcement learning: independent vs. cooperative agents. In: Proceedings of the Tenth International Conference on Machine Learning, pp. 330\u2013337 (1993)","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"1_CR40","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"355","DOI":"10.1007\/978-3-642-13495-1_44","volume-title":"Advances in Swarm Intelligence","author":"Y Tan","year":"2010","unstructured":"Tan, Y., Zhu, Y.: Fireworks algorithm for optimization. In: Tan, Y., Shi, Y., Tan, K.C. (eds.) ICSI 2010. LNCS, vol. 6145, pp. 355\u2013364. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-13495-1_44"},{"key":"1_CR41","unstructured":"Terry, J.K., Grammel, N., Hari, A., Santos, L.: Parameter sharing is surprisingly useful for multi-agent deep reinforcement learning (2020)"},{"key":"1_CR42","unstructured":"Terry, J.K., Grammel, N., Hari, A., Santos, L., Black, B.: Revisiting parameter sharing in multi-agent deep reinforcement learning. arXiv preprint arXiv:2005.13625 (2020)"},{"key":"1_CR43","unstructured":"Terry, J.K., Grammel, N., Son, S., Black, B.: Parameter sharing for heterogeneous agents in multi-agent reinforcement learning. arXiv:abs\/2005.13625 (2020)"},{"issue":"7782","key":"1_CR44","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals, O., et al.: Grandmaster level in StarCraft ii using multi-agent reinforcement learning. Nature 575(7782), 350\u2013354 (2019)","journal-title":"Nature"},{"key":"1_CR45","unstructured":"Vinyals, O., et al.: StarCraft ii: A new challenge for reinforcement learning. arXiv preprint arXiv:1708.04782 (2017)"},{"key":"1_CR46","unstructured":"Wang, J., Ren, Z., Liu, T., Yu, Y., Zhang, C.: Qplex: duplex dueling multi-agent q-learning. arXiv preprint arXiv:2008.01062 (2020)"},{"key":"1_CR47","unstructured":"Wang, T., Dong, H., Lesser, V., Zhang, C.: Roma: Multi-agent reinforcement learning with emergent roles. arXiv preprint arXiv:2003.08039 (2020)"},{"key":"1_CR48","unstructured":"Wang, T., Gupta, T., Mahajan, A., Peng, B., Whiteson, S., Zhang, C.: Rode: learning roles to decompose multi-agent tasks. arXiv preprint arXiv:2010.01523 (2020)"},{"key":"1_CR49","unstructured":"de Witt, C.S., et al.: Is independent learning all you need in the StarCraft multi-agent challenge? arXiv preprint arXiv:2011.09533 (2020)"},{"key":"1_CR50","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1016\/j.neunet.2021.03.015","volume":"139","author":"S Yang","year":"2021","unstructured":"Yang, S., Yang, B., Kang, Z., Deng, L.: IHG-MA: Inductive heterogeneous graph multi-agent reinforcement learning for multi-intersection traffic signal control. Neural Netw. 139, 265\u2013277 (2021). https:\/\/doi.org\/10.1016\/j.neunet.2021.03.015","journal-title":"Neural Netw."},{"key":"1_CR51","unstructured":"Yu, C., Velu, A., Vinitsky, E., Wang, Y., Bayen, A., Wu, Y.: The surprising effectiveness of ppo in cooperative, multi-agent games. arXiv preprint arXiv:2103.01955 (2021)"},{"key":"1_CR52","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Tan, Y.: Group explosion strategy for searching multiple targets using swarm robotic. In: 2013 IEEE Congress on Evolutionary Computation, pp. 821\u2013828. IEEE (2013)","DOI":"10.1109\/CEC.2013.6557653"},{"key":"1_CR53","unstructured":"Zhou, J., Cui, G., Zhang, Z., Yang, C., Liu, Z., Sun, M.: Graph neural networks: a review of methods and applications. arXiv:abs\/1812.08434 (2020)"},{"issue":"A11","key":"1_CR54","first-page":"125","volume":"7","author":"Y Zhou","year":"2011","unstructured":"Zhou, Y., Tan, Y.: GPU-based parallel multi-objective particle swarm optimization. Int. J. Artif. Intell. 7(A11), 125\u2013141 (2011)","journal-title":"Int. J. Artif. Intell."}],"container-title":["Communications in Computer and Information Science","Data Mining and Big Data"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-19-9297-1_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,12]],"date-time":"2024-10-12T15:30:52Z","timestamp":1728747052000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-19-9297-1_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9789811992964","9789811992971"],"references-count":54,"URL":"https:\/\/doi.org\/10.1007\/978-981-19-9297-1_1","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"20 January 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DMBD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Data Mining and Big Data","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Beijing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 November 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 November 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dmbd2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"135","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"62","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"46% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.8","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2-3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}