{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T13:15:00Z","timestamp":1783602900819,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":25,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819533428","type":"print"},{"value":"9789819533435","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,23]],"date-time":"2025-11-23T00:00:00Z","timestamp":1763856000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,23]],"date-time":"2025-11-23T00:00:00Z","timestamp":1763856000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-3343-5_37","type":"book-chapter","created":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T06:31:14Z","timestamp":1763793074000},"page":"479-491","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Large Language Model Based Multi-agent Learning for\u00a0Mixed Cooperative-Competitive Environments"],"prefix":"10.1007","author":[{"given":"Chenghua","family":"He","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiyue","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongzhe","family":"Chang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tongtong","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,11,23]]},"reference":[{"key":"37_CR1","unstructured":"Achiam, J., et\u00a0al.: Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"37_CR2","unstructured":"Agashe, S., Fan, Y., Wang, X.E.: Evaluating multi-agent coordination abilities in large language models. arXiv preprint arXiv:2310.03903 (2023)"},{"key":"37_CR3","doi-asserted-by":"crossref","unstructured":"Bradley, R.A., Terry, M.E.: Rank analysis of incomplete block designs: I. the method of paired comparisons. Biometrika 39(3\/4), 324\u2013345 (1952)","DOI":"10.1093\/biomet\/39.3-4.324"},{"key":"37_CR4","unstructured":"Burns, C., et\u00a0al.: Weak-to-strong generalization: eliciting strong capabilities with weak supervision. arXiv preprint arXiv:2312.09390 (2023)"},{"key":"37_CR5","unstructured":"Chen, Z., Deng, Y., Yuan, H., Ji, K., Gu, Q.: Self-play fine-tuning converts weak language models to strong language models. arXiv preprint arXiv:2401.01335 (2024)"},{"key":"37_CR6","doi-asserted-by":"crossref","unstructured":"Cheng, P., et al.: Self-playing adversarial language game enhances llm reasoning. arXiv preprint arXiv:2404.10642 (2024)","DOI":"10.52202\/079017-4019"},{"key":"37_CR7","unstructured":"Dubey, A., et\u00a0al.: The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)"},{"key":"37_CR8","doi-asserted-by":"crossref","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with pagedattention. In: Proceedings of the 29th Symposium on Operating Systems Principles, pp. 611\u2013626 (2023)","DOI":"10.1145\/3600006.3613165"},{"key":"37_CR9","unstructured":"Lowe, R., Wu, Y.I., Tamar, A., Harb, J., Pieter\u00a0Abbeel, O., Mordatch, I.: Multi-agent actor-critic for mixed cooperative-competitive environments. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"37_CR10","doi-asserted-by":"crossref","unstructured":"Ma, W., et al.: Large language models play starcraft ii: benchmarks and a chain of summarization approach. arXiv preprint arXiv:2312.11865 (2023)","DOI":"10.52202\/079017-4240"},{"key":"37_CR11","unstructured":"Mao, S., et al.: Alympics: language agents meet game theory. arXiv preprint arXiv:2311.03220 (2023)"},{"key":"37_CR12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-28929-8","volume-title":"A Concise Introduction to Decentralized POMDPs","author":"FA Oliehoek","year":"2016","unstructured":"Oliehoek, F.A., Amato, C., et al.: A Concise Introduction to Decentralized POMDPs, vol. 1. Springer, Cham (2016)"},{"key":"37_CR13","doi-asserted-by":"publisher","first-page":"27730","DOI":"10.52202\/068431-2011","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. Adv. Neural. Inf. Process. Syst. 35, 27730\u201327744 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"37_CR14","unstructured":"Putta, P., et al.: Agent q: advanced reasoning and learning for autonomous AI agents. arXiv preprint arXiv:2408.07199 (2024)"},{"key":"37_CR15","doi-asserted-by":"crossref","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Manning, C.D., Ermon, S., Finn, C.: Direct preference optimization: your language model is secretly a reward model. In: Advances in Neural Information Processing Systems, vol. 36 (2024)","DOI":"10.52202\/075280-2338"},{"key":"37_CR16","doi-asserted-by":"crossref","unstructured":"Song, Y., Yin, D., Yue, X., Huang, J., Li, S., Lin, B.Y.: Trial and error: exploration-based trajectory optimization for LLM agents. arXiv preprint arXiv:2403.02502 (2024)","DOI":"10.18653\/v1\/2024.acl-long.409"},{"key":"37_CR17","unstructured":"Touvron, H., et\u00a0al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"37_CR18","doi-asserted-by":"publisher","first-page":"24824","DOI":"10.52202\/068431-1800","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. Adv. Neural. Inf. Process. Syst. 35, 24824\u201324837 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"37_CR19","doi-asserted-by":"crossref","unstructured":"Xiong, W., et al.: Watch every step! llm agent learning via iterative step-level process refinement. arXiv preprint arXiv:2406.11176 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.93"},{"key":"37_CR20","unstructured":"Yang, A., et\u00a0al.: Qwen2 technical report. arXiv preprint arXiv:2407.10671 (2024)"},{"key":"37_CR21","doi-asserted-by":"crossref","unstructured":"Ye, T., Xu, Z., Li, Y., Allen-Zhu, Z.: Physics of language models: part 2.2, how to learn from mistakes on grade-school math problems (2024). https:\/\/arxiv.org\/abs\/2408.16293","DOI":"10.2139\/ssrn.5250631"},{"key":"37_CR22","doi-asserted-by":"crossref","unstructured":"Yin, D., et al.: Agent lumos: unified and modular training for open-source language agents. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 12380\u201312403 (2024)","DOI":"10.18653\/v1\/2024.acl-long.670"},{"key":"37_CR23","doi-asserted-by":"publisher","first-page":"24611","DOI":"10.52202\/068431-1787","volume":"35","author":"C Yu","year":"2022","unstructured":"Yu, C., et al.: The surprising effectiveness of PPO in cooperative multi-agent games. Adv. Neural. Inf. Process. Syst. 35, 24611\u201324624 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"37_CR24","doi-asserted-by":"crossref","unstructured":"Zeng, A., et al.: Agenttuning: Enabling generalized agent abilities for llms. arXiv preprint arXiv:2310.12823 (2023)","DOI":"10.18653\/v1\/2024.findings-acl.181"},{"key":"37_CR25","doi-asserted-by":"crossref","unstructured":"Zhang, C., et\u00a0al.: Proagent: building proactive cooperative agents with large language models. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 17591\u201317599 (2024)","DOI":"10.1609\/aaai.v38i16.29710"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Chinese Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-3343-5_37","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T12:29:46Z","timestamp":1783600186000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-3343-5_37"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,23]]},"ISBN":["9789819533428","9789819533435"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-3343-5_37","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,23]]},"assertion":[{"value":"23 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"NLPCC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"CCF International Conference on Natural Language Processing and Chinese Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nlpcc2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/tcci.ccf.org.cn\/conference\/2025\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}