{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,24]],"date-time":"2025-10-24T16:50:21Z","timestamp":1761324621409,"version":"3.44.0"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2025,4,18]],"date-time":"2025-04-18T00:00:00Z","timestamp":1744934400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,4,18]],"date-time":"2025-04-18T00:00:00Z","timestamp":1744934400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["2022R1A2B5B0100261512","2022R1A2B5B0100261512"],"award-info":[{"award-number":["2022R1A2B5B0100261512","2022R1A2B5B0100261512"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"International doctoral cluster program of University of Toronto","award":["DSF20-01"],"award-info":[{"award-number":["DSF20-01"]}]},{"name":"Becas Chile Doctorado en el Extranjero of Agencia Nacional de Investigaci\u00f3n y Desarrollo","award":["72200533"],"award-info":[{"award-number":["72200533"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s13042-025-02622-z","type":"journal-article","created":{"date-parts":[[2025,4,18]],"date-time":"2025-04-18T02:45:47Z","timestamp":1744944347000},"page":"6249-6270","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Reward design in multi-agent systems using successor features and multi-information source bayesian optimization"],"prefix":"10.1007","volume":"16","author":[{"given":"Kyeonghyeon","family":"Park","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David","family":"Molina Concha","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hyun-Rok","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Taesik","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chi-Guhn","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,18]]},"reference":[{"issue":"6","key":"2622_CR1","doi-asserted-by":"publisher","first-page":"4307","DOI":"10.1007\/s10462-021-10108-x","volume":"55","author":"S Adams","year":"2022","unstructured":"Adams S, Cody T, Beling PA (2022) A survey of inverse reinforcement learning. Artif Intell Rev 55(6):4307\u20134346","journal-title":"Artif Intell Rev"},{"key":"2622_CR2","doi-asserted-by":"crossref","unstructured":"Abbeel P, Ng AY (2004) Apprenticeship learning via inverse reinforcement learning. In: Proceedings of the twenty-first international conference on machine learning, p 1","DOI":"10.1145\/1015330.1015430"},{"issue":"2","key":"2622_CR3","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1080\/08982112.2017.1315136","volume":"30","author":"B Akteke-Ozturk","year":"2018","unstructured":"Akteke-Ozturk B, Koksal G, Weber GW (2018) Nonconvex optimization of desirability functions. Qual Eng 30(2):293\u2013310","journal-title":"Qual Eng"},{"key":"2622_CR4","unstructured":"Barreto A, Dabney W, Munos R, Hunt JJ, Schaul T, Hasselt H, Silver D (2017) Successor features for transfer in reinforcement learning. In: Proceedings of the 31st international conference on neural information processing systems, pp 4058\u20134068"},{"key":"2622_CR5","unstructured":"Christoffersen PJ, Haupt AA, Hadfield-Menell D (2023) Get it in writing: formal contracts mitigate social dilemmas in multi-agent rl. In: Proceedings of the 2023 international conference on autonomous agents and multiagent systems, pp 448\u2013456"},{"issue":"17","key":"2622_CR6","doi-asserted-by":"publisher","first-page":"9724","DOI":"10.1073\/pnas.95.17.9724","volume":"95","author":"JE Cohen","year":"1998","unstructured":"Cohen JE (1998) Cooperation and self-interest: pareto-inefficiency of nash equilibria in finite random games. Proc Natl Acad Sci USA 95(17):9724\u20139731","journal-title":"Proc Natl Acad Sci USA"},{"issue":"1","key":"2622_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1287\/moor.11.1.1","volume":"11","author":"P Dubey","year":"1986","unstructured":"Dubey P (1986) Inefficiency of nash equilibria. Math Oper Res 11(1):1\u20138","journal-title":"Math Oper Res"},{"key":"2622_CR8","unstructured":"Gupta T, Mahajan A, Peng B, Boehmer W, Whiteson S (2021) Uneven: universal value exploration for multi-agent reinforcement learning. In: International conference on machine learning. PMLR, pp 3930\u20133941"},{"key":"2622_CR9","unstructured":"Guresti B, Vanlioglu A, Ure NK (2023) Iq-flow: mechanism design for inducing cooperative behavior to self-interested agents in sequential social dilemmas. In: Proceedings of the 2023 international conference on autonomous agents and multiagent systems, pp 2143\u20132151"},{"key":"2622_CR10","unstructured":"Hansen EA, Bernstein DS, Zilberstein S (2004) Dynamic programming for partially observable stochastic games. In: Proceedings of the 19th national conference on artificial intelligence, pp 709\u2013715"},{"key":"2622_CR11","unstructured":"Hostallero DE, Kim D, Moon S, Son K, Kang WJ, Yi Y (2020) Inducing cooperation through reward reshaping based on peer evaluations in deep multi-agent reinforcement learning. In: Proceedings of the 19th international conference on autonomous agents and multiagent systems, pp 520\u2013528"},{"key":"2622_CR12","unstructured":"Hughes E, Leibo JZ, Phillips M, Tuyls K, Due\u00f1ez-Guzman E, Casta\u00f1eda AG, Dunning I, Zhu T, McKee K, Koster R, et al. (2018) Inequity aversion improves cooperation in intertemporal social dilemmas. In: Proceedings of the 32nd international conference on neural information processing systems, pp 3330\u20133340"},{"key":"2622_CR13","first-page":"1039","volume":"4","author":"J Hu","year":"2003","unstructured":"Hu J, Wellman MP (2003) Nash q-learning for general-sum stochastic games. J Mach Learn Res 4:1039\u20131069","journal-title":"J Mach Learn Res"},{"key":"2622_CR14","unstructured":"Jaques N, Lazaridou A, Hughes E, Gulcehre C, Ortega P, Strouse D, Leibo JZ, De\u00a0Freitas N (2019) Social influence as intrinsic motivation for multi-agent deep reinforcement learning. In: International conference on machine learning. PMLR, pp 3040\u20133049"},{"key":"2622_CR15","unstructured":"Kim SH, Van\u00a0Stralen N, Chowdhary G, Tran HT (2022) Disentangling successor features for coordination in multi-agent reinforcement learning. In: Proceedings of the 21st international conference on autonomous agents and multiagent systems, pp 751\u2013760"},{"issue":"9","key":"2622_CR16","doi-asserted-by":"publisher","first-page":"1673","DOI":"10.1109\/JAS.2022.105809","volume":"9","author":"W Liu","year":"2022","unstructured":"Liu W, Dong L, Niu D, Sun C (2022) Efficient exploration for multi-agent reinforcement learning via transferable successor features. IEEE\/CAA J Autom Sin 9(9):1673\u20131686","journal-title":"IEEE\/CAA J Autom Sin"},{"key":"2622_CR17","doi-asserted-by":"publisher","first-page":"200","DOI":"10.1016\/j.trc.2019.01.020","volume":"100","author":"L Lehe","year":"2019","unstructured":"Lehe L (2019) Downtown congestion pricing in practice. Transport Res Part C Emerg Technol 100:200\u2013223","journal-title":"Transport Res Part C Emerg Technol"},{"key":"2622_CR18","doi-asserted-by":"crossref","unstructured":"Littman ML (1994) Markov games as a framework for multi-agent reinforcement learning. In: Proceedings of the eleventh international conference on international conference on machine learning, pp 157\u2013163","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"2622_CR19","unstructured":"Li J, Yu J, Nie Y, Wang Z (2020) End-to-end learning and intervention in games. In: Proceedings of the 34th international conference on neural information processing systems, pp 16653\u201316665"},{"key":"2622_CR20","unstructured":"Leibo JZ, Zambaldi V, Lanctot M, Marecki J, Graepel T (2017) Multi-agent reinforcement learning in sequential social dilemmas. In: Proceedings of the 16th conference on autonomous agents and multiagent systems, pp 464\u2013473"},{"key":"2622_CR21","unstructured":"Molina\u00a0Concha D, Li J, Yin H, Park K, Lee H-R, Lee T, Sirohi D, Lee C-G (2024) Bayesian optimization framework for efficient fleet design in autonomous multi-robot exploration. arXiv preprint arXiv:2408.11751"},{"key":"2622_CR22","unstructured":"Molina\u00a0Concha D, Park K, Lee H-R, Lee T, Lee C-G (2024) Algorithmic contract design with reinforcement learning agents. arXiv preprint arXiv:2408.09686"},{"key":"2622_CR23","doi-asserted-by":"crossref","unstructured":"Mguni D, Jennings J, Cote EM (2018) Decentralised learning in systems with many, many strategic agents. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.11586"},{"key":"2622_CR24","unstructured":"Mguni D, Jennings J, Sison E, Valcarcel\u00a0Macua S, Ceppi S, Cote E (2019) Coordinating the crowd: inducing desirable equilibria in non-cooperative systems. In: Proceedings of the 18th international conference on autonomous agents and multiagent systems, pp 386\u2013394"},{"issue":"7540","key":"2622_CR25","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"issue":"5","key":"2622_CR26","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1527\/tjsai.36-5_AG21-H","volume":"36","author":"N Matsunami","year":"2021","unstructured":"Matsunami N, Okuhara S, Ito T (2021) Reward design for multi-agent reinforcement learning with a penalty based on the payment mechanism. Trans Jpn Soc Artif Intell 36(5):21\u20131","journal-title":"Trans Jpn Soc Artif Intell"},{"key":"2622_CR27","unstructured":"Metelli AM, Pirotta M, Restelli M (2017) Compatible reward inverse reinforcement learning. In: Proceedings of the 31st international conference on neural information processing systems, pp 2047\u20132056"},{"key":"2622_CR28","unstructured":"Ng AY, Russell SJ (2000) Algorithms for inverse reinforcement learning. In: Proceedings of the seventeenth international conference on machine learning, pp 663\u2013670"},{"issue":"1\u20132","key":"2622_CR29","doi-asserted-by":"publisher","first-page":"166","DOI":"10.1006\/game.1999.0790","volume":"35","author":"N Nisan","year":"2001","unstructured":"Nisan N, Ronen A (2001) Algorithmic mechanism design. Games Econ Behav 35(1\u20132):166\u2013196","journal-title":"Games Econ Behav"},{"key":"2622_CR30","unstructured":"Poloczek M, Wang J, Frazier PI (2017) Multi-information source optimization. In: Proceedings of the 31st international conference on neural information processing systems, pp 4291\u20134301"},{"key":"2622_CR31","doi-asserted-by":"publisher","DOI":"10.1016\/j.trc.2020.102738","volume":"119","author":"Z Shou","year":"2020","unstructured":"Shou Z, Di X (2020) Reward design for driver repositioning using multi-agent reinforcement learning. Transport Res Part C Emerg Technol 119:102738","journal-title":"Transport Res Part C Emerg Technol"},{"issue":"8","key":"2622_CR32","doi-asserted-by":"publisher","first-page":"1627","DOI":"10.1021\/ac60214a047","volume":"36","author":"A Savitzky","year":"1964","unstructured":"Savitzky A, Golay MJ (1964) Smoothing and differentiation of data by simplified least squares procedures. Anal Chem 36(8):1627\u20131639","journal-title":"Anal Chem"},{"key":"2622_CR33","unstructured":"Srinivas N, Krause A, Kakade S, Seeger M (2010) Gaussian process optimization in the bandit setting: no regret and experimental design. In: Proceedings of the 27th international conference on international conference on machine learning, pp 1015\u20131022"},{"key":"2622_CR34","unstructured":"Shoham Y, Powers R, Grenager T (2003) Multi-agent reinforcement learning: a critical survey. Technical report, Stanford University"},{"issue":"4","key":"2622_CR35","doi-asserted-by":"publisher","first-page":"1019","DOI":"10.1287\/trsc.2022.1188","volume":"57","author":"J Xie","year":"2023","unstructured":"Xie J, Liu Y, Chen N (2023) Two-sided deep reinforcement learning for dynamic mobility-on-demand management with mixed autonomy. Transport Sci 57(4), 1019\u20131046","journal-title":"Transport Sci"},{"issue":"16","key":"2622_CR36","doi-asserted-by":"publisher","first-page":"8004","DOI":"10.3390\/app12168004","volume":"12","author":"Y Yuan","year":"2022","unstructured":"Yuan Y, Guo T, Zhao P, Jiang H (2022) Adherence improves cooperation in sequential social dilemmas. Appl Sci 12(16):8004","journal-title":"Appl Sci"},{"key":"2622_CR37","unstructured":"Yang Y, Luo R, Li M, Zhou M, Zhang W, Wang J (2018) Mean field multi-agent reinforcement learning. In: International conference on machine learning. PMLR, pp 5571\u20135580"},{"key":"2622_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.jth.2019.100586","volume":"14","author":"W Yu","year":"2019","unstructured":"Yu W, Suh D, Song S, Jiao B, Zhang L, Muennig P (2019) The cost-effectiveness of competing congestion pricing plans in New York city. J Transp Health 14:100586","journal-title":"J Transp Health"},{"key":"2622_CR39","unstructured":"Yang J, Wang E, Trivedi R, Zhao T, Zha H (2022) Adaptive incentive design with multi-agent meta-gradient reinforcement learning. In: Proceedings of the 21st international conference on autonomous agents and multiagent systems, pp 1436\u20131445"},{"key":"2622_CR40","doi-asserted-by":"crossref","unstructured":"Zhang K, Yang Z, Liu H, Zhang T, Basar T (2018) Fully decentralized multi-agent reinforcement learning with networked agents. In: International conference on machine learning. PMLR, pp 5872\u20135881","DOI":"10.1109\/CDC.2018.8619581"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-025-02622-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-025-02622-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-025-02622-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T11:00:45Z","timestamp":1757156445000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-025-02622-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,18]]},"references-count":40,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["2622"],"URL":"https:\/\/doi.org\/10.1007\/s13042-025-02622-z","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"type":"print","value":"1868-8071"},{"type":"electronic","value":"1868-808X"}],"subject":[],"published":{"date-parts":[[2025,4,18]]},"assertion":[{"value":"11 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 March 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 April 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"No ethics approval was required for this work as it did not involve human subjects, animals, or sensitive data that would necessitate ethical review.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}]}}