{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T09:51:20Z","timestamp":1766137880264,"version":"3.37.3"},"reference-count":26,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,1,3]],"date-time":"2022-01-03T00:00:00Z","timestamp":1641168000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,1,3]],"date-time":"2022-01-03T00:00:00Z","timestamp":1641168000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2020AAA0107200"],"award-info":[{"award-number":["2020AAA0107200"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["The National Natural Science Foundation of China"],"award-info":[{"award-number":["The National Natural Science Foundation of China"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2022,3]]},"DOI":"10.1007\/s10994-021-06083-7","type":"journal-article","created":{"date-parts":[[2022,1,3]],"date-time":"2022-01-03T00:03:15Z","timestamp":1641168195000},"page":"977-995","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Improve generated adversarial imitation learning with reward variance regularization"],"prefix":"10.1007","volume":"111","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8834-4921","authenticated-orcid":false,"given":"Yi-Feng","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fan-Ming","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,1,3]]},"reference":[{"key":"6083_CR1","doi-asserted-by":"crossref","unstructured":"Abbeel, P., & Ng, A. Y. (2004). Apprenticeship learning via inverse reinforcement learning. In ICML 2004, Alberta, Canada.","DOI":"10.1145\/1015330.1015430"},{"key":"6083_CR2","unstructured":"Baram, N., Anschel, O., Caspi, I., & Mannor, S. (2017). End-to-end differentiable adversarial imitation learning. In ICML 2017, Sydney, Australia (pp. 390\u2013399)."},{"key":"6083_CR3","doi-asserted-by":"crossref","unstructured":"Bhattacharyya, R. P., Phillips, D. J., Wulfe, B., Morton, J., Kuefler, A., & Kochenderfer, M. J. (2018). Multi-agent imitation learning for driving simulation. In IROS 2018, Madrid, Spain (pp. 1534\u20131539).","DOI":"10.1109\/IROS.2018.8593758"},{"key":"6083_CR4","unstructured":"Chen, M., Wang, Y., Liu, T., Yang, Z., Li, X., Wang, Z., & Zhao, T. (2020). On computation and generalization of generative adversarial imitation learning. CoRR arXiv:2001.02792."},{"key":"6083_CR5","unstructured":"Finn, C., Christiano, P. F., Abbeel, P., & Levine, S. (2016a). A connection between generative adversarial networks, inverse reinforcement learning, and energy-based models. CoRR arXiv:1611.03852."},{"key":"6083_CR6","unstructured":"Finn, C., Levine, S., & Abbeel, P. (2016b). Guided cost learning: Deep inverse optimal control via policy optimization. In ICML 2016, New York City, NY (pp. 49\u201358)."},{"key":"6083_CR7","unstructured":"Fu, J., Luo, K., & Levine, S. (2017). Learning robust rewards with adversarial inverse reinforcement learning. CoRR arXiv:1710.11248."},{"key":"6083_CR8","unstructured":"Geng, S., Nassif, H., Manzanares, C. A., Reppen, A. M., & Sircar, R. (2020). Identifying reward functions using anchor actions. CoRR arXiv:2007.07443."},{"key":"6083_CR9","unstructured":"Gulrajani, I., Ahmed, F., Arjovsky, M., Dumoulin, V., & Courville, A. C. (2017). Improved training of Wasserstein Gans. In Advances in neural information processing systems, Long Beach, CA (Vol. 30, pp. 5767\u20135777)."},{"key":"6083_CR10","unstructured":"Heess, N., Wayne, G., Silver, D., Lillicrap, T. P., Erez, T., & Tassa, Y. (2015). Learning continuous control policies by stochastic value gradients. In Advances in neural information processing systems, Quebec, Canada (Vol. 28, pp. 2944\u20132952)."},{"key":"6083_CR11","unstructured":"Ho, J., & Ermon, S. (2016). Generative adversarial imitation learning. In Advances in neural information processing systems, Barcelona, Spain (Vol. 29, pp. 4565\u20134573)."},{"key":"6083_CR12","unstructured":"Kostrikov, I., Agrawal, K. K., Dwibedi, D., Levine, S., & Tompson, J. (2019). Discriminator-actor-critic: Addressing sample inefficiency and reward bias in adversarial imitation learning. In ICLR 2019, New Orleans, LA."},{"key":"6083_CR13","doi-asserted-by":"crossref","unstructured":"Li, Z., Kiseleva, J., & Rijke, Md. (2019). Dialogue generation: From imitation learning to inverse reinforcement learning. AAAI 2019, Honolulu, HI (pp. 6722\u20136729).","DOI":"10.1609\/aaai.v33i01.33016722"},{"key":"6083_CR14","unstructured":"Mnih, V., Badia, A. P., Mirza, M., Graves, A., Lillicrap, T. P., Harley, T., Silver, D., & Kavukcuoglu, K. (2016). Asynchronous methods for deep reinforcement learning. In ICML 2016, New York City, NY (pp. 1928\u20131937)."},{"key":"6083_CR15","unstructured":"Ng, A. Y., & Russell, S. J. (2000). Algorithms for inverse reinforcement learning. In ICML 2000, Stanford, CA (pp. 663\u2013670)."},{"key":"6083_CR16","unstructured":"Peng, X. B., Kanazawa, A., Toyer, S., Abbeel, P., & Levine, S. (2019). Variational discriminator bottleneck: Improving imitation learning, inverse RL, and GANs by constraining information flow. In ICLR 2019, New Orleans, LA."},{"key":"6083_CR17","unstructured":"Ross, S., & Bagnell, D. (2010). Efficient reductions for imitation learning. In AISTATS 2010, Sardinia, Italy (pp. 661\u2013668)."},{"key":"6083_CR18","unstructured":"Ross, S., Gordon, G. J., & Bagnell, D. (2011). A reduction of imitation learning and structured prediction to no-regret online learning. In AISTATS 2011, Fort Lauderdale, FL (pp. 627\u2013635)."},{"key":"6083_CR19","doi-asserted-by":"crossref","unstructured":"Russell, S. J. (1998). Learning agents for uncertain environments (extended abstract). In COLT 1998, Madison, WI (pp. 101\u2013103).","DOI":"10.1145\/279943.279964"},{"key":"6083_CR20","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M. I., & Moritz, P. (2015). Trust region policy optimization. In ICML 2015, Lille, France (pp. 1889\u20131897)."},{"key":"6083_CR21","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms. CoRR arXiv:1707.06347."},{"key":"6083_CR22","doi-asserted-by":"crossref","unstructured":"Shi, J. C., Yu, Y., Da, Q., Chen, S. Y., & Zeng, A. (2019). Virtual-taobao: virtualizing real-world online retail environment for reinforcement learning. In AAAI 2019, Honolulu, HI, January 27\u2013February 1, 2019 (pp. 4902\u20134909). AAAI Press.","DOI":"10.1609\/aaai.v33i01.33014902"},{"key":"6083_CR23","doi-asserted-by":"crossref","unstructured":"Slonim, N., & Tishby, N. (2000). Document clustering using word clusters via the information bottleneck method. In ACM SIGIR 2000, Greece, Athens (pp. 208\u2013215).","DOI":"10.1145\/345508.345578"},{"key":"6083_CR24","unstructured":"Song, J., Ren, H., Sadigh, D., & Ermon, S. (2018). Multi-agent generative adversarial imitation learning. In Advances in neural information processing systems, Montr\u00e9al, Canada (Vol. 31, pp. 7461\u20137472)."},{"key":"6083_CR25","volume-title":"Reinforcement learning: An introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R. S., & Barto, A. G. (1998). Reinforcement learning: An introduction. MIT Press."},{"key":"6083_CR26","doi-asserted-by":"crossref","unstructured":"Wu, A., Piergiovanni, A. J., & Ryoo, M. S. (2018). Action-conditioned convolutional future regression models for robot imitation learning. In CVPR 2018 workshops, Salt Lake City, UT (pp. 2035\u20132037).","DOI":"10.1109\/CVPRW.2018.00274"}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-021-06083-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-021-06083-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-021-06083-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,3]],"date-time":"2023-01-03T01:03:41Z","timestamp":1672707821000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-021-06083-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,1,3]]},"references-count":26,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2022,3]]}},"alternative-id":["6083"],"URL":"https:\/\/doi.org\/10.1007\/s10994-021-06083-7","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"type":"print","value":"0885-6125"},{"type":"electronic","value":"1573-0565"}],"subject":[],"published":{"date-parts":[[2022,1,3]]},"assertion":[{"value":"15 May 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 August 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 September 2021","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 January 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable. All the experiments in this paper are computer simulations of games and do not involve experiments on animals, plants, or human entities.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"Not applicable. All the experiments in this paper are computer simulations of games and do not involve experiments on animals, plants, or human entities.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable. The paper does not include data or images that require permissions to be published.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}