{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T18:24:09Z","timestamp":1783794249469,"version":"3.55.0"},"reference-count":25,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,9,14]],"date-time":"2022-09-14T00:00:00Z","timestamp":1663113600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,9,14]],"date-time":"2022-09-14T00:00:00Z","timestamp":1663113600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62003207, 61773350"],"award-info":[{"award-number":["62003207, 61773350"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010246","name":"Postdoctoral Science Foundation of Jiangsu Province","doi-asserted-by":"publisher","award":["2021M690629"],"award-info":[{"award-number":["2021M690629"]}],"id":[{"id":"10.13039\/501100010246","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s10586-022-03742-9","type":"journal-article","created":{"date-parts":[[2022,9,14]],"date-time":"2022-09-14T07:02:45Z","timestamp":1663138965000},"page":"2001-2010","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Upper confident bound advantage function proximal policy optimization"],"prefix":"10.1007","volume":"26","author":[{"given":"Guiliang","family":"Xie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1400-8612","authenticated-orcid":false,"given":"Wei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhi","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gaojian","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,9,14]]},"reference":[{"issue":"7587","key":"3742_CR1","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., Huang, A., Maddison, C.J., et al.: Mastering the game of Go with deep neural networks and tree search. Nature. 529(7587), 484\u2013489 (2016)","journal-title":"Nature."},{"key":"3742_CR2","doi-asserted-by":"crossref","unstructured":"Pang, Z. J., Liu, R. Z., Meng, Z. Y., et al.: On reinforcement learning for full-length game of starcraft. In: Thirty-third Association for the Advancement of Artificial Intelligence, pp. 4691-4698 (2019)","DOI":"10.1609\/aaai.v33i01.33014691"},{"key":"3742_CR3","unstructured":"Ye, D., Chen, G., Zhang, W., et al.: Towards playing full moba games with deep reinforcement learning. In: Thirty-forth Advances in Neural Information Processing Systems, pp. 621-632 (2020)"},{"issue":"11","key":"3742_CR4","doi-asserted-by":"publisher","first-page":"1238","DOI":"10.1177\/0278364913495721","volume":"32","author":"J Kober","year":"2013","unstructured":"Kober, J., Bagnell, J.A., Peters, J.: Reinforcement learning in robotics: A survey. Int. J. Robotic. Res. 32(11), 1238\u20131274 (2013)","journal-title":"Int. J. Robotic. Res."},{"key":"3742_CR5","doi-asserted-by":"crossref","unstructured":"Kuderer, M., Gulati, S., Burgard, W.: Learning driving styles for autonomous vehicles from demonstration. In: Proceedings of IEEE International Conference on Robotics and Automation, pp. 2641-2646 (2015)","DOI":"10.1109\/ICRA.2015.7139555"},{"issue":"7792","key":"3742_CR6","doi-asserted-by":"publisher","first-page":"706","DOI":"10.1038\/s41586-019-1923-7","volume":"577","author":"AW Senior","year":"2020","unstructured":"Senior, A.W., Evans, R., Jumper, J., et al.: Improved protein structure prediction using potentials from deep learning. Nature 577(7792), 706\u2013710 (2020)","journal-title":"Nature"},{"key":"3742_CR7","doi-asserted-by":"crossref","unstructured":"Li, M., Qin, Z., Jiao, Y., et al.: Efficient ridesharing order dispatching with mean field multi-agent reinforcement learning. In: The World Wide Web Conference, pp. 983-994 (2019)","DOI":"10.1145\/3308558.3313433"},{"key":"3742_CR8","doi-asserted-by":"crossref","unstructured":"Parr, R., Li, L., Taylor, G., et al.: An analysis of linear models, linear value-function approximation, and feature selection for reinforcement learning. In: Proceedings of the 25th International Conference on Machine learning, pp. 752-759 (2008)","DOI":"10.1145\/1390156.1390251"},{"key":"3742_CR9","doi-asserted-by":"crossref","unstructured":"Hessel, M., Modayil, J., Van, Hasselt, H., et al.: Rainbow: Combining improvements in deep reinforcement learning. arXiv preprint arXiv:1710.02298 (2017)","DOI":"10.1609\/aaai.v32i1.11796"},{"issue":"7540","key":"3742_CR10","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., et al.: Human-level control through deep reinforcement learning. Nature. 518(7540), 529\u2013533 (2015)","journal-title":"Nature."},{"key":"3742_CR11","unstructured":"Kumar, A., Zhou, A., Tucker, G., et al.: Conservative q-learning for offline reinforcement learning. In: Advances in Neural Information Processing Systems, pp. 1179-1191 (2020)"},{"key":"3742_CR12","unstructured":"Kakade. S, M.: A natural policy gradient. In: Advances in Neural Information Processing Systems, pp. 1057-1063 (2001)"},{"key":"3742_CR13","unstructured":"Silver, D., Lever, G., Heess, N., et al.: Deterministic policy gradient algorithms. In: International Conference on Machine Learning, pp. 387-395 (2014)"},{"key":"3742_CR14","unstructured":"Konda, V. R., Tsitsiklis, J. N.: Actor-critic algorithms. In: Advances in Neural Information Processing Systems, pp. 1008-1014 (2000)"},{"key":"3742_CR15","unstructured":"Lillicrap, T. P., Hunt, J, J., Pritzel, A., et al.: Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)"},{"key":"3742_CR16","unstructured":"Schulman, J., Levine, S., Abbeel, P., et al.: Trust region policy optimization. In: Proceedings of Machine Learning Research, pp. 1889-1897 (2015)"},{"key":"3742_CR17","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., et al.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"3742_CR18","doi-asserted-by":"crossref","unstructured":"Ye, D., Liu, Z., Sun, M., et al.: Mastering complex control in moba games with deep reinforcement learning. In: Thirty-forth Advances in Neural Information Processing Systems, pp. 6672-6679 (2020)","DOI":"10.1609\/aaai.v34i04.6144"},{"key":"3742_CR19","unstructured":"Chen, G., Peng, Y., Zhang, M.: An adaptive clipping approach for proximal policy optimization. arXiv preprint arXiv:1804.06461 (2018)"},{"key":"3742_CR20","unstructured":"Wang, Y., He, H., Tan, X., et al.: Trust region-guided proximal policy optimization. In: Thirty-third Advances in Neural Information Processing Systems, pp. 2061-2069 (2019)"},{"key":"3742_CR21","doi-asserted-by":"crossref","unstructured":"H?m?l?inen, P., Babadi, A., Ma, X., et al.: PPO-CMA: Proximal policy optimization with covariance matrix adaptation. In: 2020 IEEE 30th International Workshop on Machine Learning for Signal Processing (MLSP), pp. 1-6 (2020)","DOI":"10.1109\/MLSP49062.2020.9231618"},{"key":"3742_CR22","doi-asserted-by":"crossref","unstructured":"Van, Hasselt, H., Wiering, M. A.: Reinforcement learning in continuous action spaces. In: 2007 IEEE International Symposium on Approximate Dynamic Programming and Reinforcement Learning, pp. 272-279 (2007)","DOI":"10.1109\/ADPRL.2007.368199"},{"issue":"4","key":"3742_CR23","doi-asserted-by":"publisher","first-page":"2753","DOI":"10.1007\/s10586-019-03042-9","volume":"23","author":"Z Peng","year":"2020","unstructured":"Peng, Z., Lin, J., Cui, D., et al.: A multi-objective trade-off framework for cloud resource scheduling based on the deep Q-network algorithm. Cluster Comput. 23(4), 2753\u20132767 (2020)","journal-title":"Cluster Comput."},{"issue":"11","key":"3742_CR24","doi-asserted-by":"publisher","first-page":"2471","DOI":"10.1016\/j.automatica.2009.07.008","volume":"45","author":"S Bhatnagar","year":"2009","unstructured":"Bhatnagar, S., Sutton, R.S., Ghavamzadeh, M., et al.: Natural actor-critic algorithms. Automatica. 45(11), 2471\u20132482 (2009)","journal-title":"Automatica."},{"issue":"1","key":"3742_CR25","doi-asserted-by":"publisher","first-page":"795","DOI":"10.1007\/s10586-017-1303-8","volume":"22","author":"F Qiming","year":"2019","unstructured":"Qiming, F., Wen, H., Quan, L., et al.: Residual Sarsa algorithm with function approximation. Cluster Computing. 22(1), 795\u2013807 (2019)","journal-title":"Cluster Computing."}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-022-03742-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10586-022-03742-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-022-03742-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T20:54:51Z","timestamp":1727988891000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10586-022-03742-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,9,14]]},"references-count":25,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["3742"],"URL":"https:\/\/doi.org\/10.1007\/s10586-022-03742-9","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"value":"1386-7857","type":"print"},{"value":"1573-7543","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,9,14]]},"assertion":[{"value":"4 March 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 August 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 August 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 September 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Ethics approval was not required for this research.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}},{"value":"This work did not involved humans and animals.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Humans resoureces"}}]}}