{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T07:10:37Z","timestamp":1771657837861,"version":"3.50.1"},"reference-count":38,"publisher":"Zhejiang University Press","issue":"11","license":[{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,1]],"date-time":"2023-11-01T00:00:00Z","timestamp":1698796800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front Inform Technol Electron Eng"],"published-print":{"date-parts":[[2023,11]]},"DOI":"10.1631\/fitee.2300084","type":"journal-article","created":{"date-parts":[[2023,12,6]],"date-time":"2023-12-06T19:02:20Z","timestamp":1701889340000},"page":"1541-1556","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Embedding expert demonstrations into clustering buffer for effective deep reinforcement learning","\u57fa\u4e8e\u4e13\u5bb6\u793a\u6559\u805a\u7c7b\u7ecf\u9a8c\u6c60\u7684\u9ad8\u6548\u6df1\u5ea6\u5f3a\u5316\u5b66\u4e60"],"prefix":"10.1631","volume":"24","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7288-8323","authenticated-orcid":false,"given":"Shihmin","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Binqi","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengfeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junping","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0892-1213","authenticated-orcid":false,"given":"Jian","family":"Pu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"635","published-online":{"date-parts":[[2023,12,7]]},"reference":[{"key":"ref1","first-page":"5055","article-title":"Hindsight experience replay","volume-title":"Proc 31st Int Conf on Neural Information Processing Systems","author":"Andrychowicz","year":"2017"},{"key":"ref2","first-page":"1479","article-title":"Unifying count-based exploration and intrinsic motivation","volume-title":"Proc 30th Int Conf on Neural Information Processing Systems","author":"Bellemare","year":"2016"},{"key":"ref3","volume-title":"OpenAI Gym","author":"Brockman","year":"2016"},{"key":"ref4","article-title":"Reinforcement learning for non-stationary Markov decision processes: the blessing of (more) optimism","volume-title":"Proc 37th Int Conf on Machine Learning","author":"Cheung","year":"2020"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.2200323"},{"key":"ref6","volume-title":"Learning robust rewards with adversarial inverse reinforcement learning","author":"Fu","year":"2017"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"ref8","first-page":"1861","article-title":"Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc 35th Int Conf on Machine Learning","author":"Haarnoja","year":"2018"},{"key":"ref9","first-page":"394","article-title":"Deep Q-learning from demonstrations","volume-title":"Proc AAAI Conf on Artificial Intelligence","author":"Hester","year":"2018"},{"key":"ref10","first-page":"4572","article-title":"Generative adversarial imitation learning","volume-title":"Proc 30th Int Conf on Neural Information Processing Systems","author":"Ho","year":"2016"},{"key":"ref11","first-page":"1117","article-title":"VIME: variational information maximizing exploration","volume-title":"Proc 30th Int Conf on Neural Information Processing Systems","author":"Houthooft","year":"2016"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3216454"},{"key":"ref13","article-title":"Auto-encoding variational Bayes","volume-title":"Proc 2nd Int Conf on Learning Representations","author":"Kingma","year":"2014"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.2200128"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.2200109"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1561\/2200000086"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s00357-014-9161-z"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/icra.2018.8463162"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58595-2_44"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/tip.2022.3221290"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-100819-063206"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/itsc45102.2020.9294422"},{"key":"ref23","article-title":"Prioritized experience replay","volume-title":"Proc 4th Int Conf on Learning Representations","author":"Schaul","year":"2016"},{"key":"ref24","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc 32nd Int Conf on Machine Learning","author":"Schulman","year":"2015"},{"key":"ref25","article-title":"High-dimensional continuous control using generalized advantage estimation","volume-title":"Proc 4th Int Conf on Learning Representations","author":"Schulman","year":"2016"},{"key":"ref26","volume-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6049"},{"key":"ref29","volume-title":"Reinforcement Learning: an Introduction","author":"Sutton","year":"1998"},{"key":"ref30","author":"Vecerik","year":"2017","journal-title":"Leveraging demonstrations for deep reinforcement learning on robotics problems with sparse rewards"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-019-1724-z"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-63833-7_38"},{"key":"ref33","first-page":"478","article-title":"Unsupervised deep embedding for clustering analysis","volume-title":"Proc 33rd Int Conf on Machine Learning","author":"Xie","year":"2016"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.2250000"},{"key":"ref35","article-title":"Towards playing full MOBA games with deep reinforcement learning","volume-title":"Proc 34th Int Conf on Neural Information Processing Systems","author":"Ye","year":"2020"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/tiv.2023.3256982"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/tiv.2023.3264601"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1631\/fitee.2300089"}],"container-title":["Frontiers of Information Technology &amp; Electronic Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1631\/FITEE.2300084.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1631\/FITEE.2300084\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1631\/FITEE.2300084.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T06:36:56Z","timestamp":1771655816000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1631\/FITEE.2300084"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11]]},"references-count":38,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2023,11]]}},"alternative-id":["1961"],"URL":"https:\/\/doi.org\/10.1631\/fitee.2300084","relation":{},"ISSN":["2095-9184","2095-9230"],"issn-type":[{"value":"2095-9184","type":"print"},{"value":"2095-9230","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11]]},"assertion":[{"value":"12 February 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 May 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 December 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Junping ZHANG and Jian PU are an editorial board member and a corresponding expert of <i>Frontiers of Information Technology & Electronic Engineering<\/i>, respectively, and they were not involved with the peer review process of this paper. All authors declare that they have no conflict of interest.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with ethics guidelines"}}]}}