{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:01:23Z","timestamp":1750309283295,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,9,22]],"date-time":"2023-09-22T00:00:00Z","timestamp":1695340800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,9,22]]},"DOI":"10.1145\/3641584.3641781","type":"proceedings-article","created":{"date-parts":[[2024,6,14]],"date-time":"2024-06-14T22:44:43Z","timestamp":1718405083000},"page":"1310-1316","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Improvement of prioritized experience replay mechanism based on deep deterministic policy gradient algorithm"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9607-573X","authenticated-orcid":false,"given":"Xin","family":"Zhang","sequence":"first","affiliation":[{"name":"Xi'an University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5937-4514","authenticated-orcid":false,"given":"Yihuan","family":"Xu","sequence":"additional","affiliation":[{"name":"Xi'an University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,6,14]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.19678\/j.issn.1000-3428.0061116"},{"key":"e_1_3_2_1_2_1","unstructured":"Volodymyr Mnih Playing Atari with Deep Reinforcement Learning. [J]. CoRR 2013 abs\/1312.5602."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_1_4_1","volume-title":"CoRR","author":"Timothy","year":"2015","unstructured":"Timothy P. Lillicrap Continuous control with deep reinforcement learning. [J]. CoRR, 2015, abs\/1509.02971."},{"key":"e_1_3_2_1_5_1","first-page":"182","article-title":"A review of research on sparse reward problem in deep reinforcement learning[J]","volume":"2020","author":"Weiyi Yang","unstructured":"Yang Weiyi,Bai Chenjia,Cai Chao A review of research on sparse reward problem in deep reinforcement learning[J]. Computer Science,2020,47(03):182-191.","journal-title":"Computer Science"},{"key":"e_1_3_2_1_6_1","volume-title":"CoRR","author":"van Hasselt Hado","year":"2015","unstructured":"Hado van Hasselt and Arthur Guez and David Silver. Deep Reinforcement Learning with Double Q-learning. [J]. CoRR, 2015, abs\/1509.06461."},{"key":"e_1_3_2_1_7_1","unstructured":"Tom Schaul Prioritized Experience Replay. [J]. CoRR 2015 abs\/1511.05952."},{"key":"e_1_3_2_1_8_1","unstructured":"Ziyu Wang 0001 Sample Efficient Actor-Critic with Experience Replay. [J]. CoRR 2016 abs\/1611.01224."},{"key":"e_1_3_2_1_9_1","volume-title":"A novel DDPG method with prioritized experience replay. 316-321. 10.1109\/SMC.2017.8122622","author":"Lifeng Yuenan","year":"2017","unstructured":"Hou, Yuenan & Lifeng, Liu & Wei, Qing & Xu, Xudong & Chen, Chunlin. (2017). A novel DDPG method with prioritized experience replay. 316-321. 10.1109\/SMC.2017.8122622."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCC.2011.2106494"},{"key":"e_1_3_2_1_11_1","unstructured":"Greg Brockman OpenAI Gym. [J]. CoRR 2016 abs\/1606.01540."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/BF00993104"},{"key":"e_1_3_2_1_13_1","volume-title":"CoRR","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe and Christian Szegedy. Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift. [J]. CoRR, 2015, abs\/1502.03167."},{"issue":"6","key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","first-page":"910","DOI":"10.1016\/j.neuron.2009.11.016","article-title":"Rewarded Outcomes Enhance Reactivation of Experience in the Hippocampus[J]","volume":"64","author":"Singer Annabelle C.","year":"2009","unstructured":"Annabelle C. Singer and Loren M. Frank. Rewarded Outcomes Enhance Reactivation of Experience in the Hippocampus[J]. Neuron, 2009, 64(6): 910-921. Doi:10.1016\/j.neuron.2009.11.016.","journal-title":"Neuron"},{"key":"e_1_3_2_1_15_1","unstructured":"Nicolas Heess Learning Continuous Control Policies by Stochastic Value Gradients. [J]. CoRR 2015 abs\/1510.09142."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.19734\/j.issn.1001-3695.2018.06.0513"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.13195\/j.kzyjc.2017.0261"},{"volume-title":"Deep deterministic policy gradient algorithm optimization[J]","year":"2020","key":"e_1_3_2_1_18_1","unstructured":"Liu, Yang, Li, Jianjun. Deep deterministic policy gradient algorithm optimization[J]. Journal of Liaoning University of Engineering and Technology (Natural Science Edition),2020,39(06):545-549."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.19682\/j.cnki.1005-8885.2022.0025"},{"key":"e_1_3_2_1_20_1","first-page":"262","article-title":"Q-learning active sampling method based on TD-error adaptive correction[J]","volume":"2019","author":"Chenjia Bai","unstructured":"Bai Chenjia,Liu Peng,Zhao Wei Deep Q-learning active sampling method based on TD-error adaptive correction[J]. Computer Research and Development,2019,56(02):262-280.","journal-title":"Computer Research and Development"},{"key":"e_1_3_2_1_21_1","first-page":"420","article-title":"Parallel priority experience playback mechanism of MADDPG algorithm[J]","volume":"2021","author":"Gao","unstructured":"Gao A, Dong C, Li L Parallel priority experience playback mechanism of MADDPG algorithm[J]. Systems Engineering and Electronic Technology,2021,43(02):420-433.","journal-title":"Systems Engineering and Electronic Technology"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.27005\/d.cnki.gdzku.2022.001162"},{"key":"e_1_3_2_1_23_1","first-page":"37","article-title":"A deep deterministic policy gradient approach based on episode experience replay[J]","volume":"2021","author":"Jianxing Zhang","unstructured":"Zhang Jianxing,Liu Quan. A deep deterministic policy gradient approach based on episode experience replay[J]. Computer Science.,2021,48(10):37-43.","journal-title":"Computer Science."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.19292\/j.cnki.jdxxp.2021.02.010"}],"event":{"name":"AIPR 2023: 2023 6th International Conference on Artificial Intelligence and Pattern Recognition","acronym":"AIPR 2023","location":"Xiamen China"},"container-title":["2023 6th International Conference on Artificial Intelligence and Pattern Recognition (AIPR)"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641584.3641781","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641584.3641781","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:03:12Z","timestamp":1750291392000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641584.3641781"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,22]]},"references-count":24,"alternative-id":["10.1145\/3641584.3641781","10.1145\/3641584"],"URL":"https:\/\/doi.org\/10.1145\/3641584.3641781","relation":{},"subject":[],"published":{"date-parts":[[2023,9,22]]},"assertion":[{"value":"2024-06-14","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}