{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T15:13:34Z","timestamp":1784906014804,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":14,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,8,26]],"date-time":"2019-08-26T00:00:00Z","timestamp":1566777600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,8,26]]},"DOI":"10.1145\/3387168.3387199","type":"proceedings-article","created":{"date-parts":[[2020,5,26]],"date-time":"2020-05-26T00:25:51Z","timestamp":1590452751000},"page":"1-5","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":151,"title":["Twin-Delayed DDPG"],"prefix":"10.1145","author":[{"given":"Stephen","family":"Dankwa","sequence":"first","affiliation":[{"name":"Automation Engineering, University of Electronic Science and Technology of China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenfeng","family":"Zheng","sequence":"additional","affiliation":[{"name":"Automation Engineering, University of Electronic Science and Technology of China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,5,25]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Openai gym","author":"Brockman G.","year":"2016","unstructured":"Brockman , G. , Cheung , V. , Pettersson , L. , Schneider , J. , Schulman , J. , Tang , J. , and Zaremba , W . Openai gym , 2016 . Brockman, G., Cheung, V., Pettersson, L., Schneider, J., Schulman, J., Tang, J., and Zaremba, W. Openai gym, 2016."},{"key":"e_1_3_2_1_2_1","unstructured":"Erwin Coumans Yunfei Bai and Jasmine Hsu. PyBullet. Available on: https:\/\/pypi.org\/project\/pybullet\/. Retrieved: 5\/30\/2019  Erwin Coumans Yunfei Bai and Jasmine Hsu. PyBullet. Available on: https:\/\/pypi.org\/project\/pybullet\/. Retrieved: 5\/30\/2019"},{"key":"e_1_3_2_1_3_1","volume-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. arXiv preprint arXiv:1801.01290","author":"Haarnoja T.","year":"2018","unstructured":"Haarnoja , T. , Zhou , A. , Abbeel , P. , and Levine , S . Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. arXiv preprint arXiv:1801.01290 , 2018 Haarnoja, T., Zhou, A., Abbeel, P., and Levine, S. Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. arXiv preprint arXiv:1801.01290, 2018"},{"key":"e_1_3_2_1_4_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma","year":"2014","unstructured":"Kingma , Diederik and Ba, Jimmy . Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 , 2014 . Kingma, Diederik and Ba, Jimmy. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980, 2014."},{"key":"e_1_3_2_1_5_1","volume-title":"Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971","author":"Lillicrap T. P.","year":"2015","unstructured":"Lillicrap , T. P. , Hunt , J. J. , Pritzel , A. , Heess , N. , Erez , T. , Tassa , Y. , Silver , D. , and Wierstra , D . Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 , 2015 . Available online: https:\/\/arxiv.org\/abs\/1509.02971 Lillicrap, T. P., Hunt, J. J., Pritzel, A., Heess, N., Erez, T., Tassa, Y., Silver, D., and Wierstra, D. Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971, 2015. Available online: https:\/\/arxiv.org\/abs\/1509.02971"},{"key":"e_1_3_2_1_6_1","volume-title":"Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602","author":"Mnih","year":"2013","unstructured":"Mnih , Volodymyr, Kavukcuoglu , Koray, Silver , David, Graves , Alex, Antonoglou , Ioannis, Wierstra , Daan, and Riedmiller , Martin. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 , 2013 . Mnih, Volodymyr, Kavukcuoglu, Koray, Silver, David, Graves, Alex, Antonoglou, Ioannis, Wierstra, Daan, and Riedmiller, Martin. Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602, 2013."},{"key":"e_1_3_2_1_7_1","first-page":"1057","author":"Sutton D.A.","year":"2000","unstructured":"R.S. Sutton , D.A. McAllester , S.P. Singh , Y. Mansour , Policy gradient methods for reinforcement learning with function approximation, in : Advances in Neural Information Processing Systems , 2000 , pp. 1057 -- 1063 . R.S. Sutton, D.A. McAllester, S.P. Singh, Y. Mansour, Policy gradient methods for reinforcement learning with function approximation, in: Advances in Neural Information Processing Systems, 2000, pp. 1057--1063.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_8_1","first-page":"1889","volume-title":"International Conference on Machine Learning","author":"Schulman J.","year":"2015","unstructured":"Schulman , J. , Levine , S. , Abbeel , P. , Jordan , M. , and Moritz , P . Trust region policy optimization . In International Conference on Machine Learning , pp. 1889 -- 1897 , 2015 . Schulman, J., Levine, S., Abbeel, P., Jordan, M., and Moritz, P. Trust region policy optimization. In International Conference on Machine Learning, pp. 1889--1897, 2015."},{"key":"e_1_3_2_1_9_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman J.","year":"2017","unstructured":"Schulman , J. , Wolski , F. , Dhariwal , P. , Radford , A. , and Klimov , O . Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 , 2017 . Schulman, J., Wolski, F., Dhariwal, P., Radford, A., and Klimov, O. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347, 2017."},{"key":"e_1_3_2_1_10_1","volume-title":"Addressing Function Approximation Error in Actor-Critic Methods. https:\/\/arxiv.org\/pdf\/1802.09477.pdf","author":"Scott Fujimoto","year":"2018","unstructured":"Scott Fujimoto , Herke van Hoof and David Meger , Addressing Function Approximation Error in Actor-Critic Methods. https:\/\/arxiv.org\/pdf\/1802.09477.pdf , 2018 Scott Fujimoto, Herke van Hoof and David Meger, Addressing Function Approximation Error in Actor-Critic Methods. https:\/\/arxiv.org\/pdf\/1802.09477.pdf, 2018"},{"key":"e_1_3_2_1_11_1","volume-title":"ICML","author":"Silver","year":"2014","unstructured":"Silver , David, Lever , Guy, Heess , Nicolas, Degris , Thomas, Wierstra , Daan, and Riedmiller , Martin. Deterministic policy gradient algorithms . In ICML , 2014 . Silver, David, Lever, Guy, Heess, Nicolas, Degris, Thomas, Wierstra, Daan, and Riedmiller, Martin. Deterministic policy gradient algorithms. In ICML, 2014."},{"key":"e_1_3_2_1_12_1","volume-title":"Reinforcement learning: An introduction","author":"Sutton S.","year":"2039","unstructured":"Sutton , S. Richard and Andrew G. Barto . Reinforcement learning: An introduction . 2 nd edition, The MIT Press , Cambridge, Massachusetts , ISBN 978026 2039 246. Available online: http:\/\/www.incompleteideas.net\/book\/the-book.html, 2018. Sutton, S. Richard and Andrew G. Barto. Reinforcement learning: An introduction. 2nd edition, The MIT Press, Cambridge, Massachusetts, ISBN 9780262039246. Available online: http:\/\/www.incompleteideas.net\/book\/the-book.html, 2018.","edition":"2"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/3016100.3016191"},{"key":"e_1_3_2_1_14_1","first-page":"5285","volume-title":"Advances in Neural Information Processing Systems","author":"Wu Y.","year":"2017","unstructured":"Wu , Y. , Mansimov , E. , Grosse , R. B. , Liao , S. , and Ba , J . Scalable trust-region method for deep reinforcement learning using kronecker-factored approximation . In Advances in Neural Information Processing Systems , pp. 5285 -- 5294 , 2017 . Wu, Y., Mansimov, E., Grosse, R. B., Liao, S., and Ba, J. Scalable trust-region method for deep reinforcement learning using kronecker-factored approximation. In Advances in Neural Information Processing Systems, pp. 5285--5294, 2017."}],"event":{"name":"ICVISP 2019: 3rd International Conference on Vision, Image and Signal Processing","location":"Vancouver BC Canada","acronym":"ICVISP 2019"},"container-title":["Proceedings of the 3rd International Conference on Vision, Image and Signal Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3387168.3387199","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3387168.3387199","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T00:26:17Z","timestamp":1750206377000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3387168.3387199"}},"subtitle":["A Deep Reinforcement Learning Technique to Model a Continuous Movement of an Intelligent Robot Agent"],"short-title":[],"issued":{"date-parts":[[2019,8,26]]},"references-count":14,"alternative-id":["10.1145\/3387168.3387199","10.1145\/3387168"],"URL":"https:\/\/doi.org\/10.1145\/3387168.3387199","relation":{},"subject":[],"published":{"date-parts":[[2019,8,26]]},"assertion":[{"value":"2020-05-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}