{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T05:09:29Z","timestamp":1785906569992,"version":"3.56.0"},"reference-count":43,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100003696","name":"Korean government","doi-asserted-by":"publisher","award":["23ZR1100"],"award-info":[{"award-number":["23ZR1100"]}],"id":[{"id":"10.13039\/501100003696","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002380","name":"A Study of Hyper-Connected Thinking Internet Technology by autonomous connecting, controlling, and evolving ways) and the research fund of Hanyang University","doi-asserted-by":"publisher","award":["HY-201900000002966"],"award-info":[{"award-number":["HY-201900000002966"]}],"id":[{"id":"10.13039\/501100002380","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Development of Ultra-Low Power Deep Learning Processor Technology using Advanced Data Reuse for Edge Applications","award":["2020-0-01297"],"award-info":[{"award-number":["2020-0-01297"]}]},{"DOI":"10.13039\/501100003621","name":"Ministry of Science and ICT","doi-asserted-by":"publisher","award":["2020M3H2A1076786"],"award-info":[{"award-number":["2020M3H2A1076786"]}],"id":[{"id":"10.13039\/501100003621","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/access.2024.3367002","type":"journal-article","created":{"date-parts":[[2024,2,19]],"date-time":"2024-02-19T20:16:35Z","timestamp":1708373795000},"page":"36055-36065","source":"Crossref","is-referenced-by-count":4,"title":["Pruning With Scaled Policy Constraints for Light-Weight Reinforcement Learning"],"prefix":"10.1109","volume":"12","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8568-5808","authenticated-orcid":false,"given":"Seongmin","family":"Park","sequence":"first","affiliation":[{"name":"Department of Electronic Engineering, Hanyang University, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hyungmin","family":"Kim","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering, Hanyang University, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hyunhak","family":"Kim","sequence":"additional","affiliation":[{"name":"Autonomous IoT Research Section, Electronics and Telecommunications Research Institute, Daejeon, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3075-8694","authenticated-orcid":false,"given":"Jungwook","family":"Choi","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering, Hanyang University, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Single-shot pruning for offline reinforcement learning","author":"Arnob","year":"2021","journal-title":"arXiv:2112.15579"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JETCAS.2019.2910232"},{"key":"ref3","article-title":"The lottery ticket hypothesis: Finding sparse, trainable neural networks","author":"Frankle","year":"2018","journal-title":"arXiv:1803.03635"},{"key":"ref4","article-title":"D4RL: Datasets for deep data-driven reinforcement learning","author":"Fu","year":"2020","journal-title":"arXiv:2004.07219"},{"key":"ref5","first-page":"20132","article-title":"A minimalist approach to offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Fujimoto"},{"key":"ref6","first-page":"1587","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref7","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref8","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding","author":"Han","year":"2015","journal-title":"arXiv:1510.00149"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00447"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.155"},{"key":"ref11","first-page":"3821","article-title":"An information-theoretic justification for model pruning","volume-title":"Proc. Int. Conf. Artif. Intell. Stat. (AISTATS)","author":"Isik"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3301273"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2004.1389727"},{"key":"ref14","first-page":"1","article-title":"Stabilizing off-policy Q-learning via bootstrapping error reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Kumar"},{"key":"ref15","first-page":"1179","article-title":"Conservative Q-learning for offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Kumar"},{"key":"ref16","article-title":"Pruning filters for efficient ConvNets","author":"Li","year":"2016","journal-title":"arXiv:1608.08710"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00029"},{"key":"ref18","first-page":"3053","article-title":"RLlib: Abstractions for distributed reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Liang"},{"key":"ref19","article-title":"Continuous control with deep reinforcement learning","author":"Lillicrap","year":"2015","journal-title":"arXiv:1509.02971"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.298"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2020.2967566"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2206625119"},{"key":"ref23","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mnih"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01152"},{"key":"ref26","article-title":"SOSP: Efficiently capturing global correlations by second-order structured pruning","author":"Nonnenmacher","year":"2021","journal-title":"arXiv:2110.11395"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.5244\/C.31.11"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i8.26130"},{"key":"ref29","volume-title":"Markov Decision Processes: Discrete Stochastic Dynamic Programming","author":"Puterman","year":"2014"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/S1364-6613(99)01327-3"},{"key":"ref31","first-page":"1889","article-title":"Trust region policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Schulman"},{"key":"ref32","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","journal-title":"arXiv:1707.06347"},{"key":"ref33","article-title":"Safe, multi-agent, reinforcement learning for autonomous driving","author":"Shalev-Shwartz","year":"2016","journal-title":"arXiv:1610.03295"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref35","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref37","first-page":"1","article-title":"Exponentially weighted imitation learning for batched historical data","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Wang"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00088"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992698"},{"key":"ref40","first-page":"1","article-title":"Learning structured sparsity in deep neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"29","author":"Wen"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref42","article-title":"Behavior regularized offline reinforcement learning","author":"Wu","year":"2019","journal-title":"arXiv:1911.11361"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.643"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10380310\/10439169.pdf?arnumber=10439169","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,14]],"date-time":"2024-03-14T18:03:31Z","timestamp":1710439411000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10439169\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":43,"URL":"https:\/\/doi.org\/10.1109\/access.2024.3367002","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}