{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T16:11:17Z","timestamp":1779379877144,"version":"3.53.1"},"reference-count":34,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,11,1]],"date-time":"2019-11-01T00:00:00Z","timestamp":1572566400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,11,1]],"date-time":"2019-11-01T00:00:00Z","timestamp":1572566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,11,1]],"date-time":"2019-11-01T00:00:00Z","timestamp":1572566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,11]]},"DOI":"10.1109\/milcom47813.2019.9020862","type":"proceedings-article","created":{"date-parts":[[2020,3,6]],"date-time":"2020-03-06T15:00:21Z","timestamp":1583506821000},"page":"1-8","source":"Crossref","is-referenced-by-count":10,"title":["Neural Malware Control with Deep Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Yu","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jack W.","family":"Stokes","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mady","family":"Marinescu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638293"},{"key":"ref32","author":"schulman","year":"2015","journal-title":"High-dimensional continuous control using generalized advantage estimation"},{"key":"ref31","first-page":"2944","article-title":"Learning continuous control policies by stochastic value gradients","author":"heess","year":"2015","journal-title":"Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref30","author":"hausknecht","year":"2015","journal-title":"Deep Reinforcement Learning in Parameterized Action Space"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-40667-1_20"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2012.11.007"},{"key":"ref11","author":"lillicrap","year":"2015","journal-title":"Continuous control with deep reinforcement learning"},{"key":"ref12","author":"mnih","year":"2013","journal-title":"Playing atari with deep reinforcement learning"},{"key":"ref13","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"mnih","year":"2015","journal-title":"Nature"},{"key":"ref14","author":"zeiler","year":"2012","journal-title":"ADADELTA An Adaptive Learning Rate Method"},{"key":"ref15","author":"chollet","year":"2015","journal-title":"Keras"},{"key":"ref16","first-page":"473","volume":"472","author":"al-rfou","year":"2016","journal-title":"Theano A Python framework for fast computation of mathematical expressions"},{"key":"ref17","author":"sutton","year":"1984","journal-title":"Temporal credit assignment in reinforcement learning"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/B978-1-55860-335-6.50027-1"},{"key":"ref28","article-title":"Pgq: Combining policy gradient and q-learning","author":"o\u2019donoghue","year":"2017","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MILCOM.2018.8599785"},{"key":"ref27","article-title":"Q-prop: Sample-efficient policy gradient with an off-policy critic","author":"gu","year":"2017","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref3","first-page":"137","article-title":"Deep learning for classification of malware system call sequences","author":"kolosnjaji","year":"2016","journal-title":"Australasian Joint Conference on Artificial Intelligence"},{"key":"ref6","volume":"135","author":"sutton","year":"1998","journal-title":"Introduction to Reinforcement Learning"},{"key":"ref29","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"2016","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref5","author":"anderson","year":"2018","journal-title":"Learning to evade static PE machine learning malware models via reinforcement learning"},{"key":"ref8","first-page":"1334","article-title":"End-to-end training of deep visuomotor policies","volume":"17","author":"levine","year":"2016","journal-title":"J Machine Learning Research (JMLR)"},{"key":"ref7","first-page":"387","article-title":"Deterministic policy gradient algorithms","author":"silver","year":"2014","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952603"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2009.05.011"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178304"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1613\/jair.301"},{"key":"ref22","doi-asserted-by":"crossref","first-page":"354","DOI":"10.1038\/nature24270","article-title":"Mastering the game of go without human knowledge","volume":"550","author":"silver","year":"2017","journal-title":"Nature"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1038\/nature16961"},{"key":"ref24","first-page":"1995","article-title":"Dueling network architectures for deep reinforcement learning","author":"wang","year":"2016","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref23","first-page":"2094","article-title":"Deep reinforcement learning with double q-learning","volume":"16","author":"van hasselt","year":"2016","journal-title":"AAAI"},{"key":"ref26","first-page":"1329","article-title":"Bench-marking deep reinforcement learning for continuous control","author":"duan","year":"2016","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref25","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"2015","journal-title":"International Conference on Machine Learning (ICML)"}],"event":{"name":"MILCOM 2019 - 2019 IEEE Military Communications Conference (MILCOM)","location":"Norfolk, VA, USA","start":{"date-parts":[[2019,11,12]]},"end":{"date-parts":[[2019,11,14]]}},"container-title":["MILCOM 2019 - 2019 IEEE Military Communications Conference (MILCOM)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8993674\/9020712\/09020862.pdf?arnumber=9020862","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,17]],"date-time":"2022-07-17T21:47:47Z","timestamp":1658094467000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9020862\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11]]},"references-count":34,"URL":"https:\/\/doi.org\/10.1109\/milcom47813.2019.9020862","relation":{},"subject":[],"published":{"date-parts":[[2019,11]]}}}