{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T10:10:32Z","timestamp":1730283032058,"version":"3.28.0"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,9,1]],"date-time":"2020-09-01T00:00:00Z","timestamp":1598918400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,9,1]],"date-time":"2020-09-01T00:00:00Z","timestamp":1598918400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,9,1]],"date-time":"2020-09-01T00:00:00Z","timestamp":1598918400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,9]]},"DOI":"10.1109\/mlsp49062.2020.9231849","type":"proceedings-article","created":{"date-parts":[[2020,10,21]],"date-time":"2020-10-21T17:53:19Z","timestamp":1603302799000},"page":"1-6","source":"Crossref","is-referenced-by-count":2,"title":["Improving Deep Reinforcement Learning for Financial Trading Using Neural Network Distillation"],"prefix":"10.1109","author":[{"given":"Avraam","family":"Tsantekidis","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nikolaos","family":"Passalis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anastasios","family":"Tefas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref30","first-page":"742","article-title":"Learning efficient object detection models with knowledge distillation","author":"chen","year":"0","journal-title":"Proceedings of the Advances in Neural Information Processing Systems"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2012.2232935"},{"journal-title":"Distilling the knowledge in a neural network","year":"2015","author":"hinton","key":"ref11"},{"key":"ref12","first-page":"176","article-title":"Averaged-DQN: Variance reduction and stabilization for deep reinforcement learning","author":"anschel","year":"0","journal-title":"Proceedings of the International Conference on Machine Learning"},{"journal-title":"Dueling network architectures for deep reinforcement learning","year":"2015","author":"wang","key":"ref13"},{"key":"ref14","first-page":"5279","article-title":"Scalable trust-region method for deep reinforcement learning using kronecker-factored approximation","author":"yuhuai","year":"0","journal-title":"Proceedings of the Advances in Neural Information Processing Systems"},{"journal-title":"Averaging weights leads to wider optima and better generalization","year":"2018","author":"izmailov","key":"ref15"},{"key":"ref16","first-page":"4496","article-title":"Distral: Robust multitask reinforcement learning","author":"yee","year":"0","journal-title":"Proceedings of the Advances in Neural Information Processing Systems"},{"journal-title":"Progressive reinforcement learning with distillation for multi-skilled motion control","year":"2018","author":"berseth","key":"ref17"},{"journal-title":"Discorl Continual reinforcement learning via policy distillation","year":"2019","author":"traore","key":"ref18"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/72.935097"},{"key":"ref28","first-page":"1929","article-title":"Dropout: a simple way to prevent neural networks from overfitting","volume":"15","author":"srivastava","year":"2014","journal-title":"Journal of Machine Learning Research"},{"journal-title":"Ray rllib A composable and scalable reinforcement learning library","year":"2017","author":"liang","key":"ref4"},{"key":"ref27","first-page":"26","article-title":"Lecture 6.5-rmsprop: Divide the gradient by a running average of its recent magnitude","volume":"4","author":"tieleman","year":"2012","journal-title":"COURSERA Neural Networks for Machine Learning"},{"journal-title":"Continuous control with deep reinforcement learning","year":"2015","author":"timothy","key":"ref3"},{"journal-title":"Proximal policy optimization algorithms","year":"2017","author":"schulman","key":"ref6"},{"journal-title":"Fixing weight decay regularization in adam","year":"2017","author":"loshchilov","key":"ref29"},{"key":"ref5","article-title":"Rainbow: Combining improvements in deep reinforcement learning","author":"hessel","year":"0","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CBI.2017.23"},{"key":"ref7","first-page":"1889","article-title":"Trust region policy optimization","author":"schulman","year":"0","journal-title":"Proceedings of the International Conference on Machine Learning"},{"key":"ref2","first-page":"1928","article-title":"Asynchronous methods for deep reinforcement learning","author":"mnih","year":"0","journal-title":"Proceedings of the International Conference on Machine Learning"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2019.2907260"},{"journal-title":"Playing atari with deep reinforcement learning","year":"2013","author":"mnih","key":"ref1"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2016.2522401"},{"journal-title":"High-dimensional continuous control using generalized advantage estimation","year":"2015","author":"schulman","key":"ref22"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683161"},{"journal-title":"Japanese Candlestick Charting Techniques A Contemporary Guide to the Ancient Investment Techniques of the Far East","year":"2001","author":"nison","key":"ref24"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-4380-9_35"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"}],"event":{"name":"2020 IEEE 30th International Workshop on Machine Learning for Signal Processing (MLSP)","start":{"date-parts":[[2020,9,21]]},"location":"Espoo, Finland","end":{"date-parts":[[2020,9,24]]}},"container-title":["2020 IEEE 30th International Workshop on Machine Learning for Signal Processing (MLSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9217888\/9231523\/09231849.pdf?arnumber=9231849","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,28]],"date-time":"2022-06-28T21:52:57Z","timestamp":1656453177000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9231849\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,9]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/mlsp49062.2020.9231849","relation":{},"subject":[],"published":{"date-parts":[[2020,9]]}}}