{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T05:30:17Z","timestamp":1730266217627,"version":"3.28.0"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,7,1]],"date-time":"2020-07-01T00:00:00Z","timestamp":1593561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,7,1]],"date-time":"2020-07-01T00:00:00Z","timestamp":1593561600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,7,1]],"date-time":"2020-07-01T00:00:00Z","timestamp":1593561600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,7]]},"DOI":"10.1109\/ijcnn48605.2020.9207166","type":"proceedings-article","created":{"date-parts":[[2020,9,30]],"date-time":"2020-09-30T00:40:33Z","timestamp":1601426433000},"page":"1-8","source":"Crossref","is-referenced-by-count":1,"title":["On the Trend-corrected Variant of Adaptive Stochastic Optimization Methods"],"prefix":"10.1109","author":[{"given":"Bingxin","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuebin","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junbin","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","article-title":"On the variance of the adaptive learning rate and beyond","author":"liu","year":"2020","journal-title":"International Conference on Learning Representations"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijforecast.2003.09.015"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1287\/mnsc.31.10.1237"},{"journal-title":"Forecasting Principles and Practice","year":"2018","author":"hyndman","key":"ref13"},{"key":"ref14","article-title":"Fashion-MNIST: a novel image dataset for benchmarking machine learning algorithms","author":"xiao","year":"2017","journal-title":"Preprints arXiv 1708 07747"},{"key":"ref15","article-title":"Reading digits in natural images with unsupervised feature learning","author":"netzer","year":"2011","journal-title":"NIPS Workshop on Deep Learning and Unsupervised Feature Learning"},{"key":"ref16","first-page":"1929","article-title":"Dropout: a simple way to prevent neural networks from over-fitting","volume":"15","author":"srivastava","year":"2014","journal-title":"The Journal of Machine Learning Research"},{"key":"ref17","article-title":"Auto-encoding variational bayes","author":"kingma","year":"2014","journal-title":"Proceedings of International Conference on Learning Representations"},{"key":"ref18","article-title":"Stochastic backpropagation and approximate inference in deep generative models","author":"rezende","year":"2014","journal-title":"Proceedings of the 31st International Conference on Machine Learning"},{"key":"ref19","first-page":"3821","article-title":"Stochastic convex optimization: Faster local growth implies faster global convergence","author":"xu","year":"0"},{"key":"ref4","article-title":"Adadelta: an adaptive learning rate method","author":"zeiler","year":"2012","journal-title":"preprint arXiv 1212 5701"},{"key":"ref3","first-page":"26","article-title":"Lecture 6.5-rmsprop: Divide the gradient by a running average of its recent magnitude","volume":"4","author":"tieleman","year":"2012","journal-title":"COURSERA Neural Networks for Machine Learning"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/S0893-6080(98)00116-6"},{"key":"ref5","article-title":"ADAM: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"Proc of the Int Conf on Learning Representations (ICLR)"},{"key":"ref8","article-title":"Incorporating Nesterov momentum into ADAM","author":"dozat","year":"2016","journal-title":"International Conference on Learning Representations Workshops Track"},{"key":"ref7","article-title":"On the convergence of ADAM and beyond","author":"reddi","year":"2018","journal-title":"Proc of the Int Conf on Learning Representations (ICLR)"},{"key":"ref2","article-title":"An overview of gradient descent optimization algorithms","author":"ruder","year":"2016","journal-title":"preprint arXiv 1609 04747"},{"key":"ref9","article-title":"Decoupled weight decay regularization","author":"loshchilov","year":"2019","journal-title":"International Conference on Learning Representations"},{"key":"ref1","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"duchi","year":"2011","journal-title":"Journal of Machine Learning Research"},{"key":"ref20","first-page":"912","article-title":"Sadagrad: Strongly adaptive stochastic gradient methods","author":"chen","year":"2018","journal-title":"International Conference on Machine Learning"},{"key":"ref22","article-title":"Closing the generalization gap of adaptive gradient methods in training deep neural networks","author":"chen","year":"2018","journal-title":"preprint arXiv 1806 06763"},{"key":"ref21","first-page":"6500","article-title":"Online adaptive methods, universality and acceleration","author":"levy","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref24","article-title":"On the convergence of weighted AdaGrad with momentum for training deep neural networks","author":"zou","year":"2018","journal-title":"preprint arXiv 1808 03408"},{"key":"ref23","article-title":"On the convergence of adaptive gradient methods for nonconvex optimization","author":"zhou","year":"2018","journal-title":"preprint arXiv 1808 05671"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01138"},{"key":"ref25","article-title":"On the convergence of a class of ADAM-type algorithms for non-convex optimization","author":"chen","year":"2019","journal-title":"Proc of the Int Conf on Learning Representations (ICLR)"}],"event":{"name":"2020 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2020,7,19]]},"location":"Glasgow, United Kingdom","end":{"date-parts":[[2020,7,24]]}},"container-title":["2020 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9200848\/9206590\/09207166.pdf?arnumber=9207166","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,28]],"date-time":"2022-06-28T21:53:04Z","timestamp":1656453184000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9207166\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,7]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/ijcnn48605.2020.9207166","relation":{},"subject":[],"published":{"date-parts":[[2020,7]]}}}