{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,23]],"date-time":"2025-06-23T15:06:02Z","timestamp":1750691162085,"version":"3.37.3"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2019,6,24]],"date-time":"2019-06-24T00:00:00Z","timestamp":1561334400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,6,24]],"date-time":"2019-06-24T00:00:00Z","timestamp":1561334400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2020,6]]},"DOI":"10.1007\/s00521-019-04315-5","type":"journal-article","created":{"date-parts":[[2019,6,24]],"date-time":"2019-06-24T15:02:41Z","timestamp":1561388561000},"page":"8089-8100","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Accelerating SGD using flexible variance reduction on large-scale datasets"],"prefix":"10.1007","volume":"32","author":[{"given":"Mingxing","family":"Tang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linbo","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4819-373X","authenticated-orcid":false,"given":"Zhen","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinwang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxing","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueliang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,6,24]]},"reference":[{"key":"4315_CR1","first-page":"2663","volume":"2012","author":"NL Roux","year":"2012","unstructured":"Roux NL, Schmidt M, Bach FR (2012) A stochastic gradient method with an exponential convergence rate for finite training sets. Adv Neural Inf Process Syst 2012:2663\u20132671","journal-title":"Adv Neural Inf Process Syst"},{"key":"4315_CR2","first-page":"2647","volume":"2015","author":"SJ Reddi","year":"2015","unstructured":"Reddi SJ, Hefny A, Sra S, Poczos B, Smola AJ (2015) On variance reduction in stochastic gradient descent and its asynchronous variants. Adv Neural Inf Process Syst 2015:2647\u20132655","journal-title":"Adv Neural Inf Process Syst"},{"issue":"1\u20132","key":"4315_CR3","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/s10107-016-1030-6","volume":"162","author":"M Schmidt","year":"2017","unstructured":"Schmidt M, Le Roux N, Bach F (2017) Minimizing finite sums with the stochastic average gradient. Math Program 162(1\u20132):83\u2013112","journal-title":"Math Program"},{"key":"4315_CR4","first-page":"1646","volume":"2014","author":"A Defazio","year":"2014","unstructured":"Defazio A, Bach F, Lacoste-Julien S (2014) SAGA: a fast incremental gradient method with support for non-strongly convex composite objectives. Adv Neural Inf Process Syst 2014:1646\u20131654","journal-title":"Adv Neural Inf Process Syst"},{"key":"4315_CR5","doi-asserted-by":"crossref","unstructured":"De S, Goldstein T (2016) Efficient distributed SGD with variance reduction. In: 2016 IEEE 16th international conference on data mining (ICDM), 2016. pp 111\u2013120","DOI":"10.1109\/ICDM.2016.0022"},{"key":"4315_CR6","first-page":"315","volume":"2013","author":"R Johnson","year":"2013","unstructured":"Johnson R, Zhang T (2013) Accelerating stochastic gradient descent using predictive variance reduction. Adv Neural Inf Process Syst 2013:315\u2013323","journal-title":"Adv Neural Inf Process Syst"},{"key":"4315_CR7","doi-asserted-by":"crossref","unstructured":"Tang M, Huang Z, Qiao L, Du S, Peng Y, Wang C (2018) FVR-SGD: a new flexible variance-reduction method for SGD on large-scale datasets. In: The 25th international conference on neural information processing, 01, 2018. pp 181\u2013193","DOI":"10.1007\/978-3-030-04179-3_16"},{"issue":"Jul","key":"4315_CR8","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi J, Hazan E, Singer Y (2011) Adaptive subgradient methods for online learning and stochastic optimization. J Mach Learn Res 12(Jul):2121\u20132159","journal-title":"J Mach Learn Res"},{"key":"4315_CR9","unstructured":"Zeiler MD (2012) ADADELTA: an adaptive learning rate method. \narXiv:12125701"},{"issue":"2","key":"4315_CR10","first-page":"26","volume":"4","author":"T Tieleman","year":"2012","unstructured":"Tieleman T, Hinton G (2012) Lecture 6.5-rmsprop: divide the gradient by a running average of its recent magnitude. COURSERA: Neural Netw Mach Learn 4(2):26\u201331","journal-title":"COURSERA: Neural Netw Mach Learn"},{"key":"4315_CR11","unstructured":"Kingma DP, Ba J (2014) Adam: a method for stochastic optimization. \narXiv:14126980"},{"key":"4315_CR12","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-018-3495-0","author":"T Zhang","year":"2018","unstructured":"Zhang T, Hu F, Li L, Xu X, Yang Z, Chen Y (2018) An adaptive mechanism to achieve learning rate dynamically. Neural Comput Appl. \nhttps:\/\/doi.org\/10.1007\/s00521-018-3495-0","journal-title":"Neural Comput Appl"},{"issue":"3","key":"4315_CR13","doi-asserted-by":"publisher","first-page":"A1380","DOI":"10.1137\/110830629","volume":"34","author":"MP Friedlander","year":"2012","unstructured":"Friedlander MP, Schmidt M (2012) Hybrid deterministic-stochastic methods for data fitting. SIAM J Sci Comput 34(3):A1380\u2013A1405","journal-title":"SIAM J Sci Comput"},{"key":"4315_CR14","unstructured":"De S, Yadav A, Jacobs D, Goldstein T (2016) Big batch SGD: automated inference using adaptive batch sizes. \narXiv:161005792"},{"key":"4315_CR15","doi-asserted-by":"crossref","unstructured":"Li M, Zhang T, Chen Y, Smola AJ (2014) Efficient mini-batch training for stochastic optimization. In: Proceedings of the 20th ACM SIGKDD international conference on Knowledge discovery and data mining, 2014. pp 661\u2013670","DOI":"10.1145\/2623330.2623612"},{"issue":"Jan","key":"4315_CR16","first-page":"165","volume":"13","author":"O Dekel","year":"2012","unstructured":"Dekel O, Gilad-Bachrach R, Shamir O, Xiao L (2012) Optimal distributed online prediction using mini-batches. J Mach Learn Res 13(Jan):165\u2013202","journal-title":"J Mach Learn Res"},{"issue":"1","key":"4315_CR17","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1016\/S0893-6080(98)00116-6","volume":"12","author":"N Qian","year":"1999","unstructured":"Qian N (1999) On the momentum term in gradient descent learning algorithms. Neural Netw 12(1):145\u2013151","journal-title":"Neural Netw"},{"key":"4315_CR18","first-page":"543","volume":"1983","author":"YE Nesterov","year":"1983","unstructured":"Nesterov YE (1983) A method for solving the convex programming problem with convergence rate $$O(1\/k^2)$$. Dokl. Akad. Nauk SSSR 1983:543\u2013547","journal-title":"Dokl. Akad. Nauk SSSR"},{"issue":"2","key":"4315_CR19","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1007\/s00521-009-0286-7","volume":"19","author":"Mladen Hocenski","year":"2010","unstructured":"Hocenski Mladen, Filko D (2010) Accelerated gradient learning algorithm for neural network weights update. Neural Comput Appl 19(2):219\u2013225","journal-title":"Neural Comput Appl"},{"key":"4315_CR20","first-page":"2595","volume":"2010","author":"M Zinkevich","year":"2010","unstructured":"Zinkevich M, Weimer M, Li L, Smola AJ (2010) Parallelized stochastic gradient descent. Adv Neural Inf Process Syst 2010:2595\u20132603","journal-title":"Adv Neural Inf Process Syst"},{"key":"4315_CR21","first-page":"693","volume":"2011","author":"B Recht","year":"2011","unstructured":"Recht B, Re C, Wright S, Niu F (2011) Hogwild!: a lock-free approach to parallelizing stochastic gradient descent. Adv Neural Inf Process Syst 2011:693\u2013701","journal-title":"Adv Neural Inf Process Syst"},{"key":"4315_CR22","first-page":"1223","volume":"2012","author":"J Dean","year":"2012","unstructured":"Dean J, Corrado G, Monga R, Chen K, Devin M, Mao M, Senior A, Tucker P, Yang K, Le QV (2012) Large scale distributed deep networks. Adv Neural Inf Process Syst 2012:1223\u20131231","journal-title":"Adv Neural Inf Process Syst"},{"issue":"2","key":"4315_CR23","doi-asserted-by":"publisher","first-page":"341","DOI":"10.1137\/100802001","volume":"22","author":"Y Nesterov","year":"2012","unstructured":"Nesterov Y (2012) Efficiency of coordinate descent methods on huge-scale optimization problems. SIAM J Optim 22(2):341\u2013362","journal-title":"SIAM J Optim"},{"issue":"Feb","key":"4315_CR24","first-page":"567","volume":"14","author":"S Shalev-Shwartz","year":"2013","unstructured":"Shalev-Shwartz S, Zhang T (2013) Stochastic dual coordinate ascent methods for regularized loss minimization. J Mach Learn Res 14(Feb):567\u2013599","journal-title":"J Mach Learn Res"},{"key":"4315_CR25","first-page":"1073","volume":"2018","author":"T Xie","year":"2018","unstructured":"Xie T, Liu B, Xu Y, Ghavamzadeh M, Chow Y, Lyu D, Yoon D (2018) A block coordinate ascent algorithm for mean-variance optimization. Adv Neural Inf Process Syst 2018:1073\u20131083","journal-title":"Adv Neural Inf Process Syst"},{"issue":"1","key":"4315_CR26","first-page":"2939","volume":"18","author":"Y Zhang","year":"2017","unstructured":"Zhang Y, Xiao L (2017) Stochastic primal-dual coordinate method for regularized empirical risk minimization. J Mach Learn Res 18(1):2939\u20132980","journal-title":"J Mach Learn Res"},{"issue":"2","key":"4315_CR27","first-page":"1865","volume":"12","author":"S Shalev-Shwartz","year":"2011","unstructured":"Shalev-Shwartz S, Tewari A (2011) Stochastic methods for $$\\ell _1$$ regularized loss minimization. J Mach Learn Res 12(2):1865\u20131892","journal-title":"J Mach Learn Res"},{"issue":"4","key":"4315_CR28","doi-asserted-by":"publisher","first-page":"2057","DOI":"10.1137\/140961791","volume":"24","author":"L Xiao","year":"2014","unstructured":"Xiao L, Zhang T (2014) A proximal stochastic gradient method with progressive variance reduction. SIAM J Optim 24(4):2057\u20132075","journal-title":"SIAM J Optim"},{"key":"4315_CR29","first-page":"64","volume":"2014","author":"S Shalev-Shwartz","year":"2014","unstructured":"Shalev-Shwartz S, Zhang T (2014) Accelerated proximal stochastic dual coordinate ascent for regularized loss minimization. Int Conf Mach Learn 2014:64\u201372","journal-title":"Int Conf Mach Learn"},{"key":"4315_CR30","first-page":"3185","volume":"2018","author":"X Liu","year":"2018","unstructured":"Liu X, Hsieh C-J (2018) Fast variance reduction method with Stochastic batch size. Int Conf Mach Learn 2018:3185\u20133194","journal-title":"Int Conf Mach Learn"},{"key":"4315_CR31","unstructured":"Shang F, Zhou K, Cheng J, Tsang IW, Zhang L, Tao D (2018) VR-SGD: a simple stochastic variance reduction method for machine learning. \narXiv:1802.09932"},{"key":"4315_CR32","unstructured":"De S, Taylor G, Goldstein T (2015) Variance reduction for distributed stochastic gradient descent. \narXiv:151201708"},{"key":"4315_CR33","doi-asserted-by":"crossref","unstructured":"Zhao K, Matsukawa T, Suzuki E (2018) Retraining: a simple way to improve the ensemble accuracy of deep neural networks for image classification. In: 2018 24th international conference on pattern recognition (ICPR), 2018. IEEE, pp 860\u2013867","DOI":"10.1109\/ICPR.2018.8545535"},{"key":"4315_CR34","doi-asserted-by":"crossref","unstructured":"Zhang T, Wiliem A, Yang S, Lovell B (2018) Tv-gan: generative adversarial network based thermal to visible face recognition. In: 2018 international conference on biometrics (ICB), 2018. IEEE, pp 174\u2013181","DOI":"10.1109\/ICB2018.2018.00035"},{"issue":"1","key":"4315_CR35","doi-asserted-by":"publisher","first-page":"333","DOI":"10.1007\/s00521-012-0915-4","volume":"22","author":"X Yu","year":"2013","unstructured":"Yu X, Deng F (2013) Convergence of gradient method for training ridge polynomial neural network. Neural Comput Appl 22(1):333\u2013339","journal-title":"Neural Comput Appl"},{"issue":"11","key":"4315_CR36","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun Y, Bottou L, Bengio Y, Haffner P (1998) Gradient-based learning applied to document recognition. Proc IEEE 86(11):2278\u20132324","journal-title":"Proc IEEE"},{"key":"4315_CR37","unstructured":"Krizhevsky A, Hinton G (2010) Convolutional deep belief networks on cifar-10. Unpublished manuscript 40(7)"},{"issue":"Apr","key":"4315_CR38","first-page":"361","volume":"5","author":"DD Lewis","year":"2004","unstructured":"Lewis DD, Yang Y, Rose TG, Li F (2004) Rcv1: a new benchmark collection for text categorization research. J Mach Learn Res 5(Apr):361\u2013397","journal-title":"J Mach Learn Res"},{"key":"4315_CR39","doi-asserted-by":"crossref","unstructured":"Lang K (1995) Newsweeder: learning to filter netnews. In: Machine learning proceedings 1995. Elsevier, pp 331\u2013339","DOI":"10.1016\/B978-1-55860-377-6.50048-7"},{"issue":"20","key":"4315_CR40","doi-asserted-by":"publisher","first-page":"11462","DOI":"10.1073\/pnas.201162998","volume":"98","author":"M West","year":"2001","unstructured":"West M, Blanchette C, Dressman H, Huang E, Ishida S, Spang R, Zuzan H, Olson JA, Marks JR, Nevins JR (2001) Predicting the clinical status of human breast cancer by using gene expression profiles. Proc Natl Acad Sci 98(20):11462\u201311467","journal-title":"Proc Natl Acad Sci"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-019-04315-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00521-019-04315-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-019-04315-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,6,22]],"date-time":"2020-06-22T23:25:09Z","timestamp":1592868309000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00521-019-04315-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,6,24]]},"references-count":40,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2020,6]]}},"alternative-id":["4315"],"URL":"https:\/\/doi.org\/10.1007\/s00521-019-04315-5","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"type":"print","value":"0941-0643"},{"type":"electronic","value":"1433-3058"}],"subject":[],"published":{"date-parts":[[2019,6,24]]},"assertion":[{"value":"20 January 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 June 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 June 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with ethical standards"}},{"value":"We declare that we have no financial and personal relationships with other people or organizations that can inappropriately influence our work, and there is no professional or other personal interest of any nature or kind in any product, service and\/or company that could be construed as influencing the position presented in, or the review of, the manuscript entitled \u201cAccelerating SGD using Flexible Variance Reduction on Large-Scale Datasets.\u201d","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}