{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T23:15:31Z","timestamp":1771024531281,"version":"3.50.1"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2022,11,10]],"date-time":"2022-11-10T00:00:00Z","timestamp":1668038400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,11,10]],"date-time":"2022-11-10T00:00:00Z","timestamp":1668038400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"PRAIRIE 3IA Insitute","award":["ANR-19-P3IA-0001"],"award-info":[{"award-number":["ANR-19-P3IA-0001"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1007\/s10994-022-06265-x","type":"journal-article","created":{"date-parts":[[2022,11,10]],"date-time":"2022-11-10T23:02:52Z","timestamp":1668121372000},"page":"4359-4409","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["SVRG meets AdaGrad: painless variance reduction"],"prefix":"10.1007","volume":"111","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5931-8695","authenticated-orcid":false,"given":"Benjamin","family":"Dubois-Taine","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sharan","family":"Vaswani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Reza","family":"Babanezhad","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mark","family":"Schmidt","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Simon","family":"Lacoste-Julien","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,11,10]]},"reference":[{"key":"6265_CR1","unstructured":"Ahn, K., Yun, C., & Sra, S. (2020). SGD with shuffling: Optimal rates without component convexity and large epoch requirements. In Neural information processing systems 2020, NeurIPS 2020."},{"key":"6265_CR2","doi-asserted-by":"crossref","unstructured":"Allen-Zhu, Z. (2017). Katyusha: The first direct acceleration of stochastic gradient methods. In Proceedings of the 49th annual ACM SIGACT symposium on theory of computing, STOC.","DOI":"10.1145\/3055399.3055448"},{"issue":"1","key":"6265_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.2140\/pjm.1966.16.1","volume":"16","author":"L Armijo","year":"1966","unstructured":"Armijo, L. (1966). Minimization of functions having lipschitz continuous first partial derivatives. Pacific Journal of mathematics, 16(1), 1\u20133.","journal-title":"Pacific Journal of mathematics"},{"key":"6265_CR4","first-page":"2251","volume":"28","author":"R Babanezhad Harikandeh","year":"2015","unstructured":"Babanezhad Harikandeh, R., Ahmed, M. O., Virani, A., Schmidt, M., Kone\u010dn\u1ef3, J., & Sallinen, S. (2015). Stop wasting my gradients: Practical SVRG. Advances in Neural Information Processing Systems, 28, 2251\u20132259.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"1","key":"6265_CR5","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1093\/imanum\/8.1.141","volume":"8","author":"J Barzilai","year":"1988","unstructured":"Barzilai, J., & Borwein, J. M. (1988). Two-point step size gradient methods. IMA Journal of Numerical Analysis, 8(1), 141\u2013148.","journal-title":"IMA Journal of Numerical Analysis"},{"key":"6265_CR6","unstructured":"Belkin, M., Rakhlin, A., & Tsybakov, A.\u00a0B. (2019). Does data interpolation contradict statistical optimality? In The 22nd international conference on artificial intelligence and statistics (pp. 1611\u20131619). PMLR."},{"issue":"2","key":"6265_CR7","doi-asserted-by":"publisher","first-page":"545","DOI":"10.1093\/imanum\/dry009","volume":"39","author":"R Bollapragada","year":"2019","unstructured":"Bollapragada, R., Byrd, R. H., & Nocedal, J. (2019). Exact and inexact subsampled newton methods for optimization. IMA Journal of Numerical Analysis, 39(2), 545\u2013578.","journal-title":"IMA Journal of Numerical Analysis"},{"key":"6265_CR8","doi-asserted-by":"publisher","first-page":"323","DOI":"10.1051\/ps:2005018","volume":"9","author":"S Boucheron","year":"2005","unstructured":"Boucheron, S., Bousquet, O., & Lugosi, G. (2005). Theory of classification: A survey of some recent advances. ESAIM: Probability and Statistics, 9, 323\u2013375.","journal-title":"ESAIM: Probability and Statistics"},{"key":"6265_CR9","doi-asserted-by":"crossref","unstructured":"Chang, C.-C., & Lin, C.-J. (2011). LIBSVM: A library for support vector machines. ACM Transactions on Intelligent Systems and Technology, 2(3), 1\u201327. Software available at http:\/\/www.csie.ntu.edu.tw\/~cjlin\/libsvm.","DOI":"10.1145\/1961189.1961199"},{"key":"6265_CR10","unstructured":"Cutkosky, A., & Boahen, K. (2017). Online convex optimization with unconstrained domains and losses. arXiv preprint arXiv:1703.02622."},{"key":"6265_CR11","unstructured":"Cutkosky, A., & Orabona, F. (2019). Momentum-based variance reduction in non-convex SGD. arXiv preprint arXiv:1905.10018."},{"key":"6265_CR12","unstructured":"Defazio, A., & Bottou, L. (2019). On the ineffectiveness of variance reduced optimization for deep learning. In NeurIPS: In advances in neural information processing systems."},{"key":"6265_CR13","unstructured":"Defazio, A., Bach, F., & Lacoste-Julien, S. (2014). SAGA: A fast incremental gradient method with support for non-strongly convex composite objectives. In NeurIPS: In Advances in neural information processing systems."},{"key":"6265_CR14","doi-asserted-by":"crossref","unstructured":"Dubois-Taine, B., Vaswani, S., Babanezhad, R., Schmidt, M., & Lacoste-Julien, S. (2021). Svrg meets adagrad: Painless variance reduction. arXiv preprint arXiv:2102.09645.","DOI":"10.1007\/s10994-022-06265-x"},{"key":"6265_CR15","first-page":"2121","volume":"12","author":"JC Duchi","year":"2011","unstructured":"Duchi, J. C., Hazan, E., & Singer, Y. (2011). Adaptive subgradient methods for online learning and stochastic optimization. The Journal of Machine Learning Research, 12, 2121\u20132159.","journal-title":"The Journal of Machine Learning Research"},{"issue":"11","key":"6265_CR16","doi-asserted-by":"publisher","first-page":"1968","DOI":"10.1109\/JPROC.2020.3028013","volume":"108","author":"RM Gower","year":"2020","unstructured":"Gower, R. M., Schmidt, M., Bach, F., & Richtarik, P. (2020). Variance-reduced methods for machine learning. Proceedings of the IEEE, 108(11), 1968\u20131983.","journal-title":"Proceedings of the IEEE"},{"key":"6265_CR17","first-page":"2305","volume":"28","author":"T Hofmann","year":"2015","unstructured":"Hofmann, T., Lucchi, A., Lacoste-Julien, S., & McWilliams, B. (2015). Variance reduced stochastic gradient descent with neighbors. Advances in Neural Information Processing Systems, 28, 2305\u20132313.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"6265_CR18","unstructured":"Johnson, R., & Zhang, T. (2013). Accelerating stochastic gradient descent using predictive variance reduction. In NeurIPS: In advances in neural information processing systems."},{"key":"6265_CR19","unstructured":"Kone\u010dn\u1ef3, J., & Richt\u00e1rik, P. (2013). Semi-stochastic gradient descent methods. arXiv preprint arXiv:1312.1666."},{"key":"6265_CR20","unstructured":"Kovalev, D., Horv\u00e1th, S., & Richt\u00e1rik, P. (2020). Don\u2019t jump through hoops and remove those loops: SVRG and Katyusha are better without the outer loop. In Algorithmic learning theory (pp. 451\u2013467). PMLR."},{"key":"6265_CR21","unstructured":"Lan, G., Li, Z., & Zhou, Y. (2019). A unified variance-reduced accelerated gradient method for convex optimization. In Advances in neural information processing systems (pp. 10462\u201310472)."},{"key":"6265_CR22","unstructured":"Lang, H., Xiao, L., & Zhang, P. (2019). Using statistics to automate stochastic optimization. In Advances in neural information processing systems (pp. 9540\u20139550)."},{"key":"6265_CR23","unstructured":"Levy, K. Y., Yurtsever, A., & Cevher, V. (2018). Online adaptive methods, universality and acceleration. In NeurIPS: In advances in neural information processing systems."},{"key":"6265_CR24","unstructured":"Li, B., Wang, L., & Giannakis, G.\u00a0B. (2020). Almost tune-free variance reduction. In International conference on machine learning (pp. 5969\u20135978). PMLR."},{"issue":"3","key":"6265_CR25","doi-asserted-by":"publisher","first-page":"1329","DOI":"10.1214\/19-AOS1849","volume":"48","author":"T Liang","year":"2020","unstructured":"Liang, T., Rakhlin, A., et al. (2020). Just interpolate: Kernel \u201cridgeless\u2019\u2019 regression can generalize. Annals of Statistics, 48(3), 1329\u20131347.","journal-title":"Annals of Statistics"},{"key":"6265_CR26","unstructured":"Liu, M., Zhang, W., Orabona, F., & Yang, T. (2020). Adam+: A stochastic method with adaptive variance reduction. arXiv preprint arXiv:2011.11985."},{"key":"6265_CR27","unstructured":"Loizou, N., Vaswani, S., Laradji, I., & Lacoste-Julien, S. (2020). Stochastic Polyak step-size for SGD: An adaptive learning rate for fast convergence. arXiv preprint: arXiv:2002.10542."},{"key":"6265_CR28","unstructured":"Ma, S., Bassily, R., & Belkin, M. (2018). The power of interpolation: Understanding the effectiveness of SGD in modern over-parametrized learning. In Proceedings of the 35th international conference on machine learning, ICML."},{"key":"6265_CR29","unstructured":"Mahdavi, M., & Jin, R. (2013). MixedGrad: An O(1\/T) convergence rate algorithm for stochastic smooth optimization. arXiv preprint arXiv:1307.7192."},{"key":"6265_CR30","unstructured":"Mairal, J. (2013). Optimization with first-order surrogate functions. In International conference on machine learning (pp. 783\u2013791)."},{"key":"6265_CR31","unstructured":"Meng, S.\u00a0Y., Vaswani, S., Laradji, I., Schmidt, M., & Lacoste-Julien, S. (2020). Fast and furious convergence: Stochastic second order methods under interpolation. In The 23nd international conference on artificial intelligence and statistics, AISTATS."},{"key":"6265_CR32","unstructured":"Moulines, E., & Bach, F. R. (2011). Non-asymptotic analysis of stochastic approximation algorithms for machine learning. In NeurIPS: In advances in neural information processing systems."},{"key":"6265_CR33","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4419-8853-9","volume-title":"Introductory lectures on convex optimization: A basic course","author":"Y Nesterov","year":"2004","unstructured":"Nesterov, Y. (2004). Introductory lectures on convex optimization: A basic course. Berlin: Springer."},{"key":"6265_CR34","unstructured":"Nguyen, L.\u00a0M., Liu, J., Scheinberg, K., & Tak\u00e1\u010d, M. (2017). SARAH: a novel method for machine learning problems using stochastic recursive gradient. In Proceedings of the 34th international conference on machine learning (Vol. 70, pp. 2613\u20132621)."},{"key":"6265_CR35","unstructured":"Pesme, S., Dieuleveut, A., & Flammarion, N. (2020). On convergence-diagnostic based step sizes for stochastic gradient descent. arXiv preprint arXiv:2007.00534."},{"key":"6265_CR36","unstructured":"Pflug, G.\u00a0C. (1983). On the determination of the step size in stochastic quasigradient methods. collaborative paper cp-83-025. International Institute for Applied Systems Analysis (IIASA), Laxenburg, Austria."},{"key":"6265_CR37","unstructured":"Qian, Q., & Qian, X. (2019). The implicit bias of adagrad on separable data. arXiv preprint arXiv:1906.03559."},{"key":"6265_CR38","doi-asserted-by":"crossref","unstructured":"Reddi, S.\u00a0J., Hefny, A., Sra, S., Poczos, B., & Smola, A. (2016). Stochastic variance reduction for nonconvex optimization. In International conference on machine learning (pp. 314\u2013323).","DOI":"10.1109\/ALLERTON.2016.7852377"},{"key":"6265_CR39","unstructured":"Reddi, S.\u00a0J., Kale, S., & Kumar, S. (2018). On the convergence of Adam and Beyond. In International conference on learning representations."},{"key":"6265_CR40","unstructured":"Schmidt, M., & Le\u00a0Roux, N. (2013). Fast convergence of stochastic gradient descent under a strong growth condition. arXiv preprint: arXiv:1308.6370."},{"key":"6265_CR41","unstructured":"Schmidt, M., Babanezhad, R., Ahmed, M., Defazio, A., Clifton, A., & Sarkar, A. (2015). Non-uniform stochastic average gradient method for training conditional random fields. In Proceedings of the eighteenth international conference on artificial intelligence and statistics, AISTATS."},{"issue":"1\u20132","key":"6265_CR42","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/s10107-016-1030-6","volume":"162","author":"M Schmidt","year":"2017","unstructured":"Schmidt, M., Le Roux, N., & Bach, F. (2017). Minimizing finite sums with the stochastic average gradient. Mathematical Programming, 162(1\u20132), 83\u2013112.","journal-title":"Mathematical Programming"},{"key":"6265_CR43","unstructured":"Sebbouh, O., Gazagnadou, N., Jelassi, S., Bach, F., & Gower, R. (2019). Towards closing the gap between the theory and practice of SVRG. In Advances in neural information processing systems (pp. 648\u2013658)."},{"issue":"Feb","key":"6265_CR44","first-page":"567","volume":"14","author":"S Shalev-Shwartz","year":"2013","unstructured":"Shalev-Shwartz, S., & Zhang, T. (2013). Stochastic dual coordinate ascent methods for regularized loss minimization. Journal of Machine Learning Research, 14(Feb), 567\u2013599.","journal-title":"Journal of Machine Learning Research"},{"key":"6265_CR45","unstructured":"Song, C., Jiang, Y., & Ma, Y. (2020). Variance reduction via accelerated dual averaging for finite-sum optimization. Advances in Neural Information Processing Systems, 33."},{"key":"6265_CR46","first-page":"1545","volume":"21","author":"K Sridharan","year":"2008","unstructured":"Sridharan, K., Shalev-Shwartz, S., & Srebro, N. (2008). Fast rates for regularized objectives. Advances in Neural Information Processing Systems, 21, 1545\u20131552.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"6265_CR47","unstructured":"Tan, C., Ma, S., Dai, Y.-H., & Qian, Y. (2016). Barzilai-Borwein step size for stochastic gradient descent. arXiv preprint arXiv:1605.04131."},{"key":"6265_CR48","unstructured":"Vaswani, S., Bach, F., & Schmidt, M. (2019a). Fast and faster convergence of SGD for over-parameterized models and an accelerated perceptron. In The 22nd International Conference on Artificial Intelligence and Statistics, AISTATS."},{"key":"6265_CR49","unstructured":"Vaswani, S., Kunstner, F., Laradji, I., Meng, S.\u00a0Y., Schmidt, M., & Lacoste-Julien, S. (2020). Adaptive gradient methods converge faster with over-parameterization (and you can do a line-search). arXiv preprint arXiv:2006.06835."},{"key":"6265_CR50","unstructured":"Vaswani, S., Mishkin, A., Laradji, I., Schmidt, M., Gidel, G., & Lacoste-Julien, S. (2019). Painless stochastic gradient: Interpolation, line-search, and convergence rates. In NeurIPS: In advances in neural information processing systems."},{"key":"6265_CR51","unstructured":"Ward, R., Wu, X., & Bottou, L. (2019). AdaGrad stepsizes: Sharp convergence over nonconvex landscapes, from any initialization. In Proceedings of the 36th international conference on machine learning, ICML."},{"key":"6265_CR52","unstructured":"Yaida, S. (2018). Fluctuation-dissipation relations for stochastic gradient descent. arXiv preprint arXiv:1810.00004."},{"key":"6265_CR53","unstructured":"Zhang, C., Bengio, S., Hardt, M., Recht, B., & Vinyals, O. (2017). Understanding deep learning requires rethinking generalization. In 5th international conference on learning representations, ICLR."},{"key":"6265_CR54","unstructured":"Zhou, K., So, A. M.-C., & Cheng, J. (2021). Accelerating perturbed stochastic iterates in asynchronous lock-free optimization. arXiv preprint arXiv:2109.15292."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-022-06265-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-022-06265-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-022-06265-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,10]],"date-time":"2023-11-10T01:10:24Z","timestamp":1699578624000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-022-06265-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,10]]},"references-count":54,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["6265"],"URL":"https:\/\/doi.org\/10.1007\/s10994-022-06265-x","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,11,10]]},"assertion":[{"value":"20 February 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 September 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 November 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}