{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,10]],"date-time":"2026-02-10T17:46:09Z","timestamp":1770745569399,"version":"3.49.0"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2022,8,22]],"date-time":"2022-08-22T00:00:00Z","timestamp":1661126400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,8,22]],"date-time":"2022-08-22T00:00:00Z","timestamp":1661126400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1007\/s10994-022-06227-3","type":"journal-article","created":{"date-parts":[[2022,8,22]],"date-time":"2022-08-22T16:03:24Z","timestamp":1661184204000},"page":"4639-4677","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Variance reduction on general adaptive stochastic mirror descent"],"prefix":"10.1007","volume":"111","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1872-4595","authenticated-orcid":false,"given":"Wenjie","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhanyu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yichen","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guang","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,22]]},"reference":[{"key":"6227_CR1","unstructured":"Allen-Zhu, Z. (2017). Natasha: Faster non-convex stochastic optimization via strongly non-convex parameter. In Proceedings of international conference on machine learning."},{"key":"6227_CR2","unstructured":"Allen-Zhu, Z. (2018). Natasha 2: Faster non-convex optimization than SGD. Advances in Neural Information Processing Systems."},{"key":"6227_CR3","unstructured":"Asi, H., & Duchi, J. C. (2019). Modeling simple structures and geometry for better stochastic optimization algorithms. In Proceedings of international conference on artificial intelligence and statistics."},{"key":"6227_CR4","unstructured":"Chen, X., Liu, S., Sun, R., & Hong, M. (2019). On the convergence of a class of adam-type algorithm for non-convex optimization."},{"key":"6227_CR5","unstructured":"Defazio, A., Bach, F., & Lacoste-Julien, S. (2014). Saga: A fast incremental gradient method with support for non-strongly convex composite objectives. Advances in Neural Information Processing Systems"},{"key":"6227_CR6","doi-asserted-by":"crossref","unstructured":"Dubois-Taine, B., Vaswani, S., Babanezhad, R., Schmidt, M., & Lacoste-Julien, S. (2021). SVRG meets AdaGrad: Painless variance reduction. arXiv: 2102.09645","DOI":"10.1007\/s10994-022-06265-x"},{"key":"6227_CR7","unstructured":"Duchi, J., Shalev-Shwartz, S., Singer, Y., & Tewari, A. (2010). Composite objective mirror descent. In Proceedings of the twenty third annual conference on computational learning theory."},{"key":"6227_CR8","unstructured":"Duchi, J., Hazan, E., & Singer, Y. (2011). Adaptive subgradient methods for online learning and stochastic optimization. Journal of Machine Learning Research12(7)."},{"key":"6227_CR9","unstructured":"Fang, C., Li, C. J., Lin, Z., & Zhang, T. (2018). SPIDER: Near-optimal non-convex optimization via stochastic path-integrated differential estimator. Advances in Neural Information Processing Systems."},{"key":"6227_CR10","unstructured":"Ge, R., Li, Z., Wang, W., & Wang, X. (2019). Stabilized SVRG: Simple variance reduction for nonconvex optimization. In Conference on learning theory."},{"key":"6227_CR11","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1023\/A:1026319107706","volume":"53","author":"C Gentile","year":"2003","unstructured":"Gentile, C. (2003). The robustness of the $$p$$-norm algorithms. Machine Learning, 53, 265\u2013299.","journal-title":"Machine Learning"},{"key":"6227_CR12","doi-asserted-by":"crossref","unstructured":"Ghadimi, S., Lan, G., & Zhang, H. (2016). Mini-batch stochastic approximation methods for nonconvex stochastic composite optimization. arXiv preprint arXiv:1308.6594","DOI":"10.1007\/s10107-014-0846-1"},{"key":"6227_CR13","unstructured":"Goyal, P., Doll\u00e1r, P., Girshick, R., Noordhuis, P., Wesolowski, L., Kyrola, A., Tulloch, A., Jia, Y., & He, K. (2017). Accurate, large minibatch SGD: Training ImageNet in 1 hour. 1706.02677."},{"key":"6227_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2016.90"},{"key":"6227_CR15","doi-asserted-by":"crossref","unstructured":"Huang, H., Wang, C., & Dong, B. (2019). Nostalgic adam: Weighting more of the past gradients when designing the adaptive learning rate. arXiv preprint arXiv: 1805.07557","DOI":"10.24963\/ijcai.2019\/355"},{"key":"6227_CR16","unstructured":"Johnson, R., & Zhang, T. (2013). Accelerating stochastic gradient descent using predictive variance reduction. Advances in Neural Information Processing Systems."},{"key":"6227_CR17","doi-asserted-by":"crossref","unstructured":"Karimi, H., Nutini, J., & Schmidt, M. (2016). Linear convergence of gradient and proximal-gradient methods under the Polyak-\u0142ojasiewicz condition.","DOI":"10.1007\/978-3-319-46128-1_50"},{"key":"6227_CR18","unstructured":"Kingma, D. P., & Ba, J. L. (2015). Adam: A method for stochastic optimization. In Proceedings of the 3rd international conference on learning representations (ICLR)."},{"key":"6227_CR19","unstructured":"Krizhevsky, A., Nair, V., & Hinton, G. (2009). Cifar-10 (Canadian Institute for Advanced Research). 1."},{"issue":"11","key":"6227_CR20","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., & Haffner, P. (1998). Gradient-based learning applied to document recognition. Proceedings of the IEEE, 86(11), 2278\u20132324.","journal-title":"Proceedings of the IEEE"},{"issue":"2","key":"6227_CR21","doi-asserted-by":"publisher","first-page":"1473","DOI":"10.1137\/19M1256919","volume":"30","author":"L Lei","year":"2020","unstructured":"Lei, L., & Jordan, M. I. (2020). On the adaptivity of stochastic gradient-based optimization. SIAM Journal on Optimization, 30(2), 1473\u20131500.","journal-title":"SIAM Journal on Optimization"},{"key":"6227_CR22","first-page":"2348","volume":"30","author":"L Lei","year":"2017","unstructured":"Lei, L., Ju, C., Chen, J., & Jordan, M. I. (2017). Non-convex finite-sum optimization via SCSG methods. Advances in Neural Information Processing Systems, 30, 2348\u20132358.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"6227_CR23","unstructured":"Li, W., Zhang, Z., Wang, X., & Luo, P. (2020). Adax: Adaptive gradient descent with exponential long term memory. arXiv preprint arXiv:2004.09740"},{"key":"6227_CR24","first-page":"5564","volume":"31","author":"Z Li","year":"2018","unstructured":"Li, Z., & Li, J. (2018). A simple proximal stochastic gradient method for nonsmooth nonconvex optimization. Advances in Neural Information Processing Systems, 31, 5564\u20135574.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"6227_CR25","unstructured":"Li, Z., Bao, H., Zhang, X., & Richt\u00e1rik, P. (2021). PAGE: A simple and optimal probabilistic gradient estimator for nonconvex optimization. In International conference on machine learning."},{"key":"6227_CR26","unstructured":"Liu, L., Jiang, H., He, P., Chen, W., Liu, X., Gao, J., & Han, J. (2019). On the variance of the adaptive learning rate and beyond. arXiv preprint arXiv:1908.03265"},{"key":"6227_CR27","unstructured":"Liu, M., Zhang, W., Orabona, F., & Yang, T. (2020). Adam$$^+$$: A stochastic method with adaptive variance reduction. arXiv: 2011.11985"},{"key":"6227_CR28","unstructured":"Loshchilov, I., & Hutter, F. (2016). SGDR: Stochastic gradient descent with warm restarts. In International conference on learning representations."},{"key":"6227_CR29","unstructured":"Luo, L., Xiong, Y., Liu, Y., & Sun, X. (2019). Adaptive gradient methods with dynamic bound of learning rate. In Proceedings of 7th international conference on learning representations."},{"key":"6227_CR30","unstructured":"Nair, V., & Hinton, G. (2010). Rectified linear units improve restricted Boltzmann machines. In Proceedings of 27th international conference on machine learning (ICML)."},{"issue":"110","key":"6227_CR31","first-page":"1","volume":"21","author":"NH Pham","year":"2020","unstructured":"Pham, N. H., Nguyen, L. M., Phan, D. T., & Tran-Dinh, Q. (2020). ProxSARAH: An efficient algorithmic framework for stochastic composite nonconvex optimization. Journal of Machine Learning Research, 21(110), 1\u201348.","journal-title":"Journal of Machine Learning Research"},{"key":"6227_CR32","unstructured":"Pham, N. H., Nguyen, L. M., Phan, D. T., & Tran-Dinh, Q. (2021). A hybrid stochastic optimization framework for composite nonconvex optimization. Mathematical Programming A."},{"key":"6227_CR33","unstructured":"Polyak, B. T. (1963). Gradient methods for minimizing functionals. Zhurnal Vychislitel\u2019noi Matematiki i Matematicheskoi Fiziki."},{"key":"6227_CR34","unstructured":"Reddi, S., Sra, S., Poczos, B., & Smola, A. J. (2016b). Proximal stochastic methods for nonsmooth nonconvex finite-sum optimization. Advances in Neural Information Processing Systems."},{"key":"6227_CR35","doi-asserted-by":"crossref","unstructured":"Reddi, S. J., Hefny, A., Sra, S., Poczos, B., & Smola, A. (2016). Stochastic variance reduction for nonconvex optimization. In International conference on machine learning (pp. 314\u2013323).","DOI":"10.1109\/ALLERTON.2016.7852377"},{"key":"6227_CR36","unstructured":"Sch\u00f6lkopf, B., & Smola, A. J. (2002). Learning with kernels: Support vector machines, regularization, optimization, and beyond. MIT Press."},{"issue":"2","key":"6227_CR37","first-page":"26","volume":"4","author":"T Tieleman","year":"2012","unstructured":"Tieleman, T., & Hinton, G. (2012). RMSProp: Divide the gradient by a running average of its recent magnitude. COURSERA: Neural Networks for Machine Learning, 4(2), 26\u201331.","journal-title":"COURSERA: Neural Networks for Machine Learning"},{"key":"6227_CR38","unstructured":"Wang, Z., Ji, K., Zhou, Y., Liang, Y., & Tarokh, V. (2019). SpiderBoost and momentum: Faster variance reduction algorithms."},{"key":"6227_CR39","unstructured":"Wilson, A. C., Roelofs, R., Stern, M., Srebro, N., & Recht, B. (2017). The marginal value of adaptive gradient methods in machine learning. Advances in Neural Information Processing Systems."},{"key":"6227_CR40","unstructured":"Zaheer, M., Reddi, S., Sachan, D., Kale, S., & Kumar, S. (2018). Adaptive methods for nonconvex optimization. Advances in Neural Information Processing Systems."},{"key":"6227_CR41","unstructured":"Zhou, D., Tang, Y., Yang, Z., Cao, Y., & Gu, Q. (2018a). On the convergence of adaptive gradient methods for nonconvex optimization. 1808.05671."},{"key":"6227_CR42","unstructured":"Zhou, D., Xu, P., & Gu, Q. (2018b). Stochastic nested variance reduction for nonconvex optimization. Advances in Neural Information Processing Systems."},{"key":"6227_CR43","unstructured":"Zhuang, J., Tang, T., Ding, Y., Tatikonda, S., Dvornek, N., Papademetris, X., & Duncan, J. S. (2020). AdaBelief optimizer: Adapting stepsizes by the belief in observed gradients. Advances in Neural Information Processing Systems."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-022-06227-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-022-06227-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-022-06227-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,22]],"date-time":"2023-08-22T00:03:15Z","timestamp":1692662595000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-022-06227-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,22]]},"references-count":43,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["6227"],"URL":"https:\/\/doi.org\/10.1007\/s10994-022-06227-3","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,8,22]]},"assertion":[{"value":"8 June 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 July 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 August 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflicts of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"This paper is approved in ethics.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"This paper is consented to participate.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"This paper is consented for publication.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}