{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,8]],"date-time":"2026-02-08T06:11:32Z","timestamp":1770531092642,"version":"3.49.0"},"reference-count":71,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2023,4,3]],"date-time":"2023-04-03T00:00:00Z","timestamp":1680480000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,4,3]],"date-time":"2023-04-03T00:00:00Z","timestamp":1680480000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100009515","name":"Fondation Simone et Cino Del Duca","doi-asserted-by":"publisher","award":["OpSiMorE"],"award-info":[{"award-number":["OpSiMorE"]}],"id":[{"id":"10.13039\/100009515","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001665","name":"Agence Nationale de la Recherche","doi-asserted-by":"publisher","award":["ANR-19-CE23 MASDOL"],"award-info":[{"award-number":["ANR-19-CE23 MASDOL"]}],"id":[{"id":"10.13039\/501100001665","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Stat Comput"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s11222-023-10230-6","type":"journal-article","created":{"date-parts":[[2023,4,3]],"date-time":"2023-04-03T10:02:30Z","timestamp":1680516150000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Stochastic variable metric proximal gradient with variance reduction for non-convex composite optimization"],"prefix":"10.1007","volume":"33","author":[{"given":"Gersende","family":"Fort","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eric","family":"Moulines","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,4,3]]},"reference":[{"key":"10230_CR1","volume-title":"Advances in Neural Information Processing Systems","author":"Z Allen-Zhu","year":"2018","unstructured":"Allen-Zhu, Z.: Natasha 2: Faster Non-Convex Optimization Than SGD. In: Bengio, S., Wallach, H., Larochelle, H., et al. (eds.) Advances in Neural Information Processing Systems, vol. 31. Curran Associates Inc, New York (2018)"},{"key":"10230_CR2","unstructured":"Allen-Zhu, Z., Hazan, E.: Variance reduction for faster non-convex optimization. In: Balcan M, Weinberger K (eds) 33rd International Conference on Machine Learning, ICML 2016, pp 1093\u20131101 (2016)"},{"issue":"2","key":"10230_CR3","doi-asserted-by":"publisher","first-page":"391","DOI":"10.1239\/jap\/1437658605","volume":"52","author":"C Andrieu","year":"2015","unstructured":"Andrieu, C., Fort, G., Vihola, M.: Quantitative convergence rates for subgeometric Markov chains. J. Appl. Probab. 52(2), 391\u2013404 (2015). https:\/\/doi.org\/10.1239\/jap\/1437658605","journal-title":"J. Appl. Probab."},{"issue":"10","key":"10230_CR4","first-page":"1","volume":"18","author":"Y Atchad\u00e9","year":"2017","unstructured":"Atchad\u00e9, Y., Fort, G., Moulines, E.: On perturbed proximal gradient algorithms. J. Mach. Learn. Res. 18(10), 1\u201333 (2017)","journal-title":"J. Mach. Learn. Res."},{"key":"10230_CR5","doi-asserted-by":"publisher","unstructured":"Bauschke, H.H., Combettes, P.L.: Convex Analysis and Monotone Operator Theory in Hilbert Spaces, 1st edn. Springer Publishing Company, Incorporated, (2011) https:\/\/doi.org\/10.1007\/978-1-4419-9467-7","DOI":"10.1007\/978-1-4419-9467-7"},{"issue":"1137\/1","key":"10230_CR6","first-page":"9781611974997","volume":"10","author":"A Beck","year":"2017","unstructured":"Beck, A.: First-order methods in optimization. Soc. Ind. Appl. Math. 10(1137\/1), 9781611974997 (2017)","journal-title":"Soc. Ind. Appl. Math."},{"key":"10230_CR7","volume-title":"Advances in neural information processing systems","author":"S Becker","year":"2012","unstructured":"Becker, S., Fadili, J.: A Quasi-Newton Proximal Splitting Method. In: Pereira, F., Burges, C., Bottou, L., et al. (eds.) Advances in neural information processing systems, vol. 25. Curran Associates Inc (2012)"},{"issue":"4","key":"10230_CR8","doi-asserted-by":"publisher","first-page":"2445","DOI":"10.1137\/18M1167152","volume":"29","author":"S Becker","year":"2019","unstructured":"Becker, S., Fadili, J., Ochs, P.: On quasi-newton forward\u2013backward splitting: Proximal calculus and convergence. SIAM J. Optim. 29(4), 2445\u20132481 (2019). https:\/\/doi.org\/10.1137\/18M1167152","journal-title":"SIAM J. Optim."},{"key":"10230_CR9","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-75894-2","volume-title":"Adaptive Algorithms and Stochastic Approximations","author":"A Benveniste","year":"1990","unstructured":"Benveniste, A., M\u00e9tivier, M., Priouret, P.: Adaptive Algorithms and Stochastic Approximations. Springer Verlag, London (1990). https:\/\/doi.org\/10.1007\/978-3-642-75894-2"},{"issue":"113","key":"10230_CR10","doi-asserted-by":"publisher","first-page":"192","DOI":"10.1016\/j.cam.2020.113192","volume":"385","author":"S Bonettini","year":"2021","unstructured":"Bonettini, S., Porta, F., Ruggiero, V., et al.: Variable metric techniques for forward\u2013backward methods in imaging. J. Comput. Appl. Math. 385(113), 192 (2021). https:\/\/doi.org\/10.1016\/j.cam.2020.113192","journal-title":"J. Comput. Appl. Math."},{"key":"10230_CR11","doi-asserted-by":"publisher","DOI":"10.1007\/978-93-86279-38-5","volume-title":"Stochastic Approximation","author":"VS Borkar","year":"2008","unstructured":"Borkar, V.S.: Stochastic Approximation. Cambridge University Press, Hindustan Book Agency, Cambridge, New Delhi (2008). https:\/\/doi.org\/10.1007\/978-93-86279-38-5 . (a dynamical systems viewpoint)"},{"key":"10230_CR12","doi-asserted-by":"publisher","DOI":"10.1214\/lnms\/1215466757","author":"L Brown","year":"1986","unstructured":"Brown, L.: Fundamentals of statistical exponential families\u202f: with applications in statistical decision theory. Lecture notes-monograph series fundamentals of statistical exponential families. Inst. Math. Stat. (1986). https:\/\/doi.org\/10.1214\/lnms\/1215466757","journal-title":"Inst. Math. Stat."},{"issue":"3","key":"10230_CR13","doi-asserted-by":"publisher","first-page":"593","DOI":"10.1111\/j.1467-9868.2009.00698.x","volume":"71","author":"O Capp\u00e9","year":"2009","unstructured":"Capp\u00e9, O., Moulines, E.: On-line expectation maximization algorithm for latent data models. J. Roy. Stat. Soc. B Met. 71(3), 593\u2013613 (2009). https:\/\/doi.org\/10.1111\/j.1467-9868.2009.00698.x","journal-title":"J. Roy. Stat. Soc. B Met."},{"key":"10230_CR14","first-page":"73","volume":"2","author":"G Celeux","year":"1985","unstructured":"Celeux, G., Diebolt, J.: The SEM algorithm: a probabilistic teacher algorithm derived from the EM algorithm for the mixture problem. Comput. Stat. Q. 2, 73\u201382 (1985)","journal-title":"Comput. Stat. Q."},{"key":"10230_CR15","doi-asserted-by":"publisher","first-page":"421","DOI":"10.1137\/S1052623495290179","volume":"7","author":"HG Chen","year":"1997","unstructured":"Chen, H.G., Rockafellar, R.: Convergence rates in forward-backward splitting. SIAM J. Optim. 7, 421\u2013444 (1997). https:\/\/doi.org\/10.1137\/S1052623495290179","journal-title":"SIAM J. Optim."},{"key":"10230_CR16","doi-asserted-by":"publisher","unstructured":"Chen, J., Zhu, J., Teh, Y., et\u00a0al: Stochastic Expectation Maximization with Variance Reduction. In: Bengio S, Wallach H, Larochelle H, et\u00a0al (eds) Advances in Neural Information Processing Systems 31. Curran Associates, Inc., pp. 7967\u20137977, (2018) https:\/\/doi.org\/10.5555\/3327757.3327893","DOI":"10.5555\/3327757.3327893"},{"key":"10230_CR17","unstructured":"Chen, X., Liu, S., Sun, R., et\u00a0al: On the convergence of a class of adam-type algorithms for non-convex optimization. In: International Conference on Learning Representations (2019)"},{"issue":"none","key":"10230_CR18","doi-asserted-by":"publisher","first-page":"2054","DOI":"10.1214\/13-EJS837","volume":"7","author":"H Choi","year":"2013","unstructured":"Choi, H., Hobert, J.: The Polya-Gamma Gibbs sampler for Bayesian logistic regression is uniformly ergodic. Electron. J. Stat. 7(none), 2054\u20132064 (2013). https:\/\/doi.org\/10.1214\/13-EJS837","journal-title":"Electron. J. Stat."},{"issue":"1","key":"10230_CR19","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1007\/s10957-013-0465-7","volume":"162","author":"E Chouzenoux","year":"2014","unstructured":"Chouzenoux, E., Pesquet, J.C., Repetti, A.: Variable metric forward\u2013backward algorithm for minimizing the sum of a differentiable function and a convex function. J. Optim. Theory Appl. 162(1), 107\u2013132 (2014). https:\/\/doi.org\/10.1007\/s10957-013-0465-7","journal-title":"J. Optim. Theory Appl."},{"key":"10230_CR20","doi-asserted-by":"publisher","unstructured":"Combettes, P., Pesquet, J.: Proximal splitting methods in signal processing. In: Bauschke HH, Burachik RS, Combettes PL, et\u00a0al (eds) Fixed-Point Algorithms for Inverse Problems in Science and Engineering. Springer Optimization and Its Applications, pp. 185\u2013212, Springer (2011). https:\/\/doi.org\/10.1007\/978-1-4419-9569-8","DOI":"10.1007\/978-1-4419-9569-8"},{"issue":"9","key":"10230_CR21","doi-asserted-by":"publisher","first-page":"1289","DOI":"10.1080\/02331934.2012.733883","volume":"63","author":"P Combettes","year":"2014","unstructured":"Combettes, P., V\u0169, B.: Variable metric forward\u2013backward splitting with applications to monotone inclusions in duality. Optimization 63(9), 1289\u20131318 (2014). https:\/\/doi.org\/10.1080\/02331934.2012.733883","journal-title":"Optimization"},{"issue":"4","key":"10230_CR22","doi-asserted-by":"publisher","first-page":"1168","DOI":"10.1137\/050626090","volume":"4","author":"PL Combettes","year":"2005","unstructured":"Combettes, P.L., Wajs, V.R.: Signal recovery by proximal forward\u2013backward splitting. Multiscale Model. Simul. 4(4), 1168\u20131200 (2005). https:\/\/doi.org\/10.1137\/050626090","journal-title":"Multiscale Model. Simul."},{"key":"10230_CR23","unstructured":"Defazio, A., Bach, F., Lacoste-Julien, S.: Saga: A fast incremental gradient method with support for non-strongly convex composite objectives. In: Proceedings of the 27th International Conference on Neural Information Processing Systems, vol. 1, pp. 1646-1654. MIT Press, Cambridge, MA, USA, NIPS\u201914, (2014)"},{"issue":"1","key":"10230_CR24","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1214\/aos\/1018031103","volume":"27","author":"B Delyon","year":"1999","unstructured":"Delyon, B., Lavielle, M., Moulines, E.: Convergence of a stochastic approximation version of the EM algorithm. Ann. Stat. 27(1), 94\u2013128 (1999). https:\/\/doi.org\/10.1214\/aos\/1018031103","journal-title":"Ann. Stat."},{"issue":"1","key":"10230_CR25","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1111\/j.2517-6161.1977.tb01600.x","volume":"39","author":"A Dempster","year":"1977","unstructured":"Dempster, A., Laird, N., Rubin, D.: Maximum likelihood from incomplete data via the EM algorithm. J. Roy. Stat. Soc. B Met. 39(1), 1\u201338 (1977)","journal-title":"J. Roy. Stat. Soc. B Met."},{"key":"10230_CR26","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4613-8643-8","volume-title":"Non-Uniform Random Variate Generation","author":"L Devroye","year":"1986","unstructured":"Devroye, L.: Non-Uniform Random Variate Generation. Springer-Verlag, London (1986). https:\/\/doi.org\/10.1007\/978-1-4613-8643-8"},{"issue":"5\u20136","key":"10230_CR27","doi-asserted-by":"publisher","first-page":"413","DOI":"10.1080\/01630569208816489","volume":"13","author":"B Eicke","year":"1992","unstructured":"Eicke, B.: Iteration methods for convexly constrained ill-posed problems in Hilbert space. Numer. Funct. Anal. Optim. 13(5\u20136), 413\u2013429 (1992). https:\/\/doi.org\/10.1080\/01630569208816489","journal-title":"Numer. Funct. Anal. Optim."},{"key":"10230_CR28","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-009-5564-6","volume-title":"An Introduction to Latent Variable Models","author":"B Everitt","year":"1984","unstructured":"Everitt, B.: An Introduction to Latent Variable Models. Chapman and Hall London, New York (1984). https:\/\/doi.org\/10.1007\/978-94-009-5564-6"},{"key":"10230_CR29","volume-title":"Advances in neural information processing systems","author":"C Fang","year":"2018","unstructured":"Fang, C., Li, C., Lin, Z., et al.: Spider: Near-Optimal Non-convex Optimization Via Stochastic Path-integrated Differential Estimator. In: Bengio, S., Wallach, H., Larochelle, H., et al. (eds.) Advances in neural information processing systems, vol. 31. Curran Associates Inc, New York (2018)"},{"issue":"4","key":"10230_CR30","doi-asserted-by":"publisher","first-page":"1220","DOI":"10.1214\/aos\/1059655912","volume":"31","author":"G Fort","year":"2003","unstructured":"Fort, G., Moulines, E.: Convergence of the Monte Carlo expectation maximization for curved exponential families. Ann. Stat. 31(4), 1220\u20131259 (2003)","journal-title":"Ann. Stat."},{"key":"10230_CR31","doi-asserted-by":"publisher","unstructured":"Fort, G., Moulines, E.: The perturbed prox-preconditioned spider algorithm: non-asymptotic convergence bounds. In: 2021 IEEE Statistical Signal Processing Workshop (SSP), pp. 96\u2013100, (2021). https:\/\/doi.org\/10.1109\/SSP49050.2021.9513846","DOI":"10.1109\/SSP49050.2021.9513846"},{"issue":"6","key":"10230_CR32","doi-asserted-by":"publisher","first-page":"3262","DOI":"10.1214\/11-AOS938","volume":"39","author":"G Fort","year":"2011","unstructured":"Fort, G., Moulines, E., Priouret, P.: Convergence of adaptive and interacting Markov chain monte Carlo algorithms. Ann. Stat. 39(6), 3262\u20133289 (2011)","journal-title":"Ann. Stat."},{"key":"10230_CR33","doi-asserted-by":"publisher","unstructured":"Fort, G., Risser, L., Atchad\u00e9, Y., et\u00a0al: Stochastic Fista algorithms: So fast ? In: 2018 IEEE Statistical Signal Processing Workshop (SSP), pp. 796\u2013800, (2018). https:\/\/doi.org\/10.1109\/SSP.2018.8450740","DOI":"10.1109\/SSP.2018.8450740"},{"key":"10230_CR34","unstructured":"Fort, G., Moulines, E., Wai, H.T.: A stochastic path-integrated differential estimator expectation maximization algorithm. In: Proceedings of the 34th International Conference on Neural Information Processing Systems. Curran Associates Inc., Red Hook, NY, USA, NIPS\u201920 (2020)"},{"issue":"4","key":"10230_CR35","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1007\/s11222-021-10023-9","volume":"31","author":"G Fort","year":"2021","unstructured":"Fort, G., Gach, P., Moulines, E.: Fast incremental expectation maximization for finite-sum optimization: nonasymptotic convergence. Stat. Comput. 31(4), 48 (2021). https:\/\/doi.org\/10.1007\/s11222-021-10023-9","journal-title":"Stat. Comput."},{"key":"10230_CR36","doi-asserted-by":"publisher","unstructured":"Fort, G., Moulines, E., Wai, H.T.: Geom-Spider-EM: Faster variance reduced stochastic expectation maximization for nonconvex finite-sum optimization. In: ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 3135\u20133139, (2021b). https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9414271","DOI":"10.1109\/ICASSP39728.2021.9414271"},{"issue":"4","key":"10230_CR37","doi-asserted-by":"publisher","first-page":"2341","DOI":"10.1137\/120880811","volume":"23","author":"S Ghadimi","year":"2013","unstructured":"Ghadimi, S., Lan, G.: Stochastic first- and zeroth-order methods for nonconvex stochastic programming. SIAM J. Opt. 23(4), 2341\u20132368 (2013). https:\/\/doi.org\/10.1137\/120880811","journal-title":"SIAM J. Opt."},{"issue":"1\u20132","key":"10230_CR38","doi-asserted-by":"publisher","first-page":"267","DOI":"10.1007\/s10107-014-0846-1","volume":"155","author":"S Ghadimi","year":"2016","unstructured":"Ghadimi, S., Lan, G., Zhang, H.: Mini-batch stochastic approximation methods for nonconvex stochastic composite optimization. Math. Program. 155(1\u20132), 267\u2013305 (2016). https:\/\/doi.org\/10.1007\/s10107-014-0846-1","journal-title":"Math. Program."},{"key":"10230_CR39","unstructured":"Gower, R., Goldfarb, D., Richtarik, P.: Stochastic Block BFGS: Squeezing more curvature out of data. In: Balcan MF, Weinberger KQ (eds) Proceedings of The 33rd International Conference on Machine Learning, Proceedings of Machine Learning Research, vol.\u00a048, pp. 1869\u20131878. PMLR, New York, New York, USA (2016)"},{"key":"10230_CR40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-662-02796-7","volume-title":"Convex Analysis and Minimization Algorithms","author":"JB Hiriart-Urruty","year":"1996","unstructured":"Hiriart-Urruty, J.B., Lemar\u00e9chal, C.: Convex Analysis and Minimization Algorithms. Springer Verlag, Heidelberg (1996). https:\/\/doi.org\/10.1007\/978-3-662-02796-7 . (two volumes-2nd printing)"},{"issue":"2","key":"10230_CR41","doi-asserted-by":"publisher","first-page":"634","DOI":"10.1137\/21M1394308","volume":"4","author":"S Horv\u00e1th","year":"2022","unstructured":"Horv\u00e1th, S., Lei, L., Richt\u00e1rik, P., et al.: Adaptivity of stochastic gradient methods for nonconvex optimization. SIAM J. Math. Data Sci. 4(2), 634\u2013648 (2022). https:\/\/doi.org\/10.1137\/21M1394308","journal-title":"SIAM J. Math. Data Sci."},{"key":"10230_CR42","volume-title":"Advances in neural information processing systems","author":"R Johnson","year":"2013","unstructured":"Johnson, R., Zhang, T.: Accelerating Stochastic Gradient Descent Using Predictive Variance Reduction. In: Burges, C., Bottou, L., Welling, M., et al. (eds.) Advances in neural information processing systems, vol. 26. Curran Associates Inc, New York (2013)"},{"key":"10230_CR43","unstructured":"Karimi, B., Wai, H.T., Moulines, E., et\u00a0al: On the global convergence of (fast) incremental expectation maximization methods. In: Wallach H, Larochelle H, Beygelzimer A, et\u00a0al (eds) Advances in Neural Information Processing Systems 32, pp. 2837\u20132847. Curran Associates, Inc., (2019)"},{"key":"10230_CR44","doi-asserted-by":"crossref","unstructured":"Karimi, H., Nutini, J., Schmidt, M.: Linear convergence of gradient and proximal-gradient methods under the Polyak-\u0141ojasiewicz condition. In: Frasconi P, Landwehr N, Manco G, et\u00a0al (eds) Machine Learning and Knowledge Discovery in Databases, pp. 795\u2013811. Springer International Publishing (2016)","DOI":"10.1007\/978-3-319-46128-1_50"},{"key":"10230_CR45","unstructured":"Kolte, R., Erdogdu, M., Ozgur, A.: Accelerating svrg via second-order information. In: Advances in Neural Information Processing Systems - Workshop OptML, pp. 1\u20135 (2015)"},{"key":"10230_CR46","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-39568-1","volume-title":"First-order and Stochastic Optimization Methods for Machine Learning, Springer Series in the Data Sciences","author":"G Lan","year":"2020","unstructured":"Lan, G.: First-order and Stochastic Optimization Methods for Machine Learning, Springer Series in the Data Sciences. Springer International Publishing, London (2020). https:\/\/doi.org\/10.1007\/978-3-030-39568-1"},{"issue":"3","key":"10230_CR47","doi-asserted-by":"publisher","first-page":"1420","DOI":"10.1137\/130921428","volume":"24","author":"JD Lee","year":"2014","unstructured":"Lee, J.D., Sun, Y., Saunders, M.A.: Proximal Newton-Type Methods for Minimizing Composite Functions. SIAM J. Opt. 24(3), 1420\u20131443 (2014). https:\/\/doi.org\/10.1137\/130921428","journal-title":"SIAM J. Opt."},{"key":"10230_CR48","volume-title":"Advances in neural information processing systems","author":"Z Li","year":"2018","unstructured":"Li, Z., Li, J.: A Simple Proximal Stochastic Gradient Method for Nonsmooth Nonconvex Optimization. In: Bengio, S., Wallach, H., Larochelle, H., et al. (eds.) Advances in neural information processing systems, vol. 31. Curran Associates Inc, New York (2018)"},{"key":"10230_CR49","unstructured":"Li, Z., Bao, H., Zhang, X., et\u00a0al: PAGE: A simple and optimal probabilistic gradient estimator for nonconvex optimization. In: Meila M, Zhang T (eds) Proceedings of the 38th International Conference on Machine Learning, Proceedings of Machine Learning Research, vol. 139, pp. 6286\u20136295. PMLR (2021)"},{"key":"10230_CR50","doi-asserted-by":"publisher","DOI":"10.1002\/9780470191613","volume-title":"The EM Algorithm and Extensions","author":"G McLachlan","year":"2008","unstructured":"McLachlan, G., Krishnan, T.: The EM Algorithm and Extensions, 2nd edn. Wiley Series in Probability and Statistics, Wiley, New York (2008). https:\/\/doi.org\/10.1002\/9780470191613","edition":"2"},{"issue":"115","key":"10230_CR51","first-page":"1","volume":"22","author":"M Metel","year":"2021","unstructured":"Metel, M., Takeda, A.: Stochastic proximal methods for non-smooth non-convex constrained sparse optimization. J. Mach. Learn. Res. 22(115), 1\u201336 (2021)","journal-title":"J. Mach. Learn. Res."},{"key":"10230_CR52","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-3267-7","volume-title":"Markov Chains and Stochastic Stability","author":"S Meyn","year":"1993","unstructured":"Meyn, S., Tweedie, R.: Markov Chains and Stochastic Stability. Springer-Verlag, London (1993). https:\/\/doi.org\/10.1007\/978-1-4471-3267-7"},{"key":"10230_CR53","doi-asserted-by":"publisher","first-page":"273","DOI":"10.24033\/bsmf.1625","volume":"93","author":"J Moreau","year":"1965","unstructured":"Moreau, J.: Proximit\u00e9 et dualit\u00e9 dans un espace hilbertien. Bull. Soc. Math. France 93, 273\u2013299 (1965). https:\/\/doi.org\/10.24033\/bsmf.1625","journal-title":"Bull. Soc. Math. France"},{"key":"10230_CR54","unstructured":"Moritz, P., Nishihara, R., Jordan, M.: A linearly-convergent stochastic L-BFGS algorithm. In: Gretton A, Robert CC (eds) Proceedings of the 19th International Conference on Artificial Intelligence and Statistics, Proceedings of Machine Learning Research, vol.\u00a051, pp. 249\u2013258. PMLR(2016)"},{"key":"10230_CR55","doi-asserted-by":"publisher","unstructured":"Neal, R.M., Hinton, G.E.: A View of the EM Algorithm that Justifies Incremental, Sparse, and other Variants. In: Jordan MI (ed) Learning in Graphical Models, pp. 355\u2013368. Springer Netherlands, Dordrecht (1998) https:\/\/doi.org\/10.1007\/978-94-011-5014-9_12","DOI":"10.1007\/978-94-011-5014-9_12"},{"issue":"1","key":"10230_CR56","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1023\/A:1021987710829","volume":"13","author":"SK Ng","year":"2003","unstructured":"Ng, S.K., McLachlan, G.J.: On the choice of the number of blocks with the incremental EM algorithm for the fitting of normal mixtures. Stat. Comput. 13(1), 45\u201355 (2003). https:\/\/doi.org\/10.1023\/A:1021987710829","journal-title":"Stat. Comput."},{"key":"10230_CR57","unstructured":"Nguyen, L., Liu, J., Scheinberg, K., et\u00a0al: SARAH: A novel method for machine learning problems using stochastic recursive gradient. In: Precup D, Teh YW (eds) Proceedings of the 34th International Conference on Machine Learning, pp. 2613\u20132621 (2017)"},{"issue":"110","key":"10230_CR58","first-page":"1","volume":"21","author":"HP Nhan","year":"2020","unstructured":"Nhan, H.P., Lam, M.N., Dzung, T.P., et al.: ProxSARAH: an efficient algorithmic framework for stochastic composite nonconvex optimization. J. Mach. Learn. Res. 21(110), 1\u201348 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"10230_CR59","doi-asserted-by":"publisher","unstructured":"Park, Y., Dhar, S., Boyd, S., et\u00a0al: Variable metric proximal gradient method with diagonal Barzilai-Borwein stepsize. (2019) https:\/\/doi.org\/10.48550\/ARXIV.1910.07056","DOI":"10.48550\/ARXIV.1910.07056"},{"issue":"504","key":"10230_CR60","doi-asserted-by":"publisher","first-page":"1339","DOI":"10.1080\/01621459.2013.829001","volume":"108","author":"NG Polson","year":"2013","unstructured":"Polson, N.G., Scott, J., Windle, J.: Bayesian inference for logistic models using P\u2019olya-Gamma latent variables. J. Am. Stat. Assoc. 108(504), 1339\u20131349 (2013). https:\/\/doi.org\/10.1080\/01621459.2013.829001","journal-title":"J. Am. Stat. Assoc."},{"key":"10230_CR61","unstructured":"Reddi, S.J., Hefny, A., Sra, S., et\u00a0al: Stochastic variance reduction for nonconvex optimization. In: Balcan MF, Weinberger KQ (eds) Proceedings of The 33rd International Conference on Machine Learning, Proceedings of Machine Learning Research, vol.\u00a048, pp. 314\u2013323. PMLR, New York, New York, USA (2016)"},{"key":"10230_CR62","doi-asserted-by":"publisher","first-page":"1215","DOI":"10.1137\/19M1277552","volume":"31","author":"A Repetti","year":"2021","unstructured":"Repetti, A., Wiaux, Y.: Variable metric forward-backward algorithm for composite minimization problems. SIAM J. Opt. 31, 1215\u20131241 (2021). https:\/\/doi.org\/10.1137\/19M1277552","journal-title":"SIAM J. Opt."},{"key":"10230_CR63","doi-asserted-by":"publisher","unstructured":"Repetti, A., Chouzenoux, E., Pesquet, J.C.: A preconditioned forward-backward approach with application to large-scale nonconvex spectral unmixing problems. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1498\u20131502, (2014) https:\/\/doi.org\/10.1109\/ICASSP.2014.6853847","DOI":"10.1109\/ICASSP.2014.6853847"},{"key":"10230_CR64","doi-asserted-by":"publisher","unstructured":"Robert, C., Casella, G.: Monte Carlo Statistical Methods. Springer Verlag, London (2004). https:\/\/doi.org\/10.1007\/978-1-4757-4145-2)","DOI":"10.1007\/978-1-4757-4145-2"},{"key":"10230_CR65","unstructured":"Wang, Z., Ji, K., Zhou, Y., et\u00a0al: Spiderboost and momentum: faster variance reduction algorithms. In: Wallach HM, Larochelle H, Beygelzimer A, et\u00a0al (eds) Advances in Neural Information Processing Systems 32: Annual Conference on Neural Information Processing Systems 2019, pp 2403\u20132413. NeurIPS 2019, December 8-14, 2019, Vancouver, BC, Canada (2019)"},{"issue":"411","key":"10230_CR66","doi-asserted-by":"publisher","first-page":"699","DOI":"10.1080\/01621459.1990.10474930","volume":"85","author":"G Wei","year":"1990","unstructured":"Wei, G., Tanner, M.: A Monte Carlo implementation of the EM algorithm and the poor man\u2019s data augmentation algorithms. J. Am. Stat. Assoc. 85(411), 699\u2013704 (1990). https:\/\/doi.org\/10.1080\/01621459.1990.10474930","journal-title":"J. Am. Stat. Assoc."},{"issue":"1","key":"10230_CR67","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1214\/aos\/1176346060","volume":"11","author":"C Wu","year":"1983","unstructured":"Wu, C.: On the convergence properties of the EM algorithm. Ann. Stat. 11(1), 95\u2013103 (1983). https:\/\/doi.org\/10.1214\/aos\/1176346060","journal-title":"Ann. Stat."},{"key":"10230_CR68","unstructured":"Yun, J., Lozano, A.C., Yang, E.: Adaptive proximal gradient methods for structured neural networks. In: Ranzato M, Beygelzimer A, Dauphin Y, et\u00a0al (eds) Advances in Neural Information Processing Systems, vol.\u00a034, pp. 24365\u201324378. Curran Associates, Inc. (2021)"},{"key":"10230_CR69","volume-title":"Advances in neural information processing systems","author":"J Zhang","year":"2019","unstructured":"Zhang, J., Xiao, L.: A Stochastic Composite Gradient Method with Incremental Variance Reduction. In: Wallach, H., Larochelle, H., Beygelzimer, A., et al. (eds.) Advances in neural information processing systems, vol. 32. Curran Associates Inc (2019)"},{"issue":"9","key":"10230_CR70","doi-asserted-by":"publisher","first-page":"4388","DOI":"10.1109\/TNNLS.2021.3056947","volume":"33","author":"Q Zhang","year":"2022","unstructured":"Zhang, Q., Huang, F., Deng, C., et al.: Faster stochastic quasi-newton methods. IEEE Trans. Neural Netw. Learn. Syst. 33(9), 4388\u20134397 (2022). https:\/\/doi.org\/10.1109\/TNNLS.2021.3056947","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"103","key":"10230_CR71","first-page":"1","volume":"21","author":"D Zhou","year":"2020","unstructured":"Zhou, D., Xu, P., Gu, Q.: Stochastic nested variance reduction for nonconvex optimization. J. Mach. Learn. Res. 21(103), 1\u201363 (2020)","journal-title":"J. Mach. Learn. Res."}],"container-title":["Statistics and Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11222-023-10230-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11222-023-10230-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11222-023-10230-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,17]],"date-time":"2024-10-17T15:56:34Z","timestamp":1729180594000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11222-023-10230-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,3]]},"references-count":71,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["10230"],"URL":"https:\/\/doi.org\/10.1007\/s11222-023-10230-6","relation":{},"ISSN":["0960-3174","1573-1375"],"issn-type":[{"value":"0960-3174","type":"print"},{"value":"1573-1375","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,4,3]]},"assertion":[{"value":"25 September 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 March 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 April 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"65"}}