{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T00:43:07Z","timestamp":1778200987298,"version":"3.51.4"},"reference-count":93,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2015,6,11]],"date-time":"2015-06-11T00:00:00Z","timestamp":1433980800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Stat Comput"],"published-print":{"date-parts":[[2015,7]]},"DOI":"10.1007\/s11222-015-9560-y","type":"journal-article","created":{"date-parts":[[2015,6,11]],"date-time":"2015-06-11T08:12:50Z","timestamp":1434010370000},"page":"781-795","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":30,"title":["Scalable estimation strategies based on stochastic approximations: classical results and new insights"],"prefix":"10.1007","volume":"25","author":[{"given":"Panos","family":"Toulis","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edoardo M.","family":"Airoldi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,6,11]]},"reference":[{"issue":"2","key":"9560_CR1","doi-asserted-by":"crossref","first-page":"251","DOI":"10.1162\/089976698300017746","volume":"10","author":"S-I Amari","year":"1998","unstructured":"Amari, S.-I.: Natural gradient works efficiently in learning. Neural Comput. 10(2), 251\u2013276 (1998)","journal-title":"Neural Comput."},{"issue":"6","key":"9560_CR2","doi-asserted-by":"crossref","first-page":"1399","DOI":"10.1162\/089976600300015420","volume":"12","author":"S-I Amari","year":"2000","unstructured":"Amari, S.-I., Park, H., Kenji, F.: Adaptive method of realizing natural gradient learning for multilayer perceptrons. Neural Comput. 12(6), 1399\u20131409 (2000)","journal-title":"Neural Comput."},{"key":"9560_CR3","volume-title":"Stochastic Approximation: A Generalisation of the Robbins\u2013Monro Procedure","author":"JA Bather","year":"1989","unstructured":"Bather, J.A.: Stochastic Approximation: A Generalisation of the Robbins\u2013Monro Procedure, vol. 89. Cornell University, Mathematical Sciences Institute, New York (1989)"},{"issue":"3","key":"9560_CR4","doi-asserted-by":"crossref","first-page":"167","DOI":"10.1016\/S0167-6377(02)00231-6","volume":"31","author":"A Beck","year":"2003","unstructured":"Beck, A., Teboulle, M.: Mirror descent and nonlinear projected subgradient methods for convex optimization. Oper. Res. Lett. 31(3), 167\u2013175 (2003)","journal-title":"Oper. Res. Lett."},{"key":"9560_CR5","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1561\/2200000006","volume":"2","author":"Y Bengio","year":"2009","unstructured":"Bengio, Y.: Learning deep architectures for ai. Foundations and trends $$\\textregistered $$ \u00ae . Mach. Learn. 2, 1\u2013127 (2009)","journal-title":"Mach. Learn."},{"issue":"6","key":"9560_CR6","doi-asserted-by":"crossref","first-page":"1601","DOI":"10.1162\/neco.2008.11-07-647","volume":"21","author":"Y Bengio","year":"2009","unstructured":"Bengio, Y., Delalleau, O.: Justifying and generalizing contrastive divergence. Neural Comput. 21(6), 1601\u20131621 (2009)","journal-title":"Neural Comput."},{"key":"9560_CR7","volume-title":"Adaptive Algorithms and Stochastic Approximations","author":"A Benveniste","year":"2012","unstructured":"Benveniste, A., M\u00e9tivier, M., Priouret, P.: Adaptive Algorithms and Stochastic Approximations. Springer Publishing Company, Incorporated, New York (2012)"},{"key":"9560_CR8","doi-asserted-by":"crossref","unstructured":"Bertsekas, D.P., Tsitsiklis, J.N.: Neuro-dynamic programming: an overview. In: Proceedings of the 34th IEEE Conference on Decision and Control, vol. 1, pp. 560\u2013564 (1995)","DOI":"10.1109\/CDC.1995.478953"},{"key":"9560_CR9","first-page":"1737","volume":"10","author":"A Bordes","year":"2009","unstructured":"Bordes, A., Bottou, L., Gallinari, P.: Sgd-qn: careful quasi-Newton stochastic gradient descent. J. Mach. Learn. Res. 10, 1737\u20131754 (2009)","journal-title":"J. Mach. Learn. Res."},{"key":"9560_CR10","doi-asserted-by":"crossref","unstructured":"Bottou, L.: Large-scale machine learning with stochastic gradient descent. In: Proceedings of COMPSTAT\u20192010, pp. 177\u2013186. Springer, New York (2010)","DOI":"10.1007\/978-3-7908-2604-3_16"},{"issue":"2","key":"9560_CR11","doi-asserted-by":"crossref","first-page":"137","DOI":"10.1002\/asmb.538","volume":"21","author":"L Bottou","year":"2005","unstructured":"Bottou, L., Le Cun, Y.: On-line learning for very large data sets. Appl. Stoch. Models Bus. Ind. 21(2), 137\u2013151 (2005)","journal-title":"Appl. Stoch. Models Bus. Ind."},{"key":"9560_CR12","first-page":"161","volume":"20","author":"O Bousquet","year":"2008","unstructured":"Bousquet, O., Bottou, L.: The tradeoffs of large scale learning. Adv. Neural Inf. Process. Syst. 20, 161\u2013168 (2008)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"9560_CR13","doi-asserted-by":"crossref","first-page":"577","DOI":"10.1090\/S0025-5718-1965-0198670-6","volume":"19","author":"CG Broyden","year":"1965","unstructured":"Broyden, C.G.: A class of methods for solving nonlinear simultaneous equations. Math. Comput. 19, 577\u2013593 (1965)","journal-title":"Math. Comput."},{"issue":"3","key":"9560_CR14","doi-asserted-by":"crossref","first-page":"728","DOI":"10.1198\/jcgs.2011.09109","volume":"20","author":"O Capp\u00e9","year":"2011","unstructured":"Capp\u00e9, O.: Online em algorithm for hidden Markov models. J. Comput. Graph. Stat. 20(3), 728\u2013749 (2011)","journal-title":"J. Comput. Graph. Stat."},{"issue":"3","key":"9560_CR15","doi-asserted-by":"crossref","first-page":"593","DOI":"10.1111\/j.1467-9868.2009.00698.x","volume":"71","author":"O Capp\u00e9","year":"2009","unstructured":"Capp\u00e9, O., Moulines, M.: On-line expectation-maximization algorithm for latent data models. J. R. Stat. Soc. 71(3), 593\u2013613 (2009)","journal-title":"J. R. Stat. Soc."},{"key":"9560_CR16","unstructured":"Carreira-Perpinan, M.A., Hinton, G.E.: On contrastive divergence learning. In: Proceedings of the Tenth International Workshop on Artificial Intelligence and Statistics, pp. 33\u201340. Citeseer (2005)"},{"key":"9560_CR17","unstructured":"Cheng, L., Vishwanathan, S.V.N., Schuurmans, D., Wang, S., Caelli, T.: Implicit online learning with kernels. In: Proceedings of the 2006 Conference Advances in Neural Information Processing Systems 19, vol. 19, p. 249. MIT Press, Cambridge, 2007"},{"key":"9560_CR18","doi-asserted-by":"crossref","first-page":"463","DOI":"10.1214\/aoms\/1177728716","volume":"25","author":"KL Chung","year":"1954","unstructured":"Chung, K.L.: On a stochastic approximation method. Ann. Math. Stat. 25, 463\u2013483 (1954)","journal-title":"Ann. Math. Stat."},{"key":"9560_CR19","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1111\/j.2517-6161.1977.tb01600.x","volume":"39","author":"A Dempster","year":"1977","unstructured":"Dempster, A., Laird, N., Rubin, D.: Maximum likelihood from incomplete data via the EM algorithm. J. R. Stat. Soc. Ser. B 39, 1\u201338 (1977)","journal-title":"J. R. Stat. Soc. Ser. B"},{"key":"9560_CR20","first-page":"2121","volume":"999999","author":"J Duchi","year":"2011","unstructured":"Duchi, J., Hazan, E., Singer, Y.: Adaptive subgradient methods for online learning and stochastic optimization. J. Mach. Learn. Res. 999999, 2121\u20132159 (2011)","journal-title":"J. Mach. Learn. Res."},{"issue":"8","key":"9560_CR21","doi-asserted-by":"crossref","first-page":"915","DOI":"10.1109\/9.133185","volume":"36","author":"P Dupuis","year":"1991","unstructured":"Dupuis, P., Simha, R.: On sampling controlled stochastic approximation. IEEE Trans. Autom. Control 36(8), 915\u2013924 (1991)","journal-title":"IEEE Trans. Autom. Control"},{"key":"9560_CR22","doi-asserted-by":"crossref","first-page":"2757","DOI":"10.1214\/07-AOS581","volume":"36","author":"N El Karoui","year":"2008","unstructured":"El Karoui, N.: Spectrum estimation for large dimensional covariance matrices using random matrix theory. Ann. Stat. 36, 2757\u20132790 (2008)","journal-title":"Ann. Stat."},{"key":"9560_CR23","doi-asserted-by":"crossref","first-page":"1327","DOI":"10.1214\/aoms\/1177698258","volume":"39","author":"V Fabian","year":"1968","unstructured":"Fabian, V.: On asymptotic normality in stochastic approximation. Ann. Math. Stat. 39, 1327\u20131332 (1968)","journal-title":"Ann. Math. Stat."},{"key":"9560_CR24","doi-asserted-by":"crossref","first-page":"486","DOI":"10.1214\/aos\/1176342414","volume":"1","author":"V Fabian","year":"1973","unstructured":"Fabian, V.: Asymptotically efficient stochastic approximation; the RM case. Ann. Stat. 1, 486\u2013495 (1973)","journal-title":"Ann. Stat."},{"key":"9560_CR25","doi-asserted-by":"crossref","first-page":"309","DOI":"10.1098\/rsta.1922.0009","volume":"222","author":"RA Fisher","year":"1922","unstructured":"Fisher, R.A.: On the mathematical foundations of theoretical statistics. Philos. Trans. R. Soc. Lond. Ser. A 222, 309\u2013368 (1922)","journal-title":"Philos. Trans. R. Soc. Lond. Ser. A"},{"key":"9560_CR26","volume-title":"Statistical Methods for Research Workers","author":"RA Fisher","year":"1925","unstructured":"Fisher, R.A.: Statistical Methods for Research Workers. Oliver and Boyd, Edinburgh (1925a)"},{"key":"9560_CR27","doi-asserted-by":"crossref","unstructured":"Fisher, R.A.: Theory of statistical estimation. In: Mathematical Proceedings of the Cambridge Philosophical Society, vol. 22, pp. 700\u2013725. Cambridge University Press, Cambridge (1925b)","DOI":"10.1017\/S0305004100009580"},{"key":"9560_CR28","doi-asserted-by":"crossref","first-page":"721","DOI":"10.1109\/TPAMI.1984.4767596","volume":"6","author":"S Geman","year":"1984","unstructured":"Geman, S., Geman, D.: Stochastic relaxation, gibbs distributions, and the Bayesian restoration of images. IEEE Trans. Pattern Anal. Mach. Intell. 6, 721\u2013741 (1984)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"9560_CR29","doi-asserted-by":"crossref","first-page":"167","DOI":"10.1007\/s10994-006-8365-9","volume":"65","author":"AP George","year":"2006","unstructured":"George, A.P., Powell, W.B.: Adaptive stepsizes for recursive estimation with applications in approximate dynamic programming. Machine Learn. 65(1), 167\u2013198 (2006)","journal-title":"Machine Learn."},{"issue":"2","key":"9560_CR30","doi-asserted-by":"crossref","first-page":"123","DOI":"10.1111\/j.1467-9868.2010.00765.x","volume":"73","author":"M Girolami","year":"2011","unstructured":"Girolami, M.: Riemann manifold Langevin and Hamiltonian Monte Carlo methods. J. R. Stat. Soc. Ser. B 73(2), 123\u2013214 (2011)","journal-title":"J. R. Stat. Soc. Ser. B"},{"issue":"2","key":"9560_CR31","doi-asserted-by":"crossref","first-page":"178","DOI":"10.1287\/ijoc.1080.0305","volume":"21","author":"A Gosavi","year":"2009","unstructured":"Gosavi, A.: Reinforcement learning: a tutorial survey and recent advances. INFORMS J. Comput. 21(2), 178\u2013192 (2009)","journal-title":"INFORMS J. Comput."},{"key":"9560_CR32","doi-asserted-by":"crossref","first-page":"149","DOI":"10.1111\/j.2517-6161.1984.tb01288.x","volume":"46","author":"PJ Green","year":"1984","unstructured":"Green, P.J.: Iteratively reweighted least squares for maximum likelihood estimation, and some robust and resistant alternatives. J. R. Stat. Soc. Ser. B 46, 149\u2013192 (1984)","journal-title":"J. R. Stat. Soc. Ser. B"},{"key":"9560_CR33","volume-title":"The Elements of Statistical Learning: Data Mining, Inference, and Prediction","author":"T Hastie","year":"2011","unstructured":"Hastie, T., Tibshirani, R., Friedman, J.: The Elements of Statistical Learning: Data Mining, Inference, and Prediction, 2nd edn. Springer, New York (2011)","edition":"2"},{"issue":"1","key":"9560_CR34","first-page":"843","volume":"14","author":"P Hennig","year":"2013","unstructured":"Hennig, P., Kiefel, M.: Quasi-Newton methods: a new direction. J. Mach. Learn. Res. 14(1), 843\u2013865 (2013)","journal-title":"J. Mach. Learn. Res."},{"issue":"8","key":"9560_CR35","doi-asserted-by":"crossref","first-page":"1771","DOI":"10.1162\/089976602760128018","volume":"14","author":"GE Hinton","year":"2002","unstructured":"Hinton, G.E.: Training products of experts by minimizing contrastive divergence. Neural Comput. 14(8), 1771\u20131800 (2002)","journal-title":"Neural Comput."},{"issue":"1","key":"9560_CR36","first-page":"1303","volume":"14","author":"MD Hoffman","year":"2013","unstructured":"Hoffman, M.D., Blei, D.M., Wang, C., Paisley, J.: Stochastic variational inference. J. Mach. Learn. Res. 14(1), 1303\u20131347 (2013)","journal-title":"J. Mach. Learn. Res."},{"issue":"1","key":"9560_CR37","doi-asserted-by":"crossref","first-page":"73","DOI":"10.1214\/aoms\/1177703732","volume":"35","author":"PJ Huber","year":"1964","unstructured":"Huber, P.J., et al.: Robust estimation of a location parameter. Ann. Math. Stat. 35(1), 73\u2013101 (1964)","journal-title":"Ann. Math. Stat."},{"key":"9560_CR38","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-642-04898-2_594","volume-title":"Robust Statistics","author":"PJ Huber","year":"2011","unstructured":"Huber, P.J.: Robust Statistics. Springer, New York (2011)"},{"key":"9560_CR39","first-page":"315","volume":"26","author":"R Johnson","year":"2013","unstructured":"Johnson, R., Zhang, T.: Accelerating stochastic gradient descent using predictive variance reduction. Adv. Neural Inf. Process. Syst. 26, 315\u2013323 (2013)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"9560_CR40","doi-asserted-by":"crossref","unstructured":"Kivinen, J., Warmuth, M.K.: Additive versus exponentiated gradient updates for linear prediction. In: Proceedings of the Twenty-Seventh Annual ACM Symposium on Theory of Computing, pp. 209\u2013218","DOI":"10.1145\/225058.225121"},{"issue":"5","key":"9560_CR41","doi-asserted-by":"crossref","first-page":"1782","DOI":"10.1109\/TSP.2006.872551","volume":"54","author":"J Kivinen","year":"2006","unstructured":"Kivinen, J., Warmuth, M.K., Hassibi, B.: The p-norm generalization of the lms algorithm for adaptive filtering. IEEE Trans. Signal Process. 54(5), 1782\u20131793 (2006)","journal-title":"IEEE Trans. Signal Process."},{"key":"9560_CR42","unstructured":"Korattikara, A., Chen, Y., Welling, M.: Austerity in mcmc land: cutting the metropolis-hastings budget. In: Proceedings of the 31st International Conference on Machine Learning, pp. 181\u2013189 (2014)"},{"key":"9560_CR43","unstructured":"Kulis, B., Bartlett, P.L.: Implicit online learning. In: Proceedings of the 27th International Conference on Machine Learning (ICML-10), pp. 575\u2013582 (2010)"},{"key":"9560_CR44","doi-asserted-by":"crossref","first-page":"1196","DOI":"10.1214\/aos\/1176344840","volume":"7","author":"TL Lai","year":"1979","unstructured":"Lai, T.L., Robbins, H.: Adaptive design and stochastic approximation. Ann. Stat. 7, 1196\u20131221 (1979)","journal-title":"Ann. Stat."},{"key":"9560_CR45","doi-asserted-by":"crossref","first-page":"425","DOI":"10.1111\/j.2517-6161.1995.tb02037.x","volume":"57","author":"K Lange","year":"1995","unstructured":"Lange, K.: A gradient algorithm locally equivalent to the EM algorithm. J. R. Stat. Soc. Ser. B 57, 425\u2013437 (1995)","journal-title":"J. R. Stat. Soc. Ser. B"},{"key":"9560_CR46","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4419-5945-4","volume-title":"Numerical Analysis for Statisticians","author":"K Lange","year":"2010","unstructured":"Lange, K.: Numerical Analysis for Statisticians. Springer, New York (2010)"},{"key":"9560_CR47","first-page":"217","volume":"16","author":"C Le","year":"2004","unstructured":"Le, C., Bottou Yann, L., Bottou, L.: Large scale online learning. Adv. Neural Inf. Process. Syst. 16, 217 (2004)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"9560_CR48","volume-title":"Theory of Point Estimation","author":"EH Lehmann","year":"2003","unstructured":"Lehmann, E.H., Casella, G.: Theory of Point Estimation, 2nd edn. Springer, New York (2003)","edition":"2"},{"key":"9560_CR49","doi-asserted-by":"crossref","unstructured":"Li, L.: A worst-case comparison between temporal difference and residual gradient with linear function approximation. In: Proceedings of the 25th International Conference on Machine Learning, ACM, pp. 560\u2013567","DOI":"10.1145\/1390156.1390227"},{"issue":"4","key":"9560_CR50","doi-asserted-by":"crossref","first-page":"1052","DOI":"10.1016\/j.csda.2004.11.002","volume":"50","author":"Z Liu","year":"2006","unstructured":"Liu, Z., Almhana, J., Choulakian, V., McGorman, R.: Online em algorithm for mixture with application to internet traffic modeling. Comput. Stat. Data Anal. 50(4), 1052\u20131071 (2006)","journal-title":"Comput. Stat. Data Anal."},{"key":"9560_CR51","doi-asserted-by":"crossref","unstructured":"Ljung, L., Pflug, G., Walk, H.: Stochastic Approximation and Optimization of Random Systems, vol. 17. Springer, New York (1992)","DOI":"10.1007\/978-3-0348-8609-3"},{"issue":"3","key":"9560_CR52","doi-asserted-by":"crossref","first-page":"263","DOI":"10.1109\/TIT.1975.1055386","volume":"21","author":"RD Martin","year":"1975","unstructured":"Martin, R.D., Masreliez, C.: Robust estimation via stochastic approximation. IEEE Trans. Inf. Theory 21(3), 263\u2013271 (1975)","journal-title":"IEEE Trans. Inf. Theory"},{"key":"9560_CR53","series-title":"Online Learning and Neural Networks","volume-title":"A Statistical Study of On-line Learning","author":"N Murata","year":"1998","unstructured":"Murata, N.: A Statistical Study of On-line Learning. Online Learning and Neural Networks. Cambridge University Press, Cambridge (1998)"},{"issue":"3","key":"9560_CR54","doi-asserted-by":"crossref","first-page":"282","DOI":"10.1109\/TAC.1967.1098599","volume":"12","author":"J-I Nagumo","year":"1967","unstructured":"Nagumo, J.-I., Noda, A.: A learning method for system identification. IEEE Trans. Autom. Control 12(3), 282\u2013287 (1967)","journal-title":"IEEE Trans. Autom. Control"},{"key":"9560_CR55","volume-title":"Frontiers in Massive Data Analysis","author":"National Research Council","year":"2013","unstructured":"National Research Council: Frontiers in Massive Data Analysis. The National Academies Press, Washington, DC (2013)"},{"key":"9560_CR56","doi-asserted-by":"crossref","unstructured":"Neal, R.M., Hinton, G.E.: A view of the em algorithm that justifies incremental, sparse, and other variants. In: Learning in Graphical Models, pp. 355\u2013368. Springer, New York (1998)","DOI":"10.1007\/978-94-011-5014-9_12"},{"key":"9560_CR57","doi-asserted-by":"crossref","unstructured":"Neal, R.: Mcmc Using Hamiltonian Dynamics. Handbook of Markov Chain Monte Carlo 2 (2011)","DOI":"10.1201\/b10905-6"},{"key":"9560_CR58","volume-title":"Problem Complexity and Method Efficiency in Optimization","author":"AS Nemirovski","year":"1983","unstructured":"Nemirovski, A.S., Yudin, D.B.: Problem Complexity and Method Efficiency in Optimization. Wiley, Chichester (1983)"},{"issue":"4","key":"9560_CR59","doi-asserted-by":"crossref","first-page":"1574","DOI":"10.1137\/070704277","volume":"19","author":"A Nemirovski","year":"2009","unstructured":"Nemirovski, A., Juditsky, A., Lan, G., Shapiro, A.: Robust stochastic approximation approach to stochastic programming. SIAM J. Optim. 19(4), 1574\u20131609 (2009)","journal-title":"SIAM J. Optim."},{"key":"9560_CR60","volume-title":"Stochastic Approximation and Recursive Estimation","author":"MB Nevelson","year":"1973","unstructured":"Nevelson, M.B., Khasminski\u012d, R.Z.: Stochastic Approximation and Recursive Estimation, vol. 47. American Mathematical Society, Providence (1973)"},{"key":"9560_CR61","unstructured":"Nowlan, S.J.: Soft Competitive Adaptation: Neural Network Learning Algorithms Based on Fitting Statistical Mixtures. Carnegie\u00a0Mellon University, Pittsburgh (1991)"},{"issue":"3","key":"9560_CR62","first-page":"123","volume":"1","author":"N Parikh","year":"2013","unstructured":"Parikh, N., Boyd, S.: Proximal algorithms. Found. Trends Optim. 1(3), 123\u2013231 (2013)","journal-title":"Found. Trends Optim."},{"key":"9560_CR63","unstructured":"Pillai, N.S., Smith, A.: Ergodicity of approximate mcmc chains with applications to large data sets. arXiv preprint http:\/\/arxiv.org\/abs\/1405.0182 (2014)"},{"key":"9560_CR64","first-page":"74","volume":"3","author":"BT Polyak","year":"1979","unstructured":"Polyak, B.T., Tsypkin, Y.Z.: Adaptive algorithms of estimation (convergence, optimality, stability). Autom. Remote Control 3, 74\u201384 (1979)","journal-title":"Autom. Remote Control"},{"key":"9560_CR65","doi-asserted-by":"crossref","unstructured":"Polyak, B.T., Juditsky, A.B.: Acceleration of stochastic approximation by averaging. SIAM J. Control Optim. 30(4), 838\u2013855 (1992)","DOI":"10.1137\/0330046"},{"key":"9560_CR66","doi-asserted-by":"crossref","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins, H., Monro, S.: A stochastic approximation method. Ann. Math. Stat. 22, 400\u2013407 (1951)","journal-title":"Ann. Math. Stat."},{"issue":"5","key":"9560_CR67","doi-asserted-by":"crossref","first-page":"877","DOI":"10.1137\/0314056","volume":"14","author":"RT Rockafellar","year":"1976","unstructured":"Rockafellar, R.T.: Monotone operators and the proximal point algorithm. SIAM J. Control Optim. 14(5), 877\u2013898 (1976)","journal-title":"SIAM J. Control Optim."},{"key":"9560_CR68","unstructured":"Rosasco, L., Villa, S., C\u00f4ng V\u0169, B.: Convergence of stochastic proximal gradient algorithm. arXiv preprint http:\/\/arxiv.org\/abs\/1403.5074 , 2014"},{"key":"9560_CR69","unstructured":"Ruppert, D.: Efficient estimations from a slowly convergent robbins-monro process. Technical report, Cornell University Operations Research and Industrial Engineering (1988)"},{"key":"9560_CR70","unstructured":"Ryu, E.K., Boyd, S.: Stochastic proximal iteration: a non-asymptotic improvement upon stochastic gradient descent. Working paper. http:\/\/web.stanford.edu\/~eryu\/papers\/spi.pdf (2014)"},{"issue":"2","key":"9560_CR71","doi-asserted-by":"crossref","first-page":"373","DOI":"10.1214\/aoms\/1177706619","volume":"29","author":"J Sacks","year":"1958","unstructured":"Sacks, J.: Asymptotic distribution of stochastic approximation procedures. Ann. Math. Stat. 29(2), 373\u2013405 (1958)","journal-title":"Ann. Math. Stat."},{"issue":"4","key":"9560_CR72","doi-asserted-by":"crossref","first-page":"461","DOI":"10.1016\/0020-7225(65)90029-7","volume":"3","author":"DJ Sakrison","year":"1965","unstructured":"Sakrison, D.J.: Efficient recursive estimation; application to estimating the parameters of a covariance function. Int. J. Eng. Sci. 3(4), 461\u2013483 (1965)","journal-title":"Int. J. Eng. Sci."},{"key":"9560_CR73","doi-asserted-by":"crossref","unstructured":"Salakhutdinov, R., Mnih, A., Hinton, G.: Restricted boltzmann machines for collaborative filtering. In: Proceedings of the 24th International Conference on Machine Learning, ACM, pp. 791\u2013798 (2007)","DOI":"10.1145\/1273496.1273596"},{"issue":"2","key":"9560_CR74","doi-asserted-by":"crossref","first-page":"407","DOI":"10.1162\/089976600300015853","volume":"12","author":"M-A Sato","year":"2000","unstructured":"Sato, M.-A., Ishii, S.: On-line em algorithm for the normalized Gaussian network. Neural Comput. 12(2), 407\u2013432 (2000)","journal-title":"Neural Comput."},{"issue":"1","key":"9560_CR75","first-page":"982","volume":"32","author":"I Sato","year":"2014","unstructured":"Sato, I., Nakagawa, H.: Approximation analysis of stochastic gradient langevin dynamics by using Fokker-Planck equation and ito process. JMLR W&CP 32(1), 982\u2013990 (2014)","journal-title":"JMLR W&CP"},{"issue":"1\u20133","key":"9560_CR76","first-page":"95","volume":"22","author":"RE Schapire","year":"1996","unstructured":"Schapire, R.E., Warmuth, M.K.: On the worst-case analysis of temporal-difference learning algorithms. Mach. Learn. 22(1\u20133), 95\u2013121 (1996)","journal-title":"Mach. Learn."},{"key":"9560_CR77","unstructured":"Schaul, T., Zhang, S., LeCun, Y.: No more pesky learning rates. arXiv preprint. http:\/\/arxiv.org\/abs\/1206.1106 , 2012"},{"key":"9560_CR78","unstructured":"Schraudolph, N.N., Yu, J., G\u00fcnter, S.: A stochastic quasi-Newton method for online convex optimization. In: Meila M., Shen X. (eds.) Proceedings of the 11th International Conference on Artificial Intelligence and Statistics (AISTATS), vol. 2, pp. 436\u2013443. San Juan, Puerto Rico (2007)"},{"key":"9560_CR79","doi-asserted-by":"crossref","unstructured":"Slock, D.T.M.: On the convergence behavior of the LMS and the normalized LMS algorithms. IEEE Trans. Signal Process. 41(9), 2811\u20132825 (1993)","DOI":"10.1109\/78.236504"},{"issue":"1","key":"9560_CR80","first-page":"9","volume":"3","author":"RS Sutton","year":"1988","unstructured":"Sutton, R.S.: Learning to predict by the methods of temporal differences. Mach. Learn. 3(1), 9\u201344 (1988)","journal-title":"Mach. Learn."},{"key":"9560_CR81","unstructured":"Tamar, A., Toulis, P., Mannor, S., Airoldi, E.: Implicit temporal differences. In: Neural Information Processing Systems, Workshop on Large-Scale Reinforcement Learning (2014)"},{"key":"9560_CR82","first-page":"1345","volume":"19","author":"GW Taylor","year":"2006","unstructured":"Taylor, G.W., Hinton, G.E., Roweis, S.T.: Modeling human motion using binary latent variables. Adv. Neural Inf. Process. Syst. 19, 1345\u20131352 (2006)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"9560_CR83","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1111\/j.2517-6161.1984.tb01296.x","volume":"46","author":"MD Titterington","year":"1984","unstructured":"Titterington, M.D.: Recursive parameter estimation using incomplete data. J. R. Stat. Soc. Ser. B 46, 257\u2013267 (1984)","journal-title":"J. R. Stat. Soc. Ser. B"},{"key":"9560_CR84","unstructured":"Toulis, P., Airoldi, E.M.: Implicit stochastic gradient descent for principled estimation with large datasets. arXiv preprint http:\/\/arxiv.org\/abs\/1408.2923 , 2014"},{"issue":"1","key":"9560_CR85","first-page":"667","volume":"32","author":"P Toulis","year":"2014","unstructured":"Toulis, P., Airoldi, E., Rennie, J.: Statistical analysis of stochastic gradient methods for generalized linear models. JMLR W&CP 32(1), 667\u2013675 (2014)","journal-title":"JMLR W&CP"},{"key":"9560_CR86","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1214\/aoms\/1177699069","volume":"38","author":"JH Venter","year":"1967","unstructured":"Venter, J.H.: An extension of the robbins-monro procedur. Ann. Math. Stat. 38, 181\u2013190 (1967)","journal-title":"Ann. Math. Stat."},{"key":"9560_CR87","first-page":"181","volume":"26","author":"C Wang","year":"2013","unstructured":"Wang, C., Chen, X., Smola, A., Xing, E.: Variance reduction for stochastic gradient optimization. Adv. Neural Inf. Process. Syst. 26, 181\u2013189 (2013)","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"1","key":"9560_CR88","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1287\/moor.2013.0596","volume":"39","author":"M Wang","year":"2013","unstructured":"Wang, M., Bertsekas, D.P.: Stabilization of stochastic iterative methods for singular and nearly singular linear systems. Math. Oper. Res. 39(1), 1\u201330 (2013)","journal-title":"Math. Oper. Res."},{"key":"9560_CR89","first-page":"1115","volume":"3","author":"CZ Wei","year":"1987","unstructured":"Wei, C.Z.: Multivariate adaptive stochastic approximation. Ann. Stat. 3, 1115\u20131130 (1987)","journal-title":"Ann. Stat."},{"key":"9560_CR90","unstructured":"Welling, M., Teh, Y.W.: Bayesian learning via stochastic gradient langevin dynamics. In: Proceedings of the 28th International Conference on Machine Learning (ICML-11), pp. 681\u2013688 (2011)"},{"key":"9560_CR91","unstructured":"Xu, W.: Towards optimal one pass large scale learning with averaged stochastic gradient descent. arXiv preprint http:\/\/arxiv.org\/abs\/1107.2490 , 2011"},{"key":"9560_CR92","doi-asserted-by":"crossref","unstructured":"Younes, L.: On the convergence of markovian stochastic algorithms with rapidly decreasing ergodicity rates. Stochastics 65(3\u20134), 177\u2013228 (1999)","DOI":"10.1080\/17442509908834179"},{"key":"9560_CR93","doi-asserted-by":"crossref","unstructured":"Zhang, T.: Solving large scale linear prediction problems using stochastic gradient descent algorithms. In: Proceedings of the Twenty-First International Conference on Machine Learning, ACM, p. 116 (2004)","DOI":"10.1145\/1015330.1015332"}],"container-title":["Statistics and Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11222-015-9560-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11222-015-9560-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11222-015-9560-y","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,9]],"date-time":"2024-06-09T12:56:33Z","timestamp":1717937793000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11222-015-9560-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,6,11]]},"references-count":93,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2015,7]]}},"alternative-id":["9560"],"URL":"https:\/\/doi.org\/10.1007\/s11222-015-9560-y","relation":{},"ISSN":["0960-3174","1573-1375"],"issn-type":[{"value":"0960-3174","type":"print"},{"value":"1573-1375","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,6,11]]}}}