{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T12:39:44Z","timestamp":1783514384940,"version":"3.55.0"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2023,7,26]],"date-time":"2023-07-26T00:00:00Z","timestamp":1690329600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,7,26]],"date-time":"2023-07-26T00:00:00Z","timestamp":1690329600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Found Comput Math"],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1007\/s10208-023-09616-9","type":"journal-article","created":{"date-parts":[[2023,7,26]],"date-time":"2023-07-26T20:32:08Z","timestamp":1690403528000},"page":"1455-1483","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Optimality of Robust Online Learning"],"prefix":"10.1007","volume":"24","author":[{"given":"Zheng-Chu","family":"Guo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andreas","family":"Christmann","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lei","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,7,26]]},"reference":[{"key":"9616_CR1","doi-asserted-by":"publisher","first-page":"337","DOI":"10.1090\/S0002-9947-1950-0051437-7","volume":"68","author":"N Aronszajn","year":"1950","unstructured":"N. Aronszajn. Theory of reproducing kernels. Transactions of the American Mathematical Society, 68 (1950), 337\u2013404.","journal-title":"Transactions of the American Mathematical Society"},{"key":"9616_CR2","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1016\/j.jco.2006.07.001","volume":"23","author":"F Bauer","year":"2007","unstructured":"F. Bauer, S. Pereverzev, and L. Rosasco. On regularization algorithms in learning theory. Journal of complexity, 23 (2007), 52\u201372.","journal-title":"Journal of complexity"},{"key":"9616_CR3","doi-asserted-by":"publisher","first-page":"1657","DOI":"10.1109\/TPWRS.2009.2030291","volume":"24","author":"R Bessa","year":"2009","unstructured":"R. Bessa, V. Miranda, and J. Gama. Entropy and correntropy against minimum square error in offline and online three-day ahead wind power forecasting. IEEE Transactions on Power Systems, 24 (2009), 1657\u20131666.","journal-title":"IEEE Transactions on Power Systems"},{"key":"9616_CR4","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1006\/cviu.1996.0006","volume":"63","author":"M Black","year":"1996","unstructured":"M. Black and P. Anandan. The robust estimation of multiple motions: Parametric and piecewise-smooth flow fields. Computer vision and image understanding, 63 (1996), 75\u2013104.","journal-title":"Computer vision and image understanding"},{"key":"9616_CR5","doi-asserted-by":"publisher","first-page":"971","DOI":"10.1007\/s10208-017-9359-7","volume":"18","author":"G Blanchard","year":"2018","unstructured":"G. Blanchard and N. M\u00fccke. Optimal rates for regularization of statistical inverse Learning problems. Foundations of Computational Mathematics, 18 (2018), 971\u20131013.","journal-title":"Foundations of Computational Mathematics"},{"key":"9616_CR6","doi-asserted-by":"publisher","first-page":"223","DOI":"10.1137\/16M1080173","volume":"60","author":"L Bottou","year":"2018","unstructured":"L. Bottou, F. E Curtis, and J. Nocedal. Optimization methods for large-scale machine learning. SIAM Review, 60 (2018), 223\u2013311.","journal-title":"SIAM Review"},{"key":"9616_CR7","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1007\/s10208-006-0196-8","volume":"7","author":"A Caponnetto","year":"2007","unstructured":"A. Caponnetto and E. De Vito. Optimal rates for the regularized least squares algorithm. Foundations of Computational Mathematics, 7 (2007), 331\u2013368.","journal-title":"Foundations of Computational Mathematics"},{"key":"9616_CR8","doi-asserted-by":"crossref","unstructured":"X. Chen, B. Tang, J. Fan, and X. Guo. Online gradient descent algorithms for functional data learning. Journal of Complexity, page 101635, 2021.","DOI":"10.1016\/j.jco.2021.101635"},{"key":"9616_CR9","doi-asserted-by":"publisher","first-page":"331","DOI":"10.4310\/SII.2009.v2.n3.a5","volume":"2","author":"A Christmann","year":"2009","unstructured":"A. Christmann and A. Van Messem, and I. Steinwart. On consistency and robustness properties of support vector machines for heavy-tailed distributions. Statistics and Its Interface, 2 (2009), 331\u2013327.","journal-title":"Statistics and Its Interface"},{"key":"9616_CR10","doi-asserted-by":"publisher","first-page":"799","DOI":"10.3150\/07-BEJ5102","volume":"13","author":"A Christmann","year":"2007","unstructured":"A. Christmann and I. Steinwart. Consistency and robustness of kernel-based regression in convex risk minimization. Bernoulli, 13 (2007), 799\u2013819.","journal-title":"Bernoulli"},{"key":"9616_CR11","doi-asserted-by":"crossref","unstructured":"F. Cucker and D. X. Zhou. Learning Theory: An Approximation Theory Viewpoint. Cambridge Univesity Press, 2007.","DOI":"10.1017\/CBO9780511618796"},{"key":"9616_CR12","doi-asserted-by":"crossref","unstructured":"K. De Brabanter, K. Pelckmans, J. De Brabanter, M. Debruyne, J. A. K. Suykens, M. Hubert, and B. De Moor. Robustness of kernel based regression: a comparison of iterative weighting schemes. International Conference on Artificial Neural Networks, (2009), 100\u2013110.","DOI":"10.1007\/978-3-642-04274-4_11"},{"key":"9616_CR13","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1016\/j.jmva.2009.09.007","volume":"101","author":"M Debruyne","year":"2010","unstructured":"M. Debruyne, A. Christmann, M. Hubert, and J. A. K. Suykens. Robustness of reweighted least squares kernel based regression. Journal of Multivariate Analysis, 101 (2010), 447\u2013463.","journal-title":"Journal of Multivariate Analysis"},{"key":"9616_CR14","doi-asserted-by":"publisher","first-page":"455","DOI":"10.1007\/s10208-010-9064-2","volume":"10","author":"E De Vito","year":"2010","unstructured":"E. De Vito, S. Pereverzyev, and L. Rosasco. Adaptive kernel methods using the balancing principle. Foundations of Computational Mathematics, 10 (2010), 455\u2013479.","journal-title":"Foundations of Computational Mathematics"},{"key":"9616_CR15","doi-asserted-by":"publisher","first-page":"1363","DOI":"10.1214\/15-AOS1391","volume":"44","author":"A Dieuleveut","year":"2016","unstructured":"A. Dieuleveut and F. Bach. Nonparametric stochastic approximation with large step-sizes. The Annals of Statistics, 44 (2016), 1363\u20131399.","journal-title":"The Annals of Statistics"},{"key":"9616_CR16","first-page":"667","volume":"3","author":"R Fair","year":"1974","unstructured":"R. Fair. On the robust estimation of econometric models. Annals of Economic and Social Measurement, 3 (1974), 667\u2013677.","journal-title":"Annals of Economic and Social Measurement"},{"key":"9616_CR17","doi-asserted-by":"publisher","first-page":"351","DOI":"10.3934\/mfc.2022021","volume":"5","author":"H Feng","year":"2021","unstructured":"H. Feng, S. Hou, L. Wei, and D. X. Zhou. CNN models for readability of Chinese texts. Mathematical Foundations of Computing, 5 (2021), 351\u2013362.","journal-title":"Mathematical Foundations of Computing"},{"key":"9616_CR18","first-page":"993","volume":"16","author":"Y Feng","year":"2015","unstructured":"Y. Feng, X. Huang, L. Shi, Y. Yang, and J. A. K. Suykens. Learning with the maximum correntropy criterion induced losses for regression. Journal of Machine Learning Research, 16 (2015), 993\u20131034.","journal-title":"Journal of Machine Learning Research"},{"key":"9616_CR19","doi-asserted-by":"publisher","first-page":"1656","DOI":"10.1162\/neco_a_01384","volume":"33","author":"Y Feng","year":"2021","unstructured":"Y. Feng and Q. Wu. A framework of learning through empirical gain maximization. Neural Computation, 33 (2021), 1656\u20131697.","journal-title":"Neural Computation"},{"key":"9616_CR20","unstructured":"S. Ganan and D. McClure. Bayesian image analysis: An application to single photon emission tomography. Journal of the American Statistical Association, (1985), 12\u201318."},{"key":"9616_CR21","doi-asserted-by":"crossref","unstructured":"X. Guo, Z. C. Guo, and L. Shi. Capacity dependent analysis for functional online learning algorithms. Applied and Computational Harmonic Analysis, 67 (2023), 1\u201330.","DOI":"10.1016\/j.acha.2023.06.002"},{"key":"9616_CR22","doi-asserted-by":"crossref","unstructured":"Z. C. Guo, T. Hu, and L. Shi. Gradient descent for robust kernel based regression. Inverse Problems, 34 (2018), 065009(29pp).","DOI":"10.1088\/1361-6420\/aabe55"},{"key":"9616_CR23","doi-asserted-by":"crossref","unstructured":"Z. C. Guo, S. B. Lin, and D. X. Zhou. Learning theory of distribued spectral algorithms. Inverse Problems, 33 (2017), 074009(29pp).","DOI":"10.1088\/1361-6420\/aa72b2"},{"key":"9616_CR24","first-page":"1","volume":"26","author":"ZC Guo","year":"2019","unstructured":"Z. C. Guo and L. Shi. Fast and strong convergence of online learning algorithms. Advances in Computational Mathematics, 26 (2019), 1\u201326.","journal-title":"Advances in Computational Mathematics"},{"key":"9616_CR25","volume-title":"Robust statistics: The Approach Based on Influence Functions","author":"FR Hampel","year":"1986","unstructured":"F.\u00a0R. Hampel, E.\u00a0M. Ronchetti and P.\u00a0J. Rousseeuw, and W.\u00a0A. Stahel. Robust statistics: The Approach Based on Influence Functions. John Wiley & Sons, New York, 1986."},{"key":"9616_CR26","doi-asserted-by":"publisher","first-page":"1561","DOI":"10.1109\/TPAMI.2010.220","volume":"33","author":"R He","year":"2011","unstructured":"R. He, W. Zheng, and B. Hu. Maximum correntropy criterion for robust face recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 33 (2011), 1561\u20131576.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"9616_CR27","doi-asserted-by":"publisher","first-page":"813","DOI":"10.1080\/03610927708827533","volume":"6","author":"PW Holland","year":"1977","unstructured":"P. W. Holland and R. E. Welsch. Robust regression using iteratively reweighted leastsquares. Communications in Statistics-Theory and Methods, 6 (1977), 813\u2013827.","journal-title":"Communications in Statistics-Theory and Methods"},{"key":"9616_CR28","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1142\/S0219530521500044","volume":"20","author":"S Huang","year":"2022","unstructured":"S. Huang, Y. Feng, and Q. Wu, Learning theory of minimum error entropy under weak moment conditions. Analysis and Applications, 20 (2022), 121\u2013139.","journal-title":"Analysis and Applications"},{"key":"9616_CR29","doi-asserted-by":"publisher","DOI":"10.1002\/0471725250","volume-title":"Robust Statistics","author":"P Huber","year":"1981","unstructured":"P. Huber. Robust Statistics. Wiley, New York, 1981."},{"key":"9616_CR30","unstructured":"J. Lin and L. Rosasco. Optimal learning for multi-pass stochastic gradient methods. In Advances in Neural Information Processing Systems, 4556\u20134564, 2016."},{"key":"9616_CR31","doi-asserted-by":"publisher","first-page":"5286","DOI":"10.1109\/TSP.2007.896065","volume":"55","author":"W Liu","year":"2007","unstructured":"W. Liu, P. Pokharel, and J. C. Principe. Correntropy: Properties and applications in non-Gaussian signal processing. IEEE Transactions on Signal Processing, 55 (2007), 5286\u20135298.","journal-title":"IEEE Transactions on Signal Processing"},{"key":"9616_CR32","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1016\/j.acha.2018.03.001","volume":"48","author":"S Lu","year":"2020","unstructured":"S. Lu, P. Math\u00e9, and S. V. Pereverzev. Balancing principle in supervised learning for a general regularization scheme. Applied and Computational Harmonic Analysis, 48 (2020), 123\u2013148.","journal-title":"Applied and Computational Harmonic Analysis"},{"key":"9616_CR33","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1142\/S0219530519410124","volume":"19","author":"F Lv","year":"2021","unstructured":"F. Lv and J. Fan, Optimal learning with Gaussians and correntropy loss. Analysis and Applications, 19(2021), 107\u2013124.","journal-title":"Analysis and Applications"},{"key":"9616_CR34","doi-asserted-by":"publisher","DOI":"10.1002\/0470010940","volume-title":"Robust Statistics","author":"R Maronna","year":"2006","unstructured":"R. Maronna, D. Martin, and V. Yohai. Robust Statistics. John Wiley & Sons, Chichester, 2006."},{"key":"9616_CR35","doi-asserted-by":"publisher","DOI":"10.1002\/0470010940","volume-title":"Robust Statistics: Theory and Methods","author":"RA Maronna","year":"2006","unstructured":"R.\u00a0A. Maronna and R.\u00a0D. Martin and V.\u00a0J. Yohai. Robust Statistics: Theory and Methods. John Wiley & Sons, New York, 2006."},{"key":"9616_CR36","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1016\/S0167-7152(02)00057-3","volume":"57","author":"I Mizera","year":"2002","unstructured":"I. Mizera and C. M\u00fcller. Breakdown points of Cauchy regression-scale estimators. Statistics & probability letters, 57 (2002), 79\u201389.","journal-title":"Statistics & probability letters"},{"key":"9616_CR37","unstructured":"L. Pillaud-Vivien, R. Alessandro, and F. Bach. Statistical optimality of stochastic gradient descent on hard learning problems through multiple passes. In Advances in Neural Information Processing Systems, 8114\u20138124, 2018."},{"key":"9616_CR38","doi-asserted-by":"publisher","first-page":"1574","DOI":"10.1137\/070704277","volume":"19","author":"A Nemirovski","year":"2009","unstructured":"A. Nemirovski, A. Juditsky, G. Lan, and A. Shapiro. Robust stochastic approximation approach to stochastic programming. SIAM Journal on Optimization, 19 (2009), 1574\u20131609.","journal-title":"SIAM Journal on Optimization"},{"key":"9616_CR39","unstructured":"A. Rakhlin, O. Shamir, and K. Sridharan. Making gradient descent optimal for strongly convex stochastic optimization. In Proceedings of the 29th International Conference on Machine Learning (ICML-12), 449\u2013456, 2012."},{"key":"9616_CR40","first-page":"335","volume":"15","author":"G Raskutti","year":"2014","unstructured":"G. Raskutti, M. J. Wainwright, and B. Yu. Early stopping and non-parametric regression: an optimal data-dependent stopping rule. Journal of Machine Learning Research, 15 (2014), 335\u2013366.","journal-title":"Journal of Machine Learning Research"},{"key":"9616_CR41","unstructured":"L. Rosasco, A, Tacchetti, and S. Villa. Regularization by early stopping for online learning algorithms. Stat, 1050 (2014), 30 pages."},{"key":"9616_CR42","doi-asserted-by":"publisher","first-page":"2187","DOI":"10.1109\/TSP.2006.872524","volume":"54","author":"I Santamar\u00eda","year":"2006","unstructured":"I. Santamar\u00eda, P. Pokharel, and J. C. Principe. Generalized correlation function: definition, properties, and application to blind equalization. IEEE Transactions on Signal Processing, 54 (2006), 2187\u20132197.","journal-title":"IEEE Transactions on Signal Processing"},{"key":"9616_CR43","unstructured":"B. Sch\u00f6lkopf and A. J. Smola. Learning with kernels: support vector machines, regularization, optimization, and beyond. MIT press, 2018."},{"key":"9616_CR44","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1142\/S0219530503000089","volume":"1","author":"S Smale","year":"2003","unstructured":"S. Smale and D. X. Zhou. Estimating the approximation error in learning theory. Analysis and Applications, 1 (2003), 17\u201341.","journal-title":"Analysis and Applications"},{"key":"9616_CR45","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1007\/s00365-006-0659-y","volume":"26","author":"S Smale","year":"2007","unstructured":"S. Smale and D. X. Zhou. Learning theory estimates via integral operators and their approximations. Constructive Approximation, 26 (2007), 153\u2013172.","journal-title":"Constructive Approximation"},{"key":"9616_CR46","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1142\/S0219530509001293","volume":"7","author":"S Smale","year":"2009","unstructured":"S. Smale and D. X. Zhou. Online learning with Markov sampling. Analysis and Applications, 7 (2009), 87\u2013113.","journal-title":"Analysis and Applications"},{"key":"9616_CR47","doi-asserted-by":"publisher","first-page":"225","DOI":"10.1007\/s00365-006-0662-3","volume":"26","author":"I Steinwart","year":"2017","unstructured":"I. Steinwart. How to compare different loss functions and their risks. Constructive Approximation, 26 (2017), 225\u2013287.","journal-title":"Constructive Approximation"},{"key":"9616_CR48","volume-title":"Support Vector Machines","author":"I Steinwart","year":"2008","unstructured":"I. Steinwart and A. Christmann. Support Vector Machines. Springer-Verlag, New York, 2008."},{"key":"9616_CR49","unstructured":"I. Steinwart, D. R. Hush, and C. Scovel. Optimal rates for regularized least squares regression. In The 22nd Annual Conference on Learning Theory (COLT), 2009."},{"key":"9616_CR50","doi-asserted-by":"crossref","unstructured":"D. Sun, S. Roth, and M. Black. Secrets of optical flow estimation and their principles. In 2010 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR), 2432\u20132439, 2010.","DOI":"10.1109\/CVPR.2010.5539939"},{"key":"9616_CR51","unstructured":"I. Sutskever, J. Martens, G. Dahl, and G. Hinton. On the importance of initialization and momentum in deep learning. In International Conference on Machine Learning (ICML-13), 1139\u20131147, 2013."},{"key":"9616_CR52","doi-asserted-by":"publisher","first-page":"6470","DOI":"10.1109\/TIT.2010.2079010","volume":"56","author":"Y Yao","year":"2010","unstructured":"Y. Yao. On complexity issues of online learning algorithms. IEEE Transactions on Information Theory, 56 (2010), 6470\u20136481.","journal-title":"IEEE Transactions on Information Theory"},{"key":"9616_CR53","doi-asserted-by":"publisher","first-page":"561","DOI":"10.1007\/s10208-006-0237-y","volume":"8","author":"Y Ying","year":"2008","unstructured":"Y. Ying and M. Pontil. Online gradient descent learning algorithms. Foundations of Computational Mathematics, 8 (2008), 561\u2013596.","journal-title":"Foundations of Computational Mathematics"},{"key":"9616_CR54","doi-asserted-by":"publisher","first-page":"224","DOI":"10.1016\/j.acha.2015.08.007","volume":"42","author":"Y Ying","year":"2017","unstructured":"Y. Ying and D. X. Zhou. Unregularized online learning algorithms with general loss functions. Applied and Computational Harmonic Analysis, 42 (2017), 224\u2013244.","journal-title":"Applied and Computational Harmonic Analysis"},{"key":"9616_CR55","doi-asserted-by":"crossref","unstructured":"T. Zhang. Solving large scale linear prediction problems using stochastic gradient descent algorithms. In International Conference on Machine Learning (ICML-04), 919\u2013926, 2004.","DOI":"10.1145\/1015330.1015332"},{"key":"9616_CR56","doi-asserted-by":"publisher","first-page":"203","DOI":"10.3934\/mfc.2022018","volume":"6","author":"X Zhu","year":"2023","unstructured":"X. Zhu, Z. Li, and J. Sun. Expression recognition method combining convolutional features and Transformer. Mathematical Foundations of Computing, 6 (2023), 203\u2013217.","journal-title":"Mathematical Foundations of Computing"}],"container-title":["Foundations of Computational Mathematics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10208-023-09616-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10208-023-09616-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10208-023-09616-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,18]],"date-time":"2024-11-18T16:03:35Z","timestamp":1731945815000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10208-023-09616-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,26]]},"references-count":56,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2024,10]]}},"alternative-id":["9616"],"URL":"https:\/\/doi.org\/10.1007\/s10208-023-09616-9","relation":{},"ISSN":["1615-3375","1615-3383"],"issn-type":[{"value":"1615-3375","type":"print"},{"value":"1615-3383","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,7,26]]},"assertion":[{"value":"30 December 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 January 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 April 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 July 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}