{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T16:07:10Z","timestamp":1783526830421,"version":"3.55.0"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"10","license":[{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Science and Technology Innovation 2030\u2013\u201cBrain Science and Brain-Like Research\u201d Key Project","award":["2021ZD0201405"],"award-info":[{"award-number":["2021ZD0201405"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1109\/tnnls.2023.3279381","type":"journal-article","created":{"date-parts":[[2023,6,13]],"date-time":"2023-06-13T17:19:34Z","timestamp":1686676774000},"page":"14482-14490","source":"Crossref","is-referenced-by-count":23,"title":["A Unified Analysis of AdaGrad With Weighted Aggregation and Momentum Acceleration"],"prefix":"10.1109","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5659-3464","authenticated-orcid":false,"given":"Li","family":"Shen","sequence":"first","affiliation":[{"name":"JD Explore Academy, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9795-4200","authenticated-orcid":false,"given":"Congliang","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Science and Engineering, The Chinese University of Hong Kong (Shenzhen), Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9017-122X","authenticated-orcid":false,"given":"Fangyu","family":"Zou","sequence":"additional","affiliation":[{"name":"Meta Platforms Inc., Menlo Park, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3038-5891","authenticated-orcid":false,"given":"Zequn","family":"Jie","sequence":"additional","affiliation":[{"name":"Meituan Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2017-5903","authenticated-orcid":false,"given":"Ju","family":"Sun","sequence":"additional","affiliation":[{"name":"University of Minnesota at Twin Cities, Minneapolis, MN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3865-8145","authenticated-orcid":false,"given":"Wei","family":"Liu","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Deep Learning","volume":"1","author":"Goodfellow","year":"2016"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2020.3004171"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4612-5110-1_9"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1137\/16M1080173"},{"issue":"9","key":"ref5","first-page":"142","article-title":"Online learning and stochastic approximations","volume":"17","author":"Bottou","year":"1998","journal-title":"On-Line Learn. Neural Netw."},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1137\/120880811"},{"key":"ref7","first-page":"1646","article-title":"SAGA: A fast incremental gradient method with support for non-strongly convex composite objectives","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Defazio"},{"key":"ref8","first-page":"315","article-title":"Accelerating stochastic gradient descent using predictive variance reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Johnson"},{"key":"ref9","first-page":"2613","article-title":"SARAH: A novel method for machine learning problems using stochastic recursive gradient","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","volume":"70","author":"Nguyen"},{"key":"ref10","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"Duchi","year":"2011","journal-title":"J. Mach. Learn. Res."},{"key":"ref11","first-page":"1","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kingma"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/0041-5553(64)90137-5"},{"issue":"3","key":"ref13","first-page":"543","article-title":"A method for solving the convex programming problem with convergence rate o(1\/k2)","volume":"269","author":"Nesterov","year":"1983","journal-title":"Dokl. Akad. Nauk SSSR"},{"key":"ref14","first-page":"6500","article-title":"Online adaptive methods, universality and acceleration","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Levy"},{"key":"ref15","first-page":"244","article-title":"Adaptive bound optimization for online convex optimization","volume-title":"Proc. COLT","author":"McMahan"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ECC.2015.7330562"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4419-8853-9"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-015-0871-8"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/410"},{"key":"ref20","first-page":"1097","article-title":"ImageNet classification with deep convolutional neural networks","volume-title":"Proc. Adv. neural Inf. Process. Syst.","author":"Krizhevsky"},{"key":"ref21","first-page":"1139","article-title":"On the importance of initialization and momentum in deep learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Sutskever"},{"key":"ref22","first-page":"1","article-title":"On the convergence of Adam and beyond","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Reddi"},{"key":"ref23","first-page":"983","article-title":"On the convergence of stochastic gradient descent with adaptive stepsizes","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Li"},{"key":"ref24","first-page":"9739","article-title":"Adaptive methods for nonconvex optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Zaheer"},{"key":"ref25","first-page":"6677","article-title":"AdaGrad stepsizes: Sharp convergence over nonconvex landscapes","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ward"},{"key":"ref26","article-title":"WNGrad: Learn the learning rate in gradient descent","author":"Wu","year":"2018","journal-title":"arXiv:1803.02865"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01138"},{"key":"ref28","first-page":"1","article-title":"Towards practical Adam: Non-convexity, convergence theory, and mini-batch acceleration","volume":"23","author":"Chen","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref29","first-page":"1","article-title":"On the convergence of a class of Adam-type algorithms for non-convex optimization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Chen"},{"key":"ref30","article-title":"A novel convergence analysis for algorithms of the Adam family","author":"Guo","year":"2021","journal-title":"arXiv:2112.03459"},{"key":"ref31","first-page":"7664","article-title":"Surrogate losses for online learning of stepsizes in stochastic non-convex optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhuang"},{"key":"ref32","first-page":"102","article-title":"Efficient full-matrix adaptive regularization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Agarwal"},{"key":"ref33","article-title":"On the variance of the adaptive learning rate and beyond","author":"Liu","year":"2019","journal-title":"arXiv:1908.03265"},{"key":"ref34","first-page":"6257","article-title":"UniXGrad: A universal, adaptive algorithm with optimal guarantees for constrained optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kavis"},{"key":"ref35","first-page":"7202","article-title":"Zo-AdaMM: Zeroth-order adaptive momentum method for black-box optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen"},{"key":"ref36","article-title":"AdaSAM: Boosting sharpness-aware minimization with adaptive learning rate and momentum for training deep neural networks","author":"Sun","year":"2023","journal-title":"arXiv:2303.00565"},{"key":"ref37","article-title":"AdaTask: A task-aware adaptive learning rate approach to multi-task learning","author":"Yang","year":"2022","journal-title":"arXiv:2211.15055"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-emnlp.300"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/3470890"},{"key":"ref40","article-title":"Efficient-Adam: Communication-efficient distributed Adam with complexity analysis","author":"Chen","year":"2022","journal-title":"arXiv:2205.14473"},{"key":"ref41","first-page":"1613","article-title":"Online to offline conversions, universality and adaptive minibatch sizes","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Levy"},{"issue":"8","key":"ref42","first-page":"1","article-title":"Neural networks for machine learning lecture 6A overview of mini-batch gradient descent","volume":"14","author":"Hinton","year":"2012","journal-title":"Cited"},{"key":"ref43","first-page":"2545","article-title":"Variants of RMSProp and adagrad with logarithmic regret bounds","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Mukkamala"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref48","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10707065\/10149826.pdf?arnumber=10149826","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,9]],"date-time":"2024-10-09T17:51:56Z","timestamp":1728496316000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10149826\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10]]},"references-count":49,"journal-issue":{"issue":"10"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2023.3279381","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10]]}}}