{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T14:48:07Z","timestamp":1784904487047,"version":"3.55.0"},"reference-count":103,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key R&#x0026;D Program of China","award":["2022ZD0160300"],"award-info":[{"award-number":["2022ZD0160300"]}]},{"name":"NSF China","award":["62276004"],"award-info":[{"award-number":["62276004"]}]},{"DOI":"10.13039\/100005144","name":"Qualcomm","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100005144","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Singapore Ministry of Education (MOE) Academic Research Fund Tier 1"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1109\/tpami.2024.3423382","type":"journal-article","created":{"date-parts":[[2024,7,4]],"date-time":"2024-07-04T17:30:50Z","timestamp":1720114250000},"page":"9508-9520","source":"Crossref","is-referenced-by-count":137,"title":["Adan: Adaptive Nesterov Momentum Algorithm for Faster Optimizing Deep Models"],"prefix":"10.1109","volume":"46","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8756-5981","authenticated-orcid":false,"given":"Xingyu","family":"Xie","sequence":"first","affiliation":[{"name":"State Key Lab of General AI, School of Intelligence Science and Technology, Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3400-8943","authenticated-orcid":false,"given":"Pan","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Computing and Information Systems, Singapore Management University, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1286-6552","authenticated-orcid":false,"given":"Huan","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Robotics and Automatic Information Systems, College of Artificial Intelligence, Nankai University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1493-7569","authenticated-orcid":false,"given":"Zhouchen","family":"Lin","sequence":"additional","affiliation":[{"name":"State Key Lab of General AI, School of Intelligence Science and Technology, Institute for Artificial Intelligence, Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8906-3777","authenticated-orcid":false,"given":"Shuicheng","family":"Yan","sequence":"additional","affiliation":[{"name":"Sea AI Lab, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.4324\/9781410605337-29"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2010.11929"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707749"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2339736"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729586"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511569920.003"},{"issue":"7","key":"ref9","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"Duchi","year":"2011","journal-title":"J. Mach. Learn. Res."},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.4135\/9781412983907.n1717"},{"key":"ref11","article-title":"On the convergence of adam and beyond","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Reddi"},{"key":"ref12","article-title":"On the convergence of a class of adam-type algorithms for non-convex optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chen"},{"key":"ref13","article-title":"Adaptive gradient methods with dynamic bound of learning rate","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Luo"},{"key":"ref14","article-title":"On the variance of the adaptive learning rate and beyond","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Liu"},{"key":"ref15","first-page":"18795","article-title":"AdaBelief optimizer: Adapting stepsizes by the belief in observed gradients","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhuang"},{"key":"ref16","article-title":"AdamP: Slowing down the slowdown for momentum optimizers on scale-invariant weights","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Heo"},{"key":"ref17","article-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014"},{"key":"ref18","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Touvron"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01055"},{"key":"ref21","article-title":"Decoupled weight decay regularization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Loshchilov"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01204"},{"key":"ref23","article-title":"Large batch training of convolutional networks","author":"You","year":"2017"},{"key":"ref24","article-title":"Large batch optimization for deep learning: Training bert in 76 minutes","volume-title":"Proc. Int. Conf. Learn. Representations","author":"You"},{"issue":"83","key":"ref25","first-page":"1","article-title":"Win: Weight-decay-integrated nesterov acceleration for faster network training","volume":"25","author":"Zhou","year":"2024","journal-title":"J. Mach. Learn. Res."},{"issue":"3","key":"ref26","first-page":"543","article-title":"A method for solving the convex programming problem with convergence rate $\\mathcal {O}(1\/k^{2})$O(1\/k2)","volume":"269","author":"Nesterov","year":"1983","journal-title":"Doklady Akademii Nauk"},{"issue":"3","key":"ref27","first-page":"509","article-title":"On an approach to the construction of optimal methods of minimization of smooth convex functions","volume":"24","author":"Nesterov","year":"1988","journal-title":"Ekonomika i Mateaticheskie Metody"},{"key":"ref28","volume-title":"Introductory Lectures on Convex Optimization: A Basic Course","volume":"87","author":"Nesterov","year":"2003"},{"key":"ref29","article-title":"A large batch optimizer reality check: Traditional, generic optimizers suffice across batch sizes","author":"Nado","year":"2021"},{"key":"ref30","article-title":"Large-scale deep learning optimizations: A comprehensive survey","author":"He","year":"2021"},{"key":"ref31","article-title":"A novel convergence analysis for algorithms of the adam family","author":"Guo","year":"2021"},{"key":"ref32","article-title":"Adapting stepsizes by momentumized gradients improves optimization and generalization","author":"Wang","year":"2021"},{"key":"ref33","article-title":"Sharpness-aware minimization for efficiently improving generalization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Foret"},{"key":"ref34","first-page":"2260","article-title":"Momentum improves normalized SGD","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Cutkosky"},{"key":"ref35","article-title":"Adam$^+$+: A stochastic method with adaptive variance reduction","author":"Liu","year":"2020"},{"key":"ref36","first-page":"242","article-title":"Second-order information in non-convex stochastic optimization: Power and limitations","volume-title":"Proc. Conf. Learn. Theory","author":"Arjevani"},{"key":"ref37","article-title":"On the convergence of adaptive gradient methods for nonconvex optimization","author":"Zhou","year":"2018"},{"key":"ref38","first-page":"3267","article-title":"Closing the generalization gap of adaptive gradient methods in training deep neural networks","volume-title":"Proc. 29th Int. Conf. Int. Joint Conf. Artif. Intell.","author":"Chen"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-15-2910-8"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-022-01822-7"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19-1423"},{"key":"ref45","article-title":"PyTorch image models","author":"Wightman","year":"2019"},{"key":"ref46","article-title":"The DeepMind JAX ecosystem","author":"Babuschkin","year":"2020"},{"key":"ref47","article-title":"OpenMMLab\u2019s image classification toolbox and benchmark","author":"Contributors","year":"2020"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1016\/0041-5553(64)90137-5"},{"key":"ref49","first-page":"5905","article-title":"ASAM: Adaptive sharpness-aware minimization for scale-invariant learning of deep neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kwon"},{"key":"ref50","article-title":"Efficient sharpness-aware minimization for improved training of neural networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Du"},{"key":"ref51","first-page":"24430","article-title":"Adaptive inertia: Disentangling the effects of adaptive learning rate and momentum","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Xie"},{"key":"ref52","article-title":"Incorporating nesterov momentum into adam","volume-title":"Proc. Int. Conf. Learn. Representations Workshops","author":"Dozat"},{"key":"ref53","article-title":"Mixup: Beyond empirical risk minimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00612"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2020.03.001"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2021.3059101"},{"key":"ref57","first-page":"1216","article-title":"Diverse neural network learns true target functions","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Xie"},{"key":"ref58","first-page":"597","article-title":"Convergence analysis of two-layer neural networks with ReLU activation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref59","first-page":"745","article-title":"Stability and generalization of learning algorithms that converge to global optima","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Charles"},{"key":"ref60","article-title":"Towards understanding why lookahead generalizes better than SGD and beyond","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhou"},{"key":"ref61","first-page":"8119","article-title":"Tight bounds on the smallest eigenvalue of the neural tangent kernel for deep ReLU networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Nguyen"},{"key":"ref62","first-page":"11961","article-title":"Global convergence of deep networks with one wide layer followed by pyramidal topology","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Nguyen"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3181425"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/s44267-024-00043-0"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1561\/2400000003"},{"key":"ref66","article-title":"Understanding AdamW through proximal methods and scale-freeness","author":"Zhuang","year":"2022","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref67","first-page":"12901","article-title":"Restarted nonconvex accelerated gradient descent: No more polylogarithmic factor in the $\\mathcal {O}(\\epsilon ^{-7\/4})$O(\u03b5-7\/4) complexity","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li","year":"2022"},{"key":"ref68","first-page":"1042","article-title":"Accelerated gradient descent escapes saddle points faster than gradient descent","volume-title":"Proc. Conf. Learn. Theory","author":"Jin"},{"key":"ref69","article-title":"ResNet strikes back: An improved training procedure in timm","author":"Wightman","year":"2021"},{"key":"ref70","first-page":"9815","article-title":"Adaptive methods for nonconvex optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zaheer"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2024.3382294"},{"key":"ref72","article-title":"RMSProp converges with proper hyper-parameter","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Shi"},{"key":"ref73","article-title":"Deformable DETR: Deformable transformers for end-to-end object detection","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhu"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"issue":"8","key":"ref75","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00359"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_39"},{"key":"ref80","article-title":"Context autoencoder for self-supervised representation learning","author":"Chen","year":"2022"},{"key":"ref81","first-page":"15383","article-title":"Why are adaptive methods good for attention models?","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref82","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref83","article-title":"When vision transformers outperform resnets without pre-training or strong data augmentations","volume-title":"Proc. Int. Conf. Learn. Representation","author":"Chen"},{"key":"ref84","article-title":"MMDetection: Open MMLab detection toolbox and benchmark","author":"Chen","year":"2019"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref86","article-title":"Building a large annotated corpus of English: The penn treebank","volume":"273","author":"Marcinkiewicz","year":"1994","journal-title":"Using Large Corpora"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-4009"},{"key":"ref89","article-title":"RoBERTa: A robustly optimized BERT pretraining approach","author":"Liu","year":"2019"},{"key":"ref90","article-title":"The stack: 3 tb of permissively licensed source code","author":"Kocetkov","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref91","article-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"key":"ref92","article-title":"SPoC: Search-based pseudocode to code","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kulal"},{"key":"ref93","first-page":"1329","article-title":"Benchmarking deep reinforcement learning for continuous control","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Duan"},{"key":"ref94","article-title":"Tianshou: A highly modularized deep reinforcement learning library","volume":"23","author":"Weng","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref95","first-page":"22118","article-title":"Open graph benchmark: Datasets for machine learning on graphs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hu"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00936"},{"key":"ref97","article-title":"DeeperGCN: All you need to train deeper GCNs","author":"Li","year":"2020"},{"key":"ref98","first-page":"315","article-title":"Accelerating stochastic gradient descent using predictive variance reduction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Johnson"},{"key":"ref99","first-page":"382","article-title":"Algorithmic regularization in learning deep homogeneous models: Layers are automatically balanced","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Du"},{"key":"ref100","article-title":"Mixtral of experts","author":"Jiang","year":"2024"},{"key":"ref101","article-title":"RedPajama: An open dataset for training large language models","author":"Computer","year":"2023"},{"key":"ref102","first-page":"30392","article-title":"Early convolutions help transformers see better","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xiao"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20053-3_30"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/10746266\/10586270.pdf?arnumber=10586270","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:30:58Z","timestamp":1732667458000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10586270\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":103,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2024.3423382","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12]]}}}