{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T17:37:57Z","timestamp":1779385077825,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,24]],"date-time":"2024-08-24T00:00:00Z","timestamp":1724457600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NSFC","award":["No. 12326608"],"award-info":[{"award-number":["No. 12326608"]}]},{"name":"CAS Project for Young Scientists in Basic Research","award":["No. YSBR-034"],"award-info":[{"award-number":["No. YSBR-034"]}]},{"name":"Hetao Shenzhen-Hong Kong Science and Technology Innovation Cooperation Zone Project","award":["No.HZQSWS-KCCYB-2024016"],"award-info":[{"award-number":["No.HZQSWS-KCCYB-2024016"]}]},{"name":"the Strategic Priority Research Program of the Chinese Academy of Sciences","award":["No. XDB0680101"],"award-info":[{"award-number":["No. XDB0680101"]}]},{"name":"Innovation Project of ICT CAS","award":["No. E261090"],"award-info":[{"award-number":["No. E261090"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,25]]},"DOI":"10.1145\/3637528.3671718","type":"proceedings-article","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T04:54:55Z","timestamp":1724561695000},"page":"2960-2969","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Provable Adaptivity of Adam under Non-uniform Smoothness"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-6240-5203","authenticated-orcid":false,"given":"Bohan","family":"Wang","sequence":"first","affiliation":[{"name":"University of Science and Technology of China &amp; Microsoft Research Asia, Beijing, Haidian, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4464-5500","authenticated-orcid":false,"given":"Yushun","family":"Zhang","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, Shenzhen, Shenzhen, Guangdong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2711-7295","authenticated-orcid":false,"given":"Huishuai","family":"Zhang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3103-1999","authenticated-orcid":false,"given":"Qi","family":"Meng","sequence":"additional","affiliation":[{"name":"Chinese Academy of Mathematics and Systems Science, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2487-5322","authenticated-orcid":false,"given":"Ruoyu","family":"Sun","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, Shenzhen, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1169-7627","authenticated-orcid":false,"given":"Zhi-Ming","family":"Ma","sequence":"additional","affiliation":[{"name":"Chinese Academy of Mathematics and Systems Science, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0476-8020","authenticated-orcid":false,"given":"Tie-Yan","family":"Liu","sequence":"additional","affiliation":[{"name":"Microsoft, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3995-914X","authenticated-orcid":false,"given":"Zhi-Quan","family":"Luo","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, Shenzhen, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7438-5180","authenticated-orcid":false,"given":"Wei","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,8,24]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the symposium on learning and data science","volume":"8","author":"Bottou L\u00e9on","year":"2009","unstructured":"L\u00e9on Bottou. 2009. Curiously fast convergence of some stochastic gradient descent algorithms. In Proceedings of the symposium on learning and data science, Paris, Vol. 8. 2624--2633."},{"key":"e_1_3_2_1_2_1","volume-title":"Neural networks: Tricks of the trade","author":"Bottou L\u00e9on","unstructured":"L\u00e9on Bottou. 2012. Stochastic gradient descent tricks. In Neural networks: Tricks of the trade. Springer, 421--436."},{"key":"e_1_3_2_1_3_1","volume-title":"Large Scale GAN Training for High Fidelity Natural Image Synthesis. In International Conference on Learning Representations.","author":"Brock Andrew","year":"2018","unstructured":"Andrew Brock, Jeff Donahue, and Karen Simonyan. 2018. Large Scale GAN Training for High Fidelity Natural Image Synthesis. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_4_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_5_1","volume-title":"Convergence Theory, and Mini-Batch Acceleration. arXiv preprint arXiv:2101.05471","author":"Chen Congliang","year":"2021","unstructured":"Congliang Chen, Li Shen, Fangyu Zou, and Wei Liu. 2021. Towards Practical Adam: Non-Convexity, Convergence Theory, and Mini-Batch Acceleration. arXiv preprint arXiv:2101.05471 (2021)."},{"key":"e_1_3_2_1_6_1","volume-title":"On the convergence of a class of Adam-type algorithms for non-convex optimization. arXiv preprint arXiv:1808.02941","author":"Chen Xiangyi","year":"2018","unstructured":"Xiangyi Chen, Sijia Liu, Ruoyu Sun, and Mingyi Hong. 2018. On the convergence of a class of Adam-type algorithms for non-convex optimization. arXiv preprint arXiv:1808.02941 (2018)."},{"key":"e_1_3_2_1_7_1","volume-title":"Gradient descent on neural networks typically occurs at the edge of stability. arXiv preprint arXiv:2103.00065","author":"Cohen Jeremy M","year":"2021","unstructured":"Jeremy M Cohen, Simran Kaur, Yuanzhi Li, J Zico Kolter, and Ameet Talwalkar. 2021. Gradient descent on neural networks typically occurs at the edge of stability. arXiv preprint arXiv:2103.00065 (2021)."},{"key":"e_1_3_2_1_8_1","volume-title":"Robustness to Unbounded Smoothness of Generalized SignSGD. arXiv preprint arXiv:2208.11195","author":"Crawshaw Michael","year":"2022","unstructured":"Michael Crawshaw, Mingrui Liu, Francesco Orabona, Wei Zhang, and Zhenxun Zhuang. 2022. Robustness to Unbounded Smoothness of Generalized SignSGD. arXiv preprint arXiv:2208.11195 (2022)."},{"key":"e_1_3_2_1_9_1","volume-title":"Convergence guarantees for RMSProp and Adam in non-convex optimization and an empirical comparison to Nesterov acceleration. arXiv preprint arXiv:1807.06766","author":"De Soham","year":"2018","unstructured":"Soham De, Anirbit Mukherjee, and Enayat Ullah. 2018. Convergence guarantees for RMSProp and Adam in non-convex optimization and an empirical comparison to Nesterov acceleration. arXiv preprint arXiv:1807.06766 (2018)."},{"key":"e_1_3_2_1_10_1","volume-title":"A Simple Convergence Proof of Adam and AdaGrad. arXiv preprint arXiv:2003.02395","author":"D\u00e9fossez Alexandre","year":"2020","unstructured":"Alexandre D\u00e9fossez, L\u00e9on Bottou, Francis Bach, and Nicolas Usunier. 2020. A Simple Convergence Proof of Adam and AdaGrad. arXiv preprint arXiv:2003.02395 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"International Conference on Learning Representations.","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et al. 2020. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_12_1","unstructured":"Timothy Dozat. 2016. Incorporating Nesterov momentum into Adam. (2016)."},{"key":"e_1_3_2_1_13_1","volume-title":"The Power of Adaptivity in SGD: Self-Tuning Step Sizes with Unbounded Gradients and Affine Variance. arXiv preprint arXiv:2202.05791","author":"Faw Matthew","year":"2022","unstructured":"Matthew Faw, Isidoros Tziotis, Constantine Caramanis, Aryan Mokhtari, Sanjay Shakkottai, and Rachel Ward. 2022. The Power of Adaptivity in SGD: Self-Tuning Step Sizes with Unbounded Gradients and Affine Variance. arXiv preprint arXiv:2202.05791 (2022)."},{"key":"e_1_3_2_1_14_1","volume-title":"Asymptotic study of stochastic adaptive algorithm in non-convex landscape. arXiv preprint arXiv:2012.05640","author":"Gadat S\u00e9bastien","year":"2020","unstructured":"S\u00e9bastien Gadat and Ioana Gavra. 2020. Asymptotic study of stochastic adaptive algorithm in non-convex landscape. arXiv preprint arXiv:2012.05640 (2020)."},{"key":"e_1_3_2_1_15_1","volume-title":"Hamid Reza Feyzmahdavian, and Mikael Johansson","author":"Ghadimi Euhanna","year":"2015","unstructured":"Euhanna Ghadimi, Hamid Reza Feyzmahdavian, and Mikael Johansson. 2015. Global convergence of the heavy-ball method for convex optimization. In 2015 European control conference (ECC). IEEE, 310--315."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1137\/120880811"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-014-0846-1"},{"key":"e_1_3_2_1_18_1","volume-title":"A Novel Convergence Analysis for Algorithms of the Adam Family. arXiv preprint arXiv:2112.03459","author":"Guo Zhishuai","year":"2021","unstructured":"Zhishuai Guo, Yi Xu, Wotao Yin, Rong Jin, and Tianbao Yang. 2021. A Novel Convergence Analysis for Algorithms of the Adam Family. arXiv preprint arXiv:2112.03459 (2021)."},{"key":"e_1_3_2_1_19_1","first-page":"9074","article-title":"Super-Adam: faster and universal framework of adaptive gradients","volume":"34","author":"Huang Feihu","year":"2021","unstructured":"Feihu Huang, Junyi Li, and Heng Huang. 2021. Super-Adam: faster and universal framework of adaptive gradients. Advances in Neural Information Processing Systems 34 (2021), 9074--9085.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_20_1","first-page":"2771","article-title":"Non-convex distributionally robust optimization: Non-asymptotic analysis","volume":"34","author":"Jin Jikai","year":"2021","unstructured":"Jikai Jin, Bohang Zhang, Haiyang Wang, and Liwei Wang. 2021. Non-convex distributionally robust optimization: Non-asymptotic analysis. Advances in Neural Information Processing Systems 34 (2021), 2771--2782.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of NAACL-HLT. 4171--4186","author":"Ming-Wei Chang Jacob Devlin","year":"2019","unstructured":"Jacob Devlin Ming-Wei Chang Kenton and Lee Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of NAACL-HLT. 4171--4186."},{"key":"e_1_3_2_1_22_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_23_1","volume-title":"Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations.","author":"Kingma Diederik P","year":"2015","unstructured":"Diederik P Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_24_1","volume-title":"Convergence of Adam Under Relaxed Assumptions. arXiv preprint arXiv:2304.13972","author":"Li Haochuan","year":"2023","unstructured":"Haochuan Li, Ali Jadbabaie, and Alexander Rakhlin. 2023. Convergence of Adam Under Relaxed Assumptions. arXiv preprint arXiv:2304.13972 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"On the Variance of the Adaptive Learning Rate and Beyond. In International Conference on Learning Representations. https: \/\/openreview.net\/forum?id=rkgz2aEKDr","author":"Liu Liyuan","year":"2020","unstructured":"Liyuan Liu, Haoming Jiang, Pengcheng He,Weizhu Chen, Xiaodong Liu, Jianfeng Gao, and Jiawei Han. 2020. On the Variance of the Adaptive Learning Rate and Beyond. In International Conference on Learning Representations. https: \/\/openreview.net\/forum?id=rkgz2aEKDr"},{"key":"e_1_3_2_1_26_1","volume-title":"An Improved Analysis of Stochastic Gradient Descent with Momentum. Advances in Neural Information Processing Systems 33","author":"Liu Yanli","year":"2020","unstructured":"Yanli Liu, Yuan Gao, and Wotao Yin. 2020. An Improved Analysis of Stochastic Gradient Descent with Momentum. Advances in Neural Information Processing Systems 33 (2020)."},{"key":"e_1_3_2_1_27_1","volume-title":"Adaptive gradient methods with dynamic bound of learning rate. arXiv preprint arXiv:1902.09843","author":"Luo Liangchen","year":"2019","unstructured":"Liangchen Luo, Yuanhao Xiong, Yan Liu, and Xu Sun. 2019. Adaptive gradient methods with dynamic bound of learning rate. arXiv preprint arXiv:1902.09843 (2019)."},{"key":"e_1_3_2_1_28_1","volume-title":"Ahmed Khaled Ragab Bayoumi, and Peter Richt\u00e1rik","author":"Mishchenko Konstantin","year":"2020","unstructured":"Konstantin Mishchenko, Ahmed Khaled Ragab Bayoumi, and Peter Richt\u00e1rik. 2020. Random reshuffling: Simple analysis with vast improvements. Advances in Neural Information Processing Systems 33 (2020)."},{"key":"e_1_3_2_1_29_1","volume-title":"Unsupervised representation learning with deep convolutional generative adversarial networks. arXiv preprint arXiv:1511.06434","author":"Radford Alec","year":"2015","unstructured":"Alec Radford, Luke Metz, and Soumith Chintala. 2015. Unsupervised representation learning with deep convolutional generative adversarial networks. arXiv preprint arXiv:1511.06434 (2015)."},{"key":"e_1_3_2_1_30_1","volume-title":"On the Convergence of Adam and Beyond. In International Conference on Learning Representations.","author":"Reddi Sashank J","year":"2018","unstructured":"Sashank J Reddi, Satyen Kale, and Sanjiv Kumar. 2018. On the Convergence of Adam and Beyond. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_31_1","volume-title":"On the convergence of Adam and beyond. arXiv preprint arXiv:1904.09237","author":"Reddi Sashank J","year":"2019","unstructured":"Sashank J Reddi, Satyen Kale, and Sanjiv Kumar. 2019. On the convergence of Adam and beyond. arXiv preprint arXiv:1904.09237 (2019)."},{"key":"e_1_3_2_1_32_1","volume-title":"Fast convergence of stochastic gradient descent under a strong growth condition. arXiv preprint arXiv:1308.6370","author":"Schmidt Mark","year":"2013","unstructured":"Mark Schmidt and Nicolas Le Roux. 2013. Fast convergence of stochastic gradient descent under a strong growth condition. arXiv preprint arXiv:1308.6370 (2013)."},{"key":"e_1_3_2_1_33_1","volume-title":"International Conference on Learning Representations.","author":"Shi Naichen","year":"2021","unstructured":"Naichen Shi, Dawei Li, Mingyi Hong, and Ruoyu Sun. 2021. RMSprop converges with proper hyper-parameter. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_34_1","volume-title":"Advanced Optimization: Lecture 18 Proximal methods, Monotone operators. https:\/\/www.cs.cmu.edu\/suvrit\/teach\/lect18.pdf","author":"Sra Suvrit","year":"2014","unstructured":"Suvrit Sra. 2014. Advanced Optimization: Lecture 18 Proximal methods, Monotone operators. https:\/\/www.cs.cmu.edu\/suvrit\/teach\/lect18.pdf (2014)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-019-01437-5"},{"key":"e_1_3_2_1_36_1","volume-title":"International Conference on Machine Learning. PMLR, 10379--10389","author":"Tran Trang H","year":"2021","unstructured":"Trang H Tran, Lam M Nguyen, and Quoc Tran-Dinh. 2021. Smg: A shuffling gradient-based method with momentum. In International Conference on Machine Learning. PMLR, 10379--10389."},{"key":"e_1_3_2_1_37_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_1_38_1","volume-title":"The 22nd International Conference on Artificial Intelligence and Statistics. PMLR, 1195--1204","author":"Vaswani Sharan","year":"2019","unstructured":"Sharan Vaswani, Francis Bach, and Mark Schmidt. 2019. Fast and faster convergence of SGD for over-parameterized models and an accelerated perceptron. In The 22nd International Conference on Artificial Intelligence and Statistics. PMLR, 1195--1204."},{"key":"e_1_3_2_1_39_1","volume-title":"arXiv e-prints","author":"Yang Xiaodong","year":"2022","unstructured":"Xiaodong Yang, Huishuai Zhang, Wei Chen, and Tie-Yan Liu. 2022. Normalized\/ Clipped SGD with Perturbation for Differentially Private Non-Convex Optimization. arXiv e-prints (2022), arXiv-2206."},{"key":"e_1_3_2_1_40_1","volume-title":"Adaptive methods for nonconvex optimization. Advances in neural information processing systems 31","author":"Zaheer Manzil","year":"2018","unstructured":"Manzil Zaheer, Sashank Reddi, Devendra Sachan, Satyen Kale, and Sanjiv Kumar. 2018. Adaptive methods for nonconvex optimization. Advances in neural information processing systems 31 (2018)."},{"key":"e_1_3_2_1_41_1","volume-title":"Garnett (Eds.)","volume":"31","author":"Zaheer Manzil","year":"2018","unstructured":"Manzil Zaheer, Sashank Reddi, Devendra Sachan, Satyen Kale, and Sanjiv Kumar. 2018. Adaptive Methods for Nonconvex Optimization. In Advances in Neural Information Processing Systems, S. Bengio, H. Wallach, H. Larochelle, K. Grauman, N. Cesa-Bianchi, and R. Garnett (Eds.), Vol. 31. Curran Associates, Inc. https:\/\/proceedings.neurips.cc\/paper\/2018\/file\/ 90365351ccc7437a1309dc64e4db32a3-Paper.pdf"},{"key":"e_1_3_2_1_42_1","first-page":"15511","article-title":"Improved analysis of clipping algorithms for non-convex optimization","volume":"33","author":"Zhang Bohang","year":"2020","unstructured":"Bohang Zhang, Jikai Jin, Cong Fang, and LiweiWang. 2020. Improved analysis of clipping algorithms for non-convex optimization. Advances in Neural Information Processing Systems 33 (2020), 15511--15521.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_43_1","volume-title":"Why Gradient Clipping Accelerates Training: A Theoretical Justification for Adaptivity. In International Conference on Learning Representations.","author":"Zhang Jingzhao","year":"2019","unstructured":"Jingzhao Zhang, Tianxing He, Suvrit Sra, and Ali Jadbabaie. 2019. Why Gradient Clipping Accelerates Training: A Theoretical Justification for Adaptivity. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_44_1","volume-title":"Andreas Veit, Seungyeon Kim, Sashank J Reddi, Sanjiv Kumar, and Suvrit Sra.","author":"Zhang Jingzhao","year":"2019","unstructured":"Jingzhao Zhang, Sai Praneeth Karimireddy, Andreas Veit, Seungyeon Kim, Sashank J Reddi, Sanjiv Kumar, and Suvrit Sra. 2019. Why Adam beats SGD for attention models. (2019)."},{"key":"e_1_3_2_1_45_1","volume-title":"Why Transformers Need Adam: A Hessian Perspective. arXiv preprint arXiv:2402.16788","author":"Zhang Yushun","year":"2024","unstructured":"Yushun Zhang, Congliang Chen, Tian Ding, Ziniu Li, Ruoyu Sun, and Zhi-Quan Luo. 2024. Why Transformers Need Adam: A Hessian Perspective. arXiv preprint arXiv:2402.16788 (2024)."},{"key":"e_1_3_2_1_46_1","volume-title":"Adam Can Converge Without Any Modification on Update Rules. Advances in Neural Information Processing Systems","author":"Zhang Yushun","year":"2022","unstructured":"Yushun Zhang, Congliang Chen, Naichen Shi, Ruoyu Sun, and Zhi-Quan Luo. 2022. Adam Can Converge Without Any Modification on Update Rules. Advances in Neural Information Processing Systems (2022)."},{"key":"e_1_3_2_1_47_1","volume-title":"On the convergence of adaptive gradient methods for nonconvex optimization. arXiv preprint arXiv:1808.05671","author":"Zhou Dongruo","year":"2018","unstructured":"Dongruo Zhou, Jinghui Chen, Yuan Cao, Yiqi Tang, Ziyan Yang, and Quanquan Gu. 2018. On the convergence of adaptive gradient methods for nonconvex optimization. arXiv preprint arXiv:1808.05671 (2018)."},{"key":"e_1_3_2_1_48_1","volume-title":"Adashift: Decorrelation and convergence of adaptive learning rate methods. arXiv preprint arXiv:1810.00143","author":"Zhou Zhiming","year":"2018","unstructured":"Zhiming Zhou, Qingru Zhang, Guansong Lu, Hongwei Wang, Weinan Zhang, and Yong Yu. 2018. Adashift: Decorrelation and convergence of adaptive learning rate methods. arXiv preprint arXiv:1810.00143 (2018)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01138"}],"event":{"name":"KDD '24: The 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Barcelona Spain","acronym":"KDD '24","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671718","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3637528.3671718","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:06:01Z","timestamp":1750291561000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671718"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,24]]},"references-count":49,"alternative-id":["10.1145\/3637528.3671718","10.1145\/3637528"],"URL":"https:\/\/doi.org\/10.1145\/3637528.3671718","relation":{},"subject":[],"published":{"date-parts":[[2024,8,24]]},"assertion":[{"value":"2024-08-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}