{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:59:58Z","timestamp":1777874398074,"version":"3.51.4"},"reference-count":85,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100014013","name":"UKRI","doi-asserted-by":"publisher","award":["EP\/S0233356\/1"],"award-info":[{"award-number":["EP\/S0233356\/1"]}],"id":[{"id":"10.13039\/100014013","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000266","name":"Engineering and Physical Sciences Research Council","doi-asserted-by":"publisher","award":["EP\/X011364\/1"],"award-info":[{"award-number":["EP\/X011364\/1"]}],"id":[{"id":"10.13039\/501100000266","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00199","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"2055-2066","source":"Crossref","is-referenced-by-count":0,"title":["Large Learning Rates Simultaneously Achieve Robustness to Spurious Correlations and Compressibility"],"prefix":"10.1109","author":[{"given":"Melih","family":"Barsbey","sequence":"first","affiliation":[{"name":"Imperial College London"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lucas","family":"Prieto","sequence":"additional","affiliation":[{"name":"Imperial College London"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefanos","family":"Zafeiriou","sequence":"additional","affiliation":[{"name":"Imperial College London"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tolga","family":"Birdal","sequence":"additional","affiliation":[{"name":"Imperial College London"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"903","article-title":"SGD with large step sizes learns sparse features","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Andriushchenko"},{"key":"ref2","article-title":"Stronger Generalization Bounds for Deep Nets via a Compression Approach","volume-title":"International Conference on Learning Representations","author":"Arora"},{"key":"ref3","article-title":"Heavy Tails in SGD and Compressibility of Overparametrized Neural Networks","author":"Barsbey","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01270-0_28"},{"key":"ref5","article-title":"Intrinsic Dimension, Persistent Homology and Generalization in Neural Networks","author":"Birdal","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref6","first-page":"1","article-title":"A Universal Law of Robustness via Isoperimetry","author":"Bubeck","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref7","first-page":"7","author":"Chen","year":"2017","journal-title":"Rethinking Atrous Convolution for Semantic Image Segmentation"},{"key":"ref8","article-title":"Gradient Descent on Neural Networks Typically Occurs at the Edge of Stability","volume-title":"International Conference on Learning Representations","author":"Cohen"},{"key":"ref9","first-page":"1","author":"Coleman","year":"2023","journal-title":"AI\u2019s Climate Impact Goes beyond Its Emissions"},{"key":"ref10","first-page":"1","author":"Constine","year":"2023","journal-title":"The AI compute shortage explained by Nvidia, Crusoe, & MosaicML | AI Venture Capital"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0170"},{"key":"ref13","first-page":"6","article-title":"Pruning Deep Neural Networks from a Sparsity Perspective","volume-title":"The Eleventh International Conference on Learning Representations","author":"Diao"},{"key":"ref14","article-title":"A Winning Hand: Compressing Deep Networks Can Improve Out-of-Distribution Robustness","author":"Diffenderfer","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref15","first-page":"1019","article-title":"Sharp Minima Can Generalize For Deep Nets","volume-title":"Proceedings of the 34th International Conference on Machine Learning","author":"Dinh"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-64580-9_31"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3596490"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.eacl-main.129"},{"key":"ref19","first-page":"7","article-title":"Sharpness-aware Minimization for Efficiently Improving Generalization","volume-title":"International Conference on Learning Representations","author":"Foret"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.l803.03635"},{"key":"ref21","first-page":"1","author":"Fung","year":"2023","journal-title":"The big bottleneck for AI: A shortage of powerful chips | CNN Business"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-020-00257-z"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2012.2197174"},{"key":"ref24","first-page":"8","article-title":"The Heavy-Tail Phenomenon in SGD","volume-title":"International Conference on Machine Learning","author":"Gurbuzbalaban"},{"key":"ref25","first-page":"1","author":"Haan","year":"2024","journal-title":"Most Common Way Consumers Plan to Use Artificial Intelligence"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref27","first-page":"6","article-title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","volume-title":"First Conference on Language Modeling","author":"Hu"},{"key":"ref28","first-page":"336351","article-title":"Simple data balancing achieves competitive worst-group-accuracy","volume-title":"Proceedings of the First Conference on Causal Learning and Reasoning","author":"Youbi Idrissi"},{"key":"ref29","article-title":"The Break-Even Point on Optimization Trajectories of Deep Neural Networks","volume-title":"International Conference on Learning Representations","author":"Jastrzebski"},{"key":"ref30","author":"Shirish Keskar","year":"2017","journal-title":"On LargeBatch Training for Deep Learning: Generalization Gap and Sharp Minima"},{"key":"ref31","first-page":"6","article-title":"Adam: A method for stochastic optimization","volume-title":"International Conference on Learning Representations","author":"Kingma"},{"key":"ref32","article-title":"Last Layer Re-Training is Sufficient for Robustness to Spurious Correlations","volume-title":"The Eleventh International Conference on Learning Representations","author":"Kirichenko"},{"key":"ref33","first-page":"7","author":"Kokhlikyan","year":"2020","journal-title":"Captum: A unified and generic model interpretability library for pytorch"},{"key":"ref34","first-page":"4","article-title":"Why Do Better Loss Functions Lead to Less Transferable Features?","author":"Kornblith","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref36","article-title":"ASAM: Adaptive Sharpness-Aware Minimization for Scale-Invariant Learning of Deep Neural Networks","volume-title":"International Conference on Machine Learning","author":"Kwon"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN54540.2023.10191635"},{"key":"ref38","first-page":"4","article-title":"Selectivity considered harmful: Evaluating the causal impact of class selectivity in DNNs","volume-title":"International Conference on Learning Representations","author":"Leavitt"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/5.726791"},{"key":"ref40","author":"Lewkowycz","year":"2020","journal-title":"The large learning rate phase of deep learning: The catapult mechanism"},{"key":"ref41","first-page":"3","article-title":"Pruning Filters for Efficient ConvNets","volume-title":"International Conference on Learning Representations","author":"Li"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2017.2672978"},{"key":"ref43","article-title":"Towards Explaining the Regularization Effect of Initial Large Learning Rate in Training Neural Networks","author":"Li","year":"2019","journal-title":"Neural Information Processing Systems"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2858826"},{"key":"ref45","article-title":"Just Train Twice: Improving Group Robustness without Training Group Information","volume-title":"International Conference on Machine Learning","author":"Liu"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.425"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref48","first-page":"6","article-title":"SGDR: Stochastic Gradient Descent with Warm Restarts","volume-title":"International Conference on Learning Representations","author":"Loshchilov"},{"key":"ref49","first-page":"8","article-title":"Normalization and effective learning rates in reinforcement learning","author":"Lyle","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1145\/3679012"},{"key":"ref51","author":"Martin","year":"2019","journal-title":"Traditional and Heavy-Tailed Self Regularization in Neural Network Models"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.133"},{"key":"ref53","article-title":"Special Properties of Gradient Descent with Large Learning Rates","author":"Mohtashami","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref54","article-title":"Understanding the failure modes of out-ofdistribution generalization","volume-title":"International Conference on Learning Representations","author":"Nagarajan"},{"key":"ref55","first-page":"2","article-title":"Deep Double Descent: Where Bigger Models and More Data Hurt","volume-title":"Eighth International Conference on Learning Representations","author":"Nakkiran"},{"key":"ref56","article-title":"Learning from Failure: Training Debiased Classifier from Biased Classifier","author":"Nam","year":"2020","journal-title":"Advences in Neural Information Processing Systems"},{"key":"ref57","article-title":"Gradient Starvation: A Learning Proclivity in Neural Networks","author":"Pezeshki","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref58","first-page":"1","article-title":"Multi-scale Feature Learning Dynamics: Insights for Double Descent","volume-title":"International Conference on Machine Learning","author":"Pezeshki"},{"key":"ref59","article-title":"Don\u2019t blame Dataset Shift! Shortcut Learning due to Gradients and Cross Entropy","volume-title":"Thirty-Seventh Conference on Neural Information Processing Systems","author":"Manas Puli"},{"key":"ref60","article-title":"Complexity Matters: Feature Learning in the Presence of Spurious Correlations","volume-title":"International Conference on Machine Learning","author":"Qiu"},{"key":"ref61","first-page":"1292","article-title":"Algorithmic Stability of HeavyTailed Stochastic Gradient Descent on Least Squares","volume-title":"Proceedings of The 34th International Conference on Algorithmic Learning Theory","author":"Raj"},{"key":"ref62","first-page":"4","article-title":"On the special role of class-selective neurons in early training","author":"Ranadive","year":"2023","journal-title":"Transactions on Machine Learning Research"},{"key":"ref63","article-title":"Outliers with Opposing Signals Have an Outsized Effect on Neural Network Optimization","volume-title":"International Conference on Learning Representations","author":"Rosenfeld"},{"key":"ref64","article-title":"Distributionally Robust Neural Networks for Group Shifts: On the Importance of Regularization for Worst-Case Generalization","volume-title":"International Conference on Learning Representations","author":"Sagawa"},{"key":"ref65","first-page":"8346","article-title":"An investigation of why overparameterization exacerbates spurious correlations","volume-title":"Proceedings of the 37th International Conference on Machine Learning","author":"Sagawa"},{"key":"ref66","first-page":"8","article-title":"Weight normalization: A simple reparameterization to accelerate training of deep neural networks","author":"Salimans","year":"2016","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref67","first-page":"9573","article-title":"The Pitfalls of Simplicity Bias in Neural Networks","author":"Shah","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref68","first-page":"3145","article-title":"Learning important features through propagating activation differences","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Shrikumar"},{"key":"ref69","author":"Simonyan","year":"2015","journal-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition"},{"key":"ref70","first-page":"3319","article-title":"Axiomatic attribution for deep networks","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Sundararajan"},{"key":"ref71","first-page":"4","article-title":"Axiomatic attribution for deep networks","volume-title":"International Conference on Machine Learning","volume":"70","author":"Sundararajan"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/393"},{"key":"ref73","first-page":"1","article-title":"Compression based bound for non-compressed network: Unified generalization error analysis of large compressible deep neural network","volume-title":"International Conference on Learning Representations","author":"Suzuki"},{"key":"ref74","first-page":"3433034343","article-title":"Overcoming simplicity bias in deep networks using a feature sieve","volume-title":"International Conference on Machine Learning","author":"Tiwari"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.52202\/079017-4252"},{"key":"ref76","first-page":"5","article-title":"The caltech-ucsd birds-200-2011 dataset","volume-title":"Technical Report CNS-TR-2011-001, California Institute of Technology","author":"Wah","year":"2011"},{"key":"ref77","author":"Wang","year":"2018","journal-title":"Adversarial Robustness of Pruned Neural Networks"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0725"},{"key":"ref79","article-title":"A Fine-Grained Analysis on Distribution Shift","volume-title":"International Conference on Learning Representations","author":"Wiles"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/4235.585893"},{"key":"ref81","first-page":"6","article-title":"Model soups: Averaging weights of multiple fine-tuned models improves accuracy without increasing inference time","volume-title":"International Conference on Machine Learning","author":"Wortsman"},{"key":"ref82","first-page":"2953","article-title":"Identifying Spurious Biases Early in Training through the Lens of Simplicity Bias","volume-title":"Proceedings of The 27th International Conference on Artificial Intelligence and Statistics","author":"Yang"},{"key":"ref83","author":"Ye","year":"2024","journal-title":"Spurious Correlations in Machine Learning: A Survey"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.5244\/c.30.87"},{"key":"ref85","article-title":"Catapults in SGD: Spikes in the training loss and their impact on generalization through feature learning","volume-title":"International Conference on Machine Learning","author":"Zhu"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11443339.pdf?arnumber=11443339","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T19:45:03Z","timestamp":1777578303000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11443339\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":85,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00199","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}