{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T16:04:59Z","timestamp":1777392299566,"version":"3.51.4"},"reference-count":52,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1109\/tai.2024.3523252","type":"journal-article","created":{"date-parts":[[2024,12,30]],"date-time":"2024-12-30T14:50:44Z","timestamp":1735570244000},"page":"1321-1333","source":"Crossref","is-referenced-by-count":3,"title":["Revisiting LARS for Large Batch Training Generalization of Neural Networks"],"prefix":"10.1109","volume":"6","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2566-5637","authenticated-orcid":false,"given":"Khoi","family":"Do","sequence":"first","affiliation":[{"name":"School of Computer Science and Statistics, Trinity College Dublin, Dublin 2, Ireland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minh-Duong","family":"Nguyen","sequence":"additional","affiliation":[{"name":"Department of Information Convergence Engineering, Pusan National University, Busan, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4743-5012","authenticated-orcid":false,"given":"Nguyen Tien","family":"Hoa","sequence":"additional","affiliation":[{"name":"School of Electrical and Electronic Engineering, Hanoi University of Science and Technology, Hanoi, Vietnam"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1617-8316","authenticated-orcid":false,"given":"Long","family":"Tran-Thanh","sequence":"additional","affiliation":[{"name":"School of Computer Science, University of Warwick, Coventry, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7323-9213","authenticated-orcid":false,"given":"Nguyen H.","family":"Tran","sequence":"additional","affiliation":[{"name":"School of Computer Science, The University of Sydney, Darlington, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9485-9216","authenticated-orcid":false,"given":"Quoc-Viet","family":"Pham","sequence":"additional","affiliation":[{"name":"School of Computer Science and Statistics, Trinity College Dublin, Dublin 2, Ireland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/tkde.2020.3015777"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCWAMTIP53232.2021.9674150"},{"key":"ref3","article-title":"Barlow twins: Self-supervised learning via redundancy reduction","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zbontar","year":"2021"},{"key":"ref4","article-title":"Big self-supervised models are strong semi-supervised learners","volume-title":"Proc. Adv. NIPS","author":"Chen","year":"2020"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref6","article-title":"Train longer, generalize better: closing the generalization gap in large batch training of neural networks","volume-title":"Proc. Adv. NIPS","author":"Hoffer","year":"2017"},{"key":"ref7","article-title":"On large-batch training for deep learning: Generalization gap and sharp minima","volume-title":"Proc. ICLR","author":"Keskar","year":"2017"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i9.16962"},{"key":"ref9","article-title":"Large batch optimization for deep learning: Training Bert in 76 minutes","volume-title":"Proc. ICLR","author":"You","year":"2020"},{"key":"ref10","article-title":"Improving layer-wise adaptive rate methods using trust ratio clipping","author":"Fong","year":"2020"},{"key":"ref11","article-title":"Large batch training of convolutional networks","author":"You","year":"2017"},{"key":"ref12","article-title":"Sharp minima can generalize for deep nets","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dinh","year":"2017"},{"key":"ref13","article-title":"Scale out for large minibatch SGD: Residual network training on ImageNet-1k with improved accuracy and reduced time to train","author":"Codreanu","year":"2017"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00647"},{"key":"ref15","article-title":"A closer look at deep learning heuristics: Learning rate restarts, warmup and distillation","volume-title":"Proc. ICLR","author":"Gotmare","year":"2019"},{"key":"ref16","article-title":"Accurate, large minibatch SGD: Training imagenet in 1 hour","author":"Goyal","year":"2018"},{"key":"ref17","article-title":"Highly scalable deep learning training system with mixed-precision: Training imagenet in four minutes","author":"Jia","year":"2018"},{"key":"ref18","first-page":"13925","article-title":"Communication-efficient distributed learning for large batch optimization","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Liu","year":"2022"},{"key":"ref19","article-title":"Yet another accelerated SGD: ResNet-50 training on imagenet in 74.7 seconds","author":"Yamazaki","year":"2019"},{"key":"ref20","first-page":"38801","article-title":"SLAMB: Accelerated large batch training with sparse communication","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","author":"Xu","year":"2023"},{"key":"ref21","article-title":"Large-batch optimization for dense visual predictions","volume-title":"Proc. Adv. NIPS","author":"Xue","year":"2022"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.1.1"},{"key":"ref23","article-title":"Learning rates as a function of batch size: A random matrix theory approach to neural network training","volume":"23","author":"Granziol","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref24","article-title":"Towards explaining the regularization effect of initial large learning rate in training neural networks","volume-title":"Proc. Adv. NIPS","author":"Li","year":"2019"},{"key":"ref25","article-title":"SGDR: Stochastic gradient descent with warm restarts","volume-title":"Proc. ICLR","author":"Loshchilov","year":"2017"},{"key":"ref26","article-title":"Automated learning rate scheduler for large-batch training","volume-title":"Proc. 8th ICML Workshop Automated Mach. Learn. (AutoML)","author":"Kim","year":"2021"},{"key":"ref27","article-title":"How does learning rate decay help modern neural networks?","author":"You","year":"2019"},{"key":"ref28","article-title":"One weird trick for parallelizing convolutional neural networks","author":"Krizhevsky","year":"2014"},{"key":"ref29","first-page":"32","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/T-AIEE.1928.5055024"},{"key":"ref31","first-page":"7611","article-title":"Tackling the objective inconsistency problem in heterogeneous federated optimization","volume-title":"Proc. Adv. NIPS","author":"Wang","year":"2020"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511779398"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729392"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654889"},{"key":"ref38","article-title":"Tensorflow: A system for large-scale machine learning","volume-title":"Proc. 12th USENIX Conf. Operating Syst. Des. Implementation","year":"2016"},{"key":"ref39","article-title":"VICReg: Variance-invariance-covariance regularization for self-supervised learning","volume-title":"Proc. ICLR","author":"Bardes","year":"2022"},{"key":"ref40","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume-title":"Proc. Adv. NIPS","year":"2019"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01215"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00089"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-017-1043-5_17"},{"key":"ref44","article-title":"Understanding the difficulty of training deep feedforward neural networks","volume-title":"Proc. 13th Int. Conf. Artif. Intell. Statist.","author":"Glorot","year":"2010"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref46","article-title":"EfficientNet: Rethinking model scaling for convolutional neural networks","author":"Tan","year":"2020"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00140"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2010.11929"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref51","article-title":"SAM 2: Segment anything in images and videos","year":"2024"},{"key":"ref52","article-title":"Llama 2: Open foundation and fine-tuned chat models","year":"2023"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9078688\/10980621\/10817779.pdf?arnumber=10817779","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T19:01:09Z","timestamp":1764270069000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10817779\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5]]},"references-count":52,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tai.2024.3523252","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5]]}}}