{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T22:21:07Z","timestamp":1780957267171,"version":"3.54.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2022,11,24]],"date-time":"2022-11-24T00:00:00Z","timestamp":1669248000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,11,24]],"date-time":"2022-11-24T00:00:00Z","timestamp":1669248000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001866","name":"Fonds National de la Recherche Luxembourg","doi-asserted-by":"publisher","award":["CPPP17\/IS\/11643091\/IDform\/Aouada"],"award-info":[{"award-number":["CPPP17\/IS\/11643091\/IDform\/Aouada"]}],"id":[{"id":"10.13039\/501100001866","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001866","name":"Fonds National de la Recherche Luxembourg","doi-asserted-by":"publisher","award":["BRIDGES2020\/IS\/14755859\/MEET-A\/Aouada"],"award-info":[{"award-number":["BRIDGES2020\/IS\/14755859\/MEET-A\/Aouada"]}],"id":[{"id":"10.13039\/501100001866","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s10489-022-04230-8","type":"journal-article","created":{"date-parts":[[2022,11,24]],"date-time":"2022-11-24T16:58:23Z","timestamp":1669309103000},"page":"15621-15637","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":21,"title":["A new perspective for understanding generalization gap of deep neural networks trained with large batch sizes"],"prefix":"10.1007","volume":"53","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4652-1691","authenticated-orcid":false,"given":"Oyebade K.","family":"Oyedotun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Konstantinos","family":"Papadopoulos","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Djamila","family":"Aouada","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,24]]},"reference":[{"key":"4230_CR1","doi-asserted-by":"publisher","first-page":"21778","DOI":"10.1109\/ACCESS.2018.2825239","volume":"6","author":"Y-R Chien","year":"2018","unstructured":"Chien Y-R, Chen J-W, Xu SS-D (2018) A multilayer perceptron-based impulsive noise detector with application to power-line-based sensor networks. IEEE Access 6:21778\u201321787","journal-title":"IEEE Access"},{"issue":"17","key":"4230_CR2","doi-asserted-by":"publisher","first-page":"7941","DOI":"10.1007\/s00500-018-3424-2","volume":"23","author":"AA Heidari","year":"2019","unstructured":"Heidari AA, Faris H, Aljarah I, Mirjalili S (2019) An efficient hybrid multilayer perceptron neural network with grasshopper optimization. Soft Comput 23(17):7941\u20137958","journal-title":"Soft Comput"},{"issue":"8","key":"4230_CR3","doi-asserted-by":"publisher","first-page":"3560","DOI":"10.1109\/TNNLS.2017.2730179","volume":"29","author":"O Oyedotun","year":"2018","unstructured":"Oyedotun O, Khashman A (2018) Prototype-incorporated emotional neural network. IEEE Trans Neural Netw Learn Syst 29(8):3560\u20133572","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"4230_CR4","unstructured":"Schiffmann W, Joost M, Werner R (1993) Comparison of optimized backpropagation algorithms, vol 93, Citeseer, pp 97\u2013104"},{"issue":"6","key":"4230_CR5","doi-asserted-by":"publisher","first-page":"1452","DOI":"10.1109\/TPAMI.2017.2723009","volume":"40","author":"B Zhou","year":"2017","unstructured":"Zhou B, Lapedriza A, Khosla A, Oliva A, Torralba A (2017) Places: a 10 million image database for scene recognition. IEEE Trans Pattern Anal Mach Intell 40(6):1452\u20131464","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"4230_CR6","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Adv Neural Inform Process Syst:1097\u20131105"},{"key":"4230_CR7","doi-asserted-by":"crossref","unstructured":"Bansal A, Nanduri A, Castillo CD, Ranjan R, Chellappa R (2017) Umdfaces: an annotated face dataset for training deep networks. In: 2017 IEEE international joint conference on biometrics (IJCB). IEEE, pp 464\u2013473","DOI":"10.1109\/BTAS.2017.8272731"},{"key":"4230_CR8","doi-asserted-by":"crossref","unstructured":"Osawa K, Tsuji Y, Ueno Y, Naruse A, Yokota R, Matsuoka S (2019) Large-scale distributed second-order optimization using kronecker-factored approximate curvature for deep convolutional neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 12359\u201312367","DOI":"10.1109\/CVPR.2019.01264"},{"key":"4230_CR9","doi-asserted-by":"crossref","unstructured":"You Y, Zhang Z, Hsieh C-J, Demmel J, Keutzer K (2018) Imagenet training in minutes. In: Proceedings of the 47th international conference on parallel processing. ACM, pp 1","DOI":"10.1145\/3225058.3225069"},{"key":"4230_CR10","unstructured":"Hoffer E, Hubara I, Soudry D (2017) Train longer, generalize better: closing the generalization gap in large batch training of neural networks. In: Advances in neural information processing systems, pp 1731\u20131741"},{"key":"4230_CR11","unstructured":"Smith SL, Kindermans P-J, Ying C, Le QV (2018) Don\u2019t decay the learning rate, increase the batch size. In: International conference on learning representations, pp 1\u201311"},{"key":"4230_CR12","unstructured":"Goyal P, Doll\u00e1r P, Girshick R, Noordhuis P, Wesolowski L, Kyrola A, Tulloch A, Jia Y, He K (2017) Accurate, large minibatch sgd: training imagenet in 1 hour. arXiv:1706.02677"},{"key":"4230_CR13","unstructured":"Keskar NS, Mudigere D, Nocedal J, Smelyanskiy M, Tang PTP (2017) On large-batch training for deep learning: generalization gap and sharp minima. In: International conference on learning representations, pp 1\u201316"},{"key":"4230_CR14","unstructured":"Yao Z, Gholami A, Lei Q, Keutzer K, Mahoney M (2018) Hessian-based analysis of large batch training and robustness to adversaries. In: Advances in neural information processing systems, pp 4949\u20134959"},{"key":"4230_CR15","unstructured":"Masters D, Luschi C (2018) Revisiting small batch training for deep neural networks, arXiv:1804.07612"},{"issue":"1","key":"4230_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1162\/neco.1997.9.1.1","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Flat minima. Neural Comput 9(1):1\u201342","journal-title":"Neural Comput"},{"key":"4230_CR17","unstructured":"Hochreiter S, Schmidhuber J (1995) Simplifying neural nets by discovering flat minima. In: Advances in neural information processing systems, pp 529\u2013536"},{"key":"4230_CR18","unstructured":"You Y, Wang Y, Zhang H, Zhang Z, Demmel J, Hsieh C-J (2020) The limit of the batch size, arXiv:2006.08517"},{"key":"4230_CR19","unstructured":"Wen Y, Luk K, Gazeau M, Zhang G, Chan H, Ba J (2020) An empirical study of large-batch stochastic gradient descent with structured covariance noise. In: International conference on artificial intelligence and statistic, pp 1\u201315"},{"key":"4230_CR20","unstructured":"Wu J, Hu W, Xiong H, Huan J, Braverman V, Zhu Z (2020) On the noisy gradient descent that generalizes as sgd. In: International conference on machine learning. PMLR, pp 10367\u2013 10376"},{"key":"4230_CR21","unstructured":"Lin T, Kong L, Stich S, Jaggi M (2020) Extrapolation for large-batch training in deep learning. In: International conference on machine learning. PMLR, pp 6094\u20136104"},{"key":"4230_CR22","unstructured":"He F, Liu T, Tao D (2019) Control batch size and learning rate to generalize well: theoretical and empirical evidence. In: Advances in neural information processing systems, vol 32"},{"issue":"04","key":"4230_CR23","first-page":"5053","volume":"34","author":"L Ma","year":"2020","unstructured":"Ma L, Montague G, Ye J, Yao Z, Gholami A, Keutzer K, Mahoney M (2020) Inefficiency of k-fac for large batch size training. Proc AAAI Conf Artif Intell 34(04):5053\u20135060","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"4230_CR24","unstructured":"Saxe AM, McClelland JL, Ganguli S (2014) Exact solutions to the nonlinear dynamics of learning in deep linear neural networks. In: International conference on learning representations"},{"key":"4230_CR25","first-page":"586","volume":"29","author":"K Kawaguchi","year":"2016","unstructured":"Kawaguchi K (2016) Deep learning without poor local minima. Adv Neural Inform Process Syst 29:586\u2013594","journal-title":"Adv Neural Inform Process Syst"},{"key":"4230_CR26","unstructured":"Zhou Y, Liang Y (2019) Critical points of linear neural networks: analytical forms and landscape properties. In: International conference on learning representations"},{"key":"4230_CR27","unstructured":"Mulayoff R, Michaeli T (2020) Unique properties of flat minima in deep networks. In: International conference on machine learning, pp 7108\u20137118"},{"issue":"1","key":"4230_CR28","first-page":"31","volume":"20","author":"S Sonoda","year":"2019","unstructured":"Sonoda S, Murata N (2019) Transport analysis of infinitely deep neural network. J Mach Learn Res 20(1):31\u201382","journal-title":"J Mach Learn Res"},{"key":"4230_CR29","unstructured":"Laurent T, Brecht J (2018) Deep linear networks with arbitrary loss: all local minima are global. In: International conference on machine learning, pp 2902\u20132907"},{"key":"4230_CR30","doi-asserted-by":"crossref","unstructured":"Rudelson M, Vershynin R (2010) Non-asymptotic theory of random matrices: extreme singular values. In: Proceedings of the international congress of mathematicians 2010 (ICM 2010) (In 4 Volumes) Vol. I: plenary lectures and ceremonies Vols. II\u2013IV: invited Lectures. World scientific, pp 1576\u20131602","DOI":"10.1142\/9789814324359_0111"},{"issue":"3","key":"4230_CR31","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1109\/MSP.2010.936030","volume":"27","author":"G Bergqvist","year":"2010","unstructured":"Bergqvist G, Larsson EG (2010) The higher-order singular value decomposition: theory and an application [lecture notes]. IEEE Signal Process Mag 27(3):151\u2013154","journal-title":"IEEE Signal Process Mag"},{"issue":"3","key":"4230_CR32","doi-asserted-by":"publisher","first-page":"1008","DOI":"10.1137\/060655936","volume":"30","author":"R Badeau","year":"2008","unstructured":"Badeau R, Boyer R (2008) Fast multilinear singular value decomposition for structured tensors. SIAM J Matrix Anal Appl 30(3):1008\u20131021","journal-title":"SIAM J Matrix Anal Appl"},{"key":"4230_CR33","unstructured":"Krizhevsky A, Nair V, Hinton G (2020) Cifar-10, cifar-100 (canadian institute for advanced research). Last accessed, 21 August 2020, http:\/\/www.cs.toronto.edu\/kriz\/cifar.html"},{"key":"4230_CR34","unstructured":"Z Research (2020) Fashion-mnist handwritten dataset. Last accessed, 21 August 2020, https:\/\/github.com\/zalandoresearch\/fashion-mnist\/"},{"key":"4230_CR35","unstructured":"LeCun Y, Cortes C (2020) Mnist handwritten digit database. Last accessed, 21 August 2020, http:\/\/yann.lecun.com\/exdb\/mnist\/"},{"key":"4230_CR36","doi-asserted-by":"crossref","unstructured":"Ding X, Ding G, Han J, Tang S (2018) Auto-balanced filter pruning for efficient convolutional neural networks. In: Proceedings of the AAAI Conference on Artificial Intelligence, 32(1)","DOI":"10.1609\/aaai.v32i1.12262"},{"key":"4230_CR37","first-page":"1","volume":"20","author":"CJ Shallue","year":"2019","unstructured":"Shallue CJ, Lee J, Antognini J, Sohl-Dickstein J, Frostig R, Dahl GE (2019) Measuring the effects of data parallelism on neural network training. J Mach Learn Res 20:1\u201349","journal-title":"J Mach Learn Res"},{"key":"4230_CR38","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: accelerating deep network training by reducing internal covariate shift. In: International conference on machine learning, pp 448\u2013456"},{"key":"4230_CR39","unstructured":"Simonyan K, Zisserman A (2015) Very deep convolutional networks for large-scale image recognition. In: International conference on learning representations"},{"key":"4230_CR40","unstructured":"Chen L, Wang H, Zhao J, Papailiopoulos D, Koutris P (2018) The effect of network width on the performance of large-batch training. In: Advances in neural information processing systems, pp 9302\u20139309"},{"issue":"8","key":"4230_CR41","doi-asserted-by":"publisher","first-page":"1313","DOI":"10.1109\/TNN.2008.2000391","volume":"19","author":"F Cousseau","year":"2008","unstructured":"Cousseau F, Ozeki T, Amari S-i (2008) Dynamics of learning in multilayer perceptrons near singularities. IEEE Trans Neural Netw 19(8):1313\u20131328","journal-title":"IEEE Trans Neural Netw"},{"issue":"2","key":"4230_CR42","first-page":"26","volume":"4","author":"T Tieleman","year":"2012","unstructured":"Tieleman T, Hinton G (2012) Lecture 6.5-rmsprop: divide the gradient by a running average of its recent magnitude. COURSERA: Neural Netw Mach Learn 4(2):26\u201331","journal-title":"COURSERA: Neural Netw Mach Learn"},{"key":"4230_CR43","unstructured":"Akiba T, Suzuki S, Fukuda K (2017) Extremely large minibatch sgd: training resnet-50 on imagenet in 15 minutes. arXiv:1711.04325"},{"key":"4230_CR44","unstructured":"Summers C, Dinneen MJ (2020) Four things everyone should know to improve batch normalization. In: International conference for learning representations, pp 1\u201318"},{"key":"4230_CR45","doi-asserted-by":"crossref","unstructured":"Xie S, Girshick R, Doll\u00e1r P, Tu Z, He K (2017) Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1492\u20131500","DOI":"10.1109\/CVPR.2017.634"},{"issue":"1","key":"4230_CR46","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1007\/BF01013465","volume":"55","author":"N Etemadi","year":"1981","unstructured":"Etemadi N (1981) An elementary proof of the strong law of large numbers. Zeitschrift f\u00fc,r Wahrscheinlichkeitstheorie und verwandte Gebiete 55(1):119\u2013122","journal-title":"Zeitschrift f\u00fc,r Wahrscheinlichkeitstheorie und verwandte Gebiete"},{"key":"4230_CR47","first-page":"17","volume":"1","author":"ID Dinov","year":"2009","unstructured":"Dinov ID, Christou N, Gould R (2009) Law of large numbers: the theory, applications and technology-based education. J Stat Educ 1:17","journal-title":"J Stat Educ"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-04230-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-022-04230-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-022-04230-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,6,1]],"date-time":"2023-06-01T04:02:15Z","timestamp":1685592135000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-022-04230-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,24]]},"references-count":47,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["4230"],"URL":"https:\/\/doi.org\/10.1007\/s10489-022-04230-8","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,11,24]]},"assertion":[{"value":"1 October 2022","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 November 2022","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}