{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T15:32:44Z","timestamp":1786980764706,"version":"3.56.0"},"reference-count":79,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2019,12,5]],"date-time":"2019-12-05T00:00:00Z","timestamp":1575504000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,12,5]],"date-time":"2019-12-05T00:00:00Z","timestamp":1575504000000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100006766","name":"Iran University of Science and Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100006766","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"published-print":{"date-parts":[[2020,8]]},"DOI":"10.1007\/s10462-019-09784-7","type":"journal-article","created":{"date-parts":[[2019,12,5]],"date-time":"2019-12-05T07:02:47Z","timestamp":1575529367000},"page":"3947-3986","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":267,"title":["A survey of regularization strategies for deep models"],"prefix":"10.1007","volume":"53","author":[{"given":"Reza","family":"Moradi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1302-571X","authenticated-orcid":false,"given":"Reza","family":"Berangi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Behrouz","family":"Minaei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,12,5]]},"reference":[{"issue":"1","key":"9784_CR1","first-page":"37","volume":"6","author":"D Aha","year":"1991","unstructured":"Aha D, Kibler D, Albert M (1991) Instance-based learning algorithms. Mach Learn 6(1):37\u201366","journal-title":"Mach Learn"},{"key":"9784_CR2","unstructured":"Ba JL, Kiros JR, Hinton GE (2016) Layer normalization, arXiv:1607.06450"},{"key":"9784_CR3","doi-asserted-by":"crossref","DOI":"10.1002\/9781118164471","volume-title":"The elements of integration and Lebesgue measure","author":"RG Bartle","year":"1995","unstructured":"Bartle RG (1995) The elements of integration and Lebesgue measure. Wiley, New Yor"},{"key":"9784_CR4","doi-asserted-by":"crossref","unstructured":"Bordes A, Chopra S, Weston J (2014) Question answering with subgraph embeddings. In: Empirical methods in natural language processing","DOI":"10.3115\/v1\/D14-1067"},{"key":"9784_CR5","unstructured":"Bouthillier X, Konda K, Vincent P, Memisevic R (2015) Dropout as data augmentation, arXiv:1506.08700"},{"key":"9784_CR6","unstructured":"Breiman L (1994) Bagging predictors. Mach Learn, pp 123\u2013140"},{"key":"9784_CR7","doi-asserted-by":"crossref","unstructured":"Chatfield K, Simonyan K, Vedaldi A, Zisserman A (2014) Return of the devil in the details: delving deep into convolutional nets. In: British machine vision","DOI":"10.5244\/C.28.6"},{"key":"9784_CR8","doi-asserted-by":"crossref","unstructured":"Chen PY, Zhang H, Sharma Y, Yi J, Hsieh CJ (2017) Zoo: zeroth order optimization based black-box attacks to deep neural networks without training substitute models. In: Proceedings of the 10th ACM workshop on artificial intelligence and security","DOI":"10.1145\/3128572.3140448"},{"key":"9784_CR9","doi-asserted-by":"crossref","unstructured":"Cohen D, Mitra B, Hofmann K, Croft WB (2018) Cross domain regularization for neural ranking models using adversarial learning, arXiv:1805.03403v1","DOI":"10.1145\/3209978.3210141"},{"key":"9784_CR10","first-page":"2493","volume":"12","author":"R Collobert","year":"2011","unstructured":"Collobert R, Weston J, Bottou L, Karlen M, Kavukcuoglu K, Kuksa P (2011) Natural language processing (almost) from scratch. J Mach Learn Res 12:2493\u20132537","journal-title":"J Mach Learn Res"},{"key":"9784_CR11","unstructured":"DeVries T, Taylor GW (2017) Improved regularization of convolutional neural networks with cutout, arXiv:1708.04552, 2017"},{"key":"9784_CR12","unstructured":"Domingos P (2000) A unified bias-variance decomposition and its applications. In: International conference on machine learning"},{"issue":"10","key":"9784_CR13","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1145\/2347736.2347755","volume":"55","author":"P Domingos","year":"2012","unstructured":"Domingos P (2012) A few useful things to know about machine learning. Commun ACM 55(10):78\u201387","journal-title":"Commun ACM"},{"key":"9784_CR14","unstructured":"Dong Y, Liao F, Pang T, Su H, Hu X, Li J, Zhu J (2017) Boosting adversarial attacks with momentum, arXiv:1710.06081"},{"key":"9784_CR15","unstructured":"Erhan D, Manzagol PA, Bengio Y, Bengio S, Vincent P (2009) The difficulty of training deep architectures and the effect of unsupervised pre-training. In: AISTATS"},{"key":"9784_CR16","doi-asserted-by":"crossref","unstructured":"Fraz\u00e3o XF, Alexandre LA (2014) DropAll: generalization of two convolutional neural network regularization methods. In: International conference on image analysis and recognition","DOI":"10.1007\/978-3-319-11758-4_31"},{"key":"9784_CR17","unstructured":"Gal Y, Ghahramani Z (2016) Dropout as a Bayesian approximation: representing model uncertainity in deep learning. In: Proceedings of the international conference on machine learning"},{"key":"9784_CR18","unstructured":"Gastaldi X (2017) Shake\u2013shake regularization, arXiv:1705.07485"},{"key":"9784_CR19","doi-asserted-by":"crossref","first-page":"452","DOI":"10.1038\/nature14541","volume":"521","author":"Z Ghahramani","year":"2015","unstructured":"Ghahramani Z (2015) Probabilistic machine learning and artificial intelligence. Nature 521:452\u2013459","journal-title":"Nature"},{"key":"9784_CR20","unstructured":"Gitman I, Ginsburg B (2017) Comparison of batch and weight normalization algorithms for largescale image classification, arXiv:1709.08145"},{"key":"9784_CR21","unstructured":"Goodfellow IJ, Warde-Farley D, Mirza M, Courville A, Bengio Y (2013) Maxout networks. In: International conference on machine learning"},{"key":"9784_CR22","unstructured":"Goodfellow IJ, Shlens J, Szegedy C (2014) Explaining and harnessing adversarial examples. CoRR. arXiv:1412.6572"},{"key":"9784_CR24","volume-title":"Deep learning","author":"I Goodfellow","year":"2016","unstructured":"Goodfellow I, Bengio Y, Courville A (2016) Deep learning. MIT Press, Cambridge"},{"key":"9784_CR25","unstructured":"Graham B (2015) Fractional max-pooling, arXiv:1412.6071"},{"key":"9784_CR26","unstructured":"He K, Zhang X, Ren S, Sun J (2015) Delving deep into rectifiers: surpassing human-level performance on ImageNet classification, arXiv:1502.01852"},{"key":"9784_CR27","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition, arXiv:1512.03385v1","DOI":"10.1109\/CVPR.2016.90"},{"key":"9784_CR28","doi-asserted-by":"crossref","first-page":"168","DOI":"10.1038\/nature12346","volume":"500","author":"M Helmstaedter","year":"2013","unstructured":"Helmstaedter M, Briggman KL, Turaga SC, Jain V, Seung HS, Denk W (2013) Connectomic reconstruction of the inner plexiform layer in the mouse retina. Nature 500:168\u2013174","journal-title":"Nature"},{"key":"9784_CR29","unstructured":"Henke N, Bughin J, Chui M, Manyika J, Saleh T, Wiseman B, Sethupathy G (2016) The age of analytics: competing in a data-driven world. In: McKinsey Global Institute"},{"issue":"6","key":"9784_CR30","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton G, Deng L, Yu D, Dahl G, Mohamed AR, Jaitly N, Senior A, Vanhoucke V, Nguyen P, Sainath TN, Kingsbury B (2012a) Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process Mag 29(6):82\u201397","journal-title":"IEEE Signal Process Mag"},{"issue":"6","key":"9784_CR31","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton G, Deng L, Yu D, Dahl G, Mohamed A, Jaitly N, Senior A, Vanhoucke V, Nguyen P, Sainath T, Kingsbury B (2012b) Deep neural networks for acoustic modeling in speech recognition. IEEE Signal Process Mag 29(6):82\u201397","journal-title":"IEEE Signal Process Mag"},{"key":"9784_CR32","unstructured":"Hinton G, Srivastava N, Krizhevsky A, Sutskever I, Salakhutdinov R (2012c) Improving neural networks by preventing co-adaptation of feature detectors, arXiv:1207.0580"},{"key":"9784_CR34","unstructured":"Hochreiter S, Schmidhuber J (1995) Simplifying neural nets by discovering flat minima. In: Advances in neural information processing systems, vol 7"},{"issue":"8","key":"9784_CR35","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"9784_CR36","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Weinberger KQ, Maaten L (2016a) Densely connected convolutional networks, arXiv:1608.06993","DOI":"10.1109\/CVPR.2017.243"},{"key":"9784_CR37","doi-asserted-by":"crossref","unstructured":"Huang G, Sun Y, Liu Z, Sedra D, Weinberger K (2016b) Deep networks with stochastic depth, arXiv:1603.09382","DOI":"10.1007\/978-3-319-46493-0_39"},{"key":"9784_CR38","unstructured":"Huang L, Liu X, Lang B, Yu AW, Wang W, Li B (2017) Orthogonal weight normalization: solution to optimization over multiple dependent stiefel manifolds in deep neural networks, arXiv:1709.06079"},{"key":"9784_CR39","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: accelerating deep network training by reducing internal covariate shift. arXiv:1502.03167"},{"key":"9784_CR40","doi-asserted-by":"crossref","unstructured":"Jakubovitz D, Giryes R (2018) Improving DNN robustness to adversarial attacks using jacobian regularization. In: European conference on computer vision","DOI":"10.1007\/978-3-030-01258-8_32"},{"key":"9784_CR41","doi-asserted-by":"crossref","unstructured":"Kang G, Li J, Tao D (2016) Shakeout: a new regularized deep neural network training scheme. In: Proceedings of the thirtieth AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v30i1.10202"},{"key":"9784_CR42","unstructured":"Krizhevsky A, Sutskever I, Hinton G (2012) ImageNet classification with deep convolutional neural networks. In: Advances in neural information processing systems"},{"issue":"6","key":"9784_CR43","doi-asserted-by":"crossref","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky A, Sutskever I, Hinton G (2017) ImageNet classification with deep convolutional neural networks. Commun ACM 60(6):84\u201390","journal-title":"Commun ACM"},{"key":"9784_CR44","unstructured":"Laarhoven TV (2017) L2 regularization versus batch and weight normalization. arXiv:1706.05350"},{"key":"9784_CR45","unstructured":"Larsson G, Maire M, Shakhnarovich G (2017) FractalNet: ultra-deep neural networks without residuals, arXiv:1605.07648"},{"key":"9784_CR46","doi-asserted-by":"crossref","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun Y, Bengio Y, Hinton G (2015) Deep learning. Nature 521:436\u2013444","journal-title":"Nature"},{"issue":"2","key":"9784_CR47","doi-asserted-by":"crossref","first-page":"263","DOI":"10.1021\/ci500747n","volume":"55","author":"J Ma","year":"2015","unstructured":"Ma J, Sheridan RP, Liaw A, Dahl GE, Svetnik V (2015) Deep neural nets as a method for quantitative structure-activity relationships. J Chem Inf Model 55(2):263\u2013274","journal-title":"J Chem Inf Model"},{"key":"9784_CR48","unstructured":"Maeda SI (2014) A Bayesian encourages dropout, arXiv:1412.7003"},{"key":"9784_CR49","doi-asserted-by":"crossref","unstructured":"Mash R, Borghetti B, Pecarina J (2016) Improved aircraft recognition for aerial refueling through data augmentation in convolutional neural networks. In: International symposium on visual computing","DOI":"10.1007\/978-3-319-50835-1_11"},{"key":"9784_CR50","doi-asserted-by":"crossref","first-page":"142","DOI":"10.1016\/j.eswa.2018.10.012","volume":"119","author":"R Moradi","year":"2019","unstructured":"Moradi R, Berangi R, Minaei B (2019) SparseMaps: convolutional networks with sparse feature maps for tiny image classification. Expert Syst Appl 119:142\u2013154","journal-title":"Expert Syst Appl"},{"key":"9784_CR51","unstructured":"Morerio P, Cavazza J, Volpi R, Vidal R, Murino V (2017) Curriculum dropout, arXiv:1703.06229"},{"key":"9784_CR52","unstructured":"Ng AY (1997) Preventing \u201coverfitting\u201d of cross-validation data. In: International conference on machine learning"},{"key":"9784_CR53","doi-asserted-by":"crossref","unstructured":"Peng H, Mou L, Li G, Chen Y, Lu Y, Jin Z (2015) A comparative study on regularization strategies for embedding-based neural networks. In: Empirical methods in natural language processing","DOI":"10.18653\/v1\/D15-1252"},{"key":"9784_CR54","unstructured":"Poole B, Sohl-Dickstein J, Ganguli S (2014) Analyzing noise in autoencoders and deep networks. In: CoRR"},{"key":"9784_CR55","unstructured":"Roth K, Lucchi A, Nowozin S, Hofmann T (2018) Adversarially Robust training through structured gradient regularization. arXiv:1805.08736v1"},{"key":"9784_CR56","doi-asserted-by":"crossref","unstructured":"Rozsa A, Rudd EM, Boult TE (2016) Adversarial diversity and hard positive generation. In: Proceedings of the IEEE conference on computer vision and pattern recognition workshops","DOI":"10.1109\/CVPRW.2016.58"},{"key":"9784_CR58","unstructured":"Salimans T, Kingma DP (2016) Weight normalization: a simple reparameterization to accelerate training of deep neural networks, arXiv:1602.07868"},{"key":"9784_CR59","doi-asserted-by":"crossref","unstructured":"Sankaranarayanan S, Jain A, Chellappa R, Lim SN (2018) Regularizing deep networks using efficient layerwise adversarial training. In: AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v32i1.11688"},{"key":"9784_CR60","unstructured":"Santurkar S, Tsipras D, Ilyas A, Madry A (2018) How does batch normalization help optimization? arXiv:1805.11604"},{"key":"9784_CR61","doi-asserted-by":"crossref","unstructured":"Shalev-Shwartz S, Ben-David S (2014) Rademacher complexities. In: Understanding machine learning\u2014from theory to algorithms. Cambridge University Press, Cambridge, pp 325\u2013336","DOI":"10.1017\/CBO9781107298019.027"},{"key":"9784_CR62","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava N, Hinton G, Krizhevsky A, Sutskever I, Salakhutdinov R (2014a) Dropout: a simple way to prevent neural networks from overfitting. J Mach Learn Res 15:1929\u20131958","journal-title":"J Mach Learn Res"},{"issue":"1","key":"9784_CR63","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava N, Hinton GE, Krizhevsky A, Sutskever I (2014b) A simple way to prevent neural network to prevent overfitting. Mach Learn Res 15(1):1929\u20131958","journal-title":"Mach Learn Res"},{"key":"9784_CR64","unstructured":"Su J, Vargas DV, Kouichi S (2017) One pixel attack for fooling deep neural networks, arXiv:1710.08864"},{"key":"9784_CR65","unstructured":"Sutskever I, Vinyals O, Le QV (2014) Sequence to sequence learning with neural networks. In: Advances in neural information processing systems"},{"key":"9784_CR66","unstructured":"Szegedy C, Zaremba W, Sutskever I, Bruna J, Erhan D, Goodfellow I, Fergus R (2013) Intriguing properties of neural networks, arXiv:1312.6199"},{"key":"9784_CR68","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2014) Going deeper with convolutions, arXiv:1409.4842"},{"key":"9784_CR69","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2015) Rethinking the inception architecture for computer vision, arXiv:1512.00567"},{"key":"9784_CR70","unstructured":"Taylor L, Nitschke G (2017) Improving deep learning using generic data augmentation, arXiv:1708.06020"},{"key":"9784_CR71","unstructured":"Wager S, Wang S, Liang PS (2013) Dropout training as adaptive regularizationauthor. In: Advances in neural information processing systems"},{"key":"9784_CR72","unstructured":"Wan L, Zeiler M, Zhang S, LeCun Y, Fergus R (2013) Regularization of neural networks using DropConnect. In: ICML, Department of Computer Science, Courant Institute of Mathematical Science, New York University, [Online]. Available: https:\/\/cs.nyu.edu\/~wanli\/dropc\/"},{"key":"9784_CR74","unstructured":"Wang Q, JaJa J (2013) From maxout to Channel-Out: Encoding information on sparse pathways, arXiv:1312.1909"},{"key":"9784_CR75","unstructured":"Wang S, Manning C (2013) Fast dropout training. In: International conference on machine learning"},{"key":"9784_CR76","unstructured":"Wen W, Wu C, Wang W, Chen Y, Li H (2016) Learning structured sparsity in deep neural networks. In: Advances in neural information processing systems"},{"key":"9784_CR77","doi-asserted-by":"crossref","first-page":"1341","DOI":"10.1162\/neco.1996.8.7.1341","volume":"8","author":"D Wolpert","year":"1996","unstructured":"Wolpert D (1996) The lack of a priori distinctions between learning algorithms. Neural Comput 8:1341\u20131390","journal-title":"Neural Comput"},{"key":"9784_CR78","unstructured":"Wu Y, He K (2018) Group normalization, arXiv:1803.08494"},{"issue":"6218","key":"9784_CR79","doi-asserted-by":"crossref","first-page":"1254806","DOI":"10.1126\/science.1254806","volume":"347","author":"HY Xiong","year":"2015","unstructured":"Xiong HY, Alipanahi B, Lee LJ, Bretschneider H, Merico D, Yuen RKC, Frey BJ (2015) The human splicing code reveals new insights into the genetic determinants of disease. Science 347(6218):1254806","journal-title":"Science"},{"key":"9784_CR80","unstructured":"Yu K, Xu W, Gong Y (2009) Deep learning with kernel regularization for visual recognition. In: Advances in neural information processing systems, vol 21"},{"key":"9784_CR81","unstructured":"Yuan X, He P, Zhu Q, Bhat RR, Li X (2017) Adversarial examples: attacks and defenses for deep learning, arXiv:1712.07107"},{"key":"9784_CR82","unstructured":"Zeiler MD, Fergus R (2013) Stochastic pooling for regularization of deep convolutional neural networks, arXiv:1301.3557"},{"key":"9784_CR83","doi-asserted-by":"crossref","unstructured":"Zeiler M, Fergus R (2014) Visualizing and understanding convolutional networks. In: IEEE European conference on computer vision","DOI":"10.1007\/978-3-319-10590-1_53"},{"key":"9784_CR84","unstructured":"Zhao Z, Dua D, Singh S (2017) Generating natural adversarial examples, arXiv:1710.11342"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-019-09784-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10462-019-09784-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-019-09784-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,7]],"date-time":"2022-10-07T15:37:49Z","timestamp":1665157069000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10462-019-09784-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12,5]]},"references-count":79,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2020,8]]}},"alternative-id":["9784"],"URL":"https:\/\/doi.org\/10.1007\/s10462-019-09784-7","relation":{},"ISSN":["0269-2821","1573-7462"],"issn-type":[{"value":"0269-2821","type":"print"},{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,12,5]]},"assertion":[{"value":"5 December 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}