{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,13]],"date-time":"2026-03-13T04:52:03Z","timestamp":1773377523529,"version":"3.50.1"},"reference-count":47,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,7,12]],"date-time":"2021-07-12T00:00:00Z","timestamp":1626048000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,7,12]],"date-time":"2021-07-12T00:00:00Z","timestamp":1626048000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,7,12]]},"DOI":"10.1109\/isit45174.2021.9517811","type":"proceedings-article","created":{"date-parts":[[2021,9,1]],"date-time":"2021-09-01T16:52:42Z","timestamp":1630515162000},"page":"819-824","source":"Crossref","is-referenced-by-count":1,"title":["Self-Regularity of Output Weights for Overparameterized Two-Layer Neural Networks"],"prefix":"10.1109","author":[{"given":"David","family":"Gamarnik","sequence":"first","affiliation":[{"name":"MIT"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eren C.","family":"K\u0131z\u0131lda\u011f","sequence":"additional","affiliation":[{"name":"MIT"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ilias","family":"Zadik","sequence":"additional","affiliation":[{"name":"NYU"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","article-title":"Mildly overparametrized neural nets can memorize training data efficiently","author":"ge","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref38","first-page":"1783","article-title":"Learning one convolutional layer with overlapping patches","author":"goel","year":"0","journal-title":"Int Conference on Machine Learning"},{"key":"ref33","first-page":"1514","article-title":"Algorithms and sq lower bounds for pac learning one-hidden-layer relu networks","author":"diakonikolas","year":"0","journal-title":"Conference on Learning Theory"},{"key":"ref32","article-title":"Learning one-hidden-layer neural networks with landscape design","author":"ge","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref31","article-title":"Self-regularity of nonnegative output weights for overparameterized two-layer neural networks","author":"gamarnik","year":"2021","journal-title":"ArXiv Preprint"},{"key":"ref30","article-title":"Neural networks and polynomial regression. demystifying the overparametrization phenomena","author":"emschwiller","year":"2020","journal-title":"ArXiv Preprint"},{"key":"ref37","first-page":"1524","article-title":"Learning one-hidden-layer relu networks via gradient descent","author":"zhang","year":"0","journal-title":"In The 22nd International Conference on Artificial Intelligence and Statistics"},{"key":"ref36","first-page":"4433","article-title":"Spurious local minima are common in two-layer relu neural networks","author":"safran","year":"0","journal-title":"Int Conference on Machine Learning"},{"key":"ref35","first-page":"1329","article-title":"On the power of over-parametrization in neural networks with quadratic activation","author":"du","year":"0","journal-title":"Int Conference on Machine Learning"},{"key":"ref34","first-page":"2613","article-title":"Learning over-parametrized two-layer neural networks beyond ntk","author":"li","year":"0","journal-title":"Conference on Learning Theory"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1903070116"},{"key":"ref40","first-page":"1675","article-title":"Gradient descent finds global minima of deep neural networks","author":"du","year":"0","journal-title":"Int Conference on Machine Learning"},{"key":"ref11","first-page":"8139","article-title":"On exact computation with an infinitely wide neural net","author":"arora","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref12","first-page":"1064","article-title":"Nearly-tight vc-dimension bounds for piecewise linear neural networks","author":"harvey","year":"0","journal-title":"Conference on Learning Theory"},{"key":"ref13","first-page":"1","article-title":"Nearly-tight vc-dimension and pseudodimension bounds for piecewise linear neural networks","volume":"20","author":"bartlett","year":"2019","journal-title":"Journal of Machine Learning Research"},{"key":"ref14","first-page":"1376","article-title":"Norm-based capacity control in neural networks","author":"neyshabur","year":"0","journal-title":"Conference on Learning Theory"},{"key":"ref15","first-page":"6240","article-title":"Spectrally-normalized margin bounds for neural networks","author":"bartlett","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref16","article-title":"Fisher-rao metric, geometry, and complexity of neural networks","author":"liang","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref17","article-title":"Size-independent sample complexity of neural networks","author":"golowich","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref18","article-title":"Computing nonvacuous generalization bounds for deep (stochastic) neural networks with many more parameters than training data","author":"dziugaite","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref19","article-title":"A pac-bayesian approach to spectrally-normalized margin bounds for neural networks","author":"neyshabur","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref28","first-page":"605","article-title":"Generalization bounds of sgld for non-convex learning: Two theoretical viewpoints","author":"mou","year":"0","journal-title":"Conference on Learning Theory"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390177"},{"key":"ref27","first-page":"10 836","article-title":"Generalization bounds of stochastic gradient descent for wide and deep neural networks","author":"cao","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2109382"},{"key":"ref6","article-title":"Gradient descent provably optimizes over-parameterized neural networks","author":"du","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/18.661502"},{"key":"ref5","article-title":"Understanding deep learning requires rethinking generalization","author":"zhang","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref8","first-page":"9461","article-title":"Implicit bias of gradient descent on linear convolutional networks","author":"gunasekar","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref7","first-page":"8157","article-title":"Learning overparameterized neural networks via stochastic gradient descent on structured data","author":"li","year":"2018","journal-title":"Advances in neural information processing systems"},{"key":"ref2","first-page":"1097","article-title":"Imagenet classification with deep convolutional neural networks","author":"krizhevsky","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref9","first-page":"6979","article-title":"Dynamics of stochastic gradient descent for two-layer neural networks in the teacher-student setup","author":"goldt","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref46","first-page":"315","article-title":"Deep sparse rectifier neural networks","author":"glorot","year":"0","journal-title":"Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics"},{"key":"ref20","first-page":"5947","article-title":"Exploring generalization in deep learning","author":"neyshabur","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1017\/9781108591034"},{"key":"ref22","article-title":"Fine-grained analysis of optimization and generalization for overparameterized two-layer neural networks","author":"arora","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1006\/jcss.1996.0033"},{"key":"ref21","article-title":"Stronger generalization bounds for deep nets via a compression approach","author":"arora","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref42","article-title":"Why are convolutional nets more sample-efficient than fully-connected nets?","author":"li","year":"2020","journal-title":"ArXiv Preprint"},{"key":"ref24","first-page":"14 797","article-title":"Algorithm-dependent generalization bounds for overparameterized deep residual networks","author":"frei","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref41","first-page":"1470","article-title":"Learning neural networks with two nonlinear layers in polynomial time","author":"goel","year":"0","journal-title":"Conference on Learning Theory"},{"key":"ref23","first-page":"605","article-title":"Globally optimal gradient descent for a convnet with gaussian inputs","volume":"70","author":"brutzkus","year":"0","journal-title":"Proceedings of the 34th International Conference on Machine Learning"},{"key":"ref44","volume":"47","author":"vershynin","year":"2018","journal-title":"High-Dimensional Probability An Introduction with Applications in Data Science"},{"key":"ref26","article-title":"Sgd learns over-parameterized networks that provably generalize on linearly separable data","author":"brutzkus","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref43","article-title":"Introduction to the non-asymptotic analysis of random matrices","author":"vershynin","year":"2010","journal-title":"ArXiv Preprint"},{"key":"ref25","first-page":"1225","article-title":"Train faster, generalize better: Stability of stochastic gradient descent","author":"hardt","year":"0","journal-title":"Int Conference on Machine Learning"}],"event":{"name":"2021 IEEE International Symposium on Information Theory (ISIT)","location":"Melbourne, Australia","start":{"date-parts":[[2021,7,12]]},"end":{"date-parts":[[2021,7,20]]}},"container-title":["2021 IEEE International Symposium on Information Theory (ISIT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9517708\/9517709\/09517811.pdf?arnumber=9517811","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,12]],"date-time":"2026-03-12T20:34:15Z","timestamp":1773347655000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9517811\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,12]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/isit45174.2021.9517811","relation":{},"subject":[],"published":{"date-parts":[[2021,7,12]]}}}