{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:44:48Z","timestamp":1776883488635,"version":"3.51.2"},"reference-count":65,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"NSF","award":["DMS-2015517"],"award-info":[{"award-number":["DMS-2015517"]}]},{"name":"CDS Moore-Sloan Postdoctoral Fellowship"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Signal Process."],"published-print":{"date-parts":[[2022]]},"DOI":"10.1109\/tsp.2022.3156702","type":"journal-article","created":{"date-parts":[[2022,3,7]],"date-time":"2022-03-07T20:44:42Z","timestamp":1646685882000},"page":"1310-1319","source":"Crossref","is-referenced-by-count":2,"title":["Self-Regularity of Non-Negative Output Weights for Overparameterized Two-Layer Neural Networks"],"prefix":"10.1109","volume":"70","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8898-8778","authenticated-orcid":false,"given":"David","family":"Gamarnik","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0411-7161","authenticated-orcid":false,"given":"Eren C.","family":"Kzldag","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ilias","family":"Zadik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT45174.2021.9517811"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref3","first-page":"1097","article-title":"ImageNet classification with deep convolutional neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Krizhevsky","year":"2012"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2109382"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390177"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1038\/nature24270"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3446776"},{"key":"ref8","article-title":"Gradient descent provably optimizes over-parameterized neural networks","volume-title":"Proc. 7th Int. Conf. Learn. Representations","author":"Du","year":"2019"},{"key":"ref9","first-page":"8157","article-title":"Learning overparameterized neural networks via stochastic gradient descent on structured data","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Li","year":"2018"},{"key":"ref10","first-page":"9461","article-title":"Implicit bias of gradient descent on linear convolutional networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Gunasekar","year":"2018"},{"key":"ref11","first-page":"6979","article-title":"Dynamics of stochastic gradient descent for two-layer neural networks in the teacher-student setup","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Goldt","year":"2019"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1903070116"},{"key":"ref13","first-page":"8139","article-title":"On exact computation with an infinitely wide neural net","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Arora","year":"2019"},{"key":"ref14","first-page":"1064","article-title":"Nearly-tight VC-dimension bounds for piecewise linear neural networks","volume-title":"Proc. Conf. Learn. Theory","author":"Harvey","year":"2017"},{"issue":"63","key":"ref15","first-page":"1","article-title":"Nearly-tight VC-dimension and pseudodimension bounds for piecewise linear neural networks","volume":"20","author":"Bartlett","year":"2019","journal-title":"J. Mach. Learn. Res."},{"key":"ref16","first-page":"1376","article-title":"Norm-based capacity control in neural networks","volume-title":"Proc. Conf. Learn. Theory","author":"Neyshabur","year":"2015"},{"key":"ref17","first-page":"6240","article-title":"Spectrally-normalized margin bounds for neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Bartlett","year":"2017"},{"key":"ref18","first-page":"888","article-title":"Fisher-Rao metric, geometry, and complexity of neural networks","volume-title":"Proc. 22nd Int. Conf. Artif. Intell. Statist.","author":"Liang","year":"2019"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1093\/imaiai\/iaz007"},{"key":"ref20","article-title":"Computing nonvacuous generalization bounds for deep (stochastic) neural networks with many more parameters than training data","volume-title":"Proc. 33rd Conf. Uncertainty Artif. Intell.","author":"Dziugaite","year":"2017"},{"key":"ref21","article-title":"A pac-Bayesian approach to spectrally-normalized margin bounds for neural networks","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Neyshabur","year":"2018"},{"key":"ref22","first-page":"5947","article-title":"Exploring generalization in deep learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Neyshabur","year":"2017"},{"key":"ref23","first-page":"254","article-title":"Stronger generalization bounds for deep nets via a compression approach","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Arora","year":"2018"},{"key":"ref24","first-page":"322","article-title":"Fine-grained analysis of optimization and generalization for overparameterized two-layer neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Arora","year":"2019"},{"key":"ref25","first-page":"605","article-title":"Globally optimal gradient descent for a convnet with Gaussian inputs","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","volume":"70","author":"Brutzkus","year":"2017"},{"key":"ref26","first-page":"14797","article-title":"Algorithm-dependent generalization bounds for overparameterized deep residual networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Frei","year":"2019"},{"key":"ref27","first-page":"1225","article-title":"Train faster, generalize better: Stability of stochastic gradient descent","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Hardt","year":"2016"},{"key":"ref28","article-title":"SGD learns over-parameterized networks that provably generalize on linearly separable data","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Brutzkus","year":"2018"},{"key":"ref29","first-page":"10836","article-title":"Generalization bounds of stochastic gradient descent for wide and deep neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Cao","year":"2019"},{"key":"ref30","first-page":"605","article-title":"Generalization bounds of SGLD for non-convex learning: Two theoretical viewpoints","volume-title":"Proc. Conf. Learn. Theory","author":"Mou","year":"2018"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/18.661502"},{"key":"ref32","article-title":"Neural networks and polynomial regression. demystifying the overparametrization phenomena","author":"Emschwiller","year":"2020"},{"key":"ref33","first-page":"855","article-title":"On the computational efficiency of training neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Livni","year":"2014"},{"key":"ref34","first-page":"1417","article-title":"Towards provable learning of polynomial neural networks using low-rank matrix estimation","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Soltani","year":"2018"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2019.2916743"},{"key":"ref36","article-title":"Provable algorithms for nonlinear models in machine learning and signal processing","author":"Soltani","year":"2021"},{"key":"ref37","first-page":"133","article-title":"Spurious valleys in one-hidden-layer neural network optimization landscapes","volume":"20","author":"Venturi","year":"2019","journal-title":"J. Mach. Learn. Res."},{"key":"ref38","article-title":"Learning one-hidden-layer neural networks with landscape design","volume-title":"Proc. 6th Int. Conf. Learn. Representations","author":"Ge","year":"2018"},{"key":"ref39","first-page":"1514","article-title":"Algorithms and SQ lower bounds for PAC learning one-hidden-layer Relu networks","volume-title":"Proc. Conf. Learn. Theory","author":"Diakonikolas","year":"2020"},{"key":"ref40","first-page":"2613","article-title":"Learning over-parametrized two-layer neural networks beyond NTK","volume-title":"Proc. Conf. Learn. Theory","author":"Li","year":"2020"},{"key":"ref41","first-page":"1329","article-title":"On the power of over-parametrization in neural networks with quadratic activation","volume-title":"Proc. Int. Conf. Mach. Learn","author":"Du","year":"2018"},{"key":"ref42","first-page":"4433","article-title":"Spurious local minima are common in two-layer Relu neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Safran","year":"2018"},{"key":"ref43","first-page":"1524","article-title":"Learning one-hidden-layer Relu networks via gradient descent","volume-title":"Proc. 22nd Int. Conf. Artif. Intell. Statist.","author":"Zhang","year":"2019"},{"key":"ref44","first-page":"1783","article-title":"Learning one convolutional layer with overlapping patches","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Goel","year":"2018"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952123"},{"key":"ref46","article-title":"Non-negative matrix factorization","year":"2021"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1137\/130913869"},{"key":"ref48","article-title":"Towards an understanding of neural networks: Mean-field incursions","author":"Gabri","year":"2019"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1038\/44565"},{"key":"ref50","article-title":"Algorithms for non-negative matrix factorization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"13","author":"Lee","year":"2001"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/NNSP.2002.1030067"},{"key":"ref52","article-title":"Mildly overparametrized neural nets can memorize training data efficiently","author":"Ge","year":"2019"},{"key":"ref53","first-page":"1675","article-title":"Gradient descent finds global minima of deep neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Du","year":"2019"},{"key":"ref54","first-page":"1470","article-title":"Learning neural networks with two nonlinear layers in polynomial time","volume-title":"Proc. Conf. Learn. Theory","author":"Goel","year":"2019"},{"key":"ref55","article-title":"Why are convolutional nets more sample-efficient than fully-connected nets","volume-title":"Proc. 9th Int. Conf. Learn. Representations","author":"Li","year":"2021"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-8176-4948-7"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1017\/cbo9780511794308.006"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1017\/9781108231596"},{"key":"ref59","first-page":"315","article-title":"Deep sparse rectifier neural networks","volume-title":"Proc. 14th Int. Conf. Artif. Intell. Statist.","author":"Glorot","year":"2011"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ISIT45174.2021.9517811"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1016\/S0022-0000(05)80062-5"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1006\/jcss.1996.0033"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1016\/0890-5401(92)90010-D"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1145\/263867.263927"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780199535255.001.0001"}],"container-title":["IEEE Transactions on Signal Processing"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/78\/9675017\/9729548-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/78\/9675017\/09729548.pdf?arnumber=9729548","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,17]],"date-time":"2024-01-17T23:31:33Z","timestamp":1705534293000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9729548\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"references-count":65,"URL":"https:\/\/doi.org\/10.1109\/tsp.2022.3156702","relation":{},"ISSN":["1053-587X","1941-0476"],"issn-type":[{"value":"1053-587X","type":"print"},{"value":"1941-0476","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]}}}