{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,5]],"date-time":"2025-10-05T04:15:59Z","timestamp":1759637759754,"version":"3.40.5"},"reference-count":64,"publisher":"Society for Industrial & Applied Mathematics (SIAM)","issue":"4","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["SIAM J. Optim."],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1137\/19m1299074","type":"journal-article","created":{"date-parts":[[2022,11,17]],"date-time":"2022-11-17T16:21:40Z","timestamp":1668702100000},"page":"2797-2827","source":"Crossref","is-referenced-by-count":3,"title":["Revisiting Landscape Analysis in Deep Neural Networks: Eliminating Decreasing Paths to Infinity"],"prefix":"10.1137","volume":"32","author":[{"given":"Shiyu","family":"Liang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2487-5322","authenticated-orcid":true,"given":"Ruoyu","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"R.","family":"Srikant","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"351","published-online":{"date-parts":[[2022,11,17]]},"reference":[{"volume-title":"A Convergence Theory for Deep Learning via Over-parameterization, preprint, https:\/\/arxiv.org\/abs\/1811.03962","year":"2018","author":"Allen-Zhu Z.","key":"atypb1"},{"volume-title":"On the Optimization of Deep Networks: Implicit Acceleration by Overparameterization, preprint, https:\/\/arxiv.org\/abs\/1802.06509","year":"2018","author":"Arora S.","key":"atypb2"},{"volume-title":"On Exact Computation with an Infinitely Wide Neural Net, preprint, https:\/\/arxiv.org\/abs\/1904.11955","year":"2019","author":"Arora S.","key":"atypb3"},{"key":"atypb4","first-page":"6240","author":"Bartlett P. L.","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"atypb5","volume-title":"Nonlinear Programming","author":"Bertsekas D. P.","year":"1999","edition":"2"},{"key":"atypb6","first-page":"3873","author":"Bhojanapalli S.","year":"2016","journal-title":"Advances in Neural Information Processing Systems"},{"key":"atypb7","first-page":"605","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Brutzkus A.","year":"2017"},{"volume-title":"International Conference on Learning Representations (ICLR)","year":"2018","author":"Brutzkus A.","key":"atypb8"},{"key":"atypb9","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2015.2399924"},{"volume-title":"Gradient Descent with Random Initialization: Fast Global Convergence for Nonconvex Phase Retrieval, preprint, https:\/\/arxiv.org\/abs\/1803.07726","year":"2018","author":"Chen Y.","key":"atypb10"},{"key":"atypb11","first-page":"3040","author":"Chizat L.","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"key":"atypb12","first-page":"192","author":"Choromanska A.","year":"2015","journal-title":"Artificial Intelligence and Statistics"},{"volume-title":"Optimization Online","year":"2019","author":"Ding T.","key":"atypb13"},{"volume-title":"On the Power of Over-parametrization in Neural Networks with Quadratic Activation, preprint, https:\/\/arxiv.org\/abs\/1803.01206","year":"2018","author":"Du S. S.","key":"atypb14"},{"volume-title":"Gradient Descent Finds Global Minima of Deep Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1811.03804","year":"2018","author":"Du S. S.","key":"atypb15"},{"volume-title":"Porcupine Neural Networks: (Almost) All Local Optima Are Global, preprint, https:\/\/arxiv.org\/abs\/1710.02196","year":"2017","author":"Feizi S.","key":"atypb16"},{"volume-title":"Topology and Geometry of Half-Rectified Network Optimization, preprint, https:\/\/arxiv.org\/abs\/1611.01540","year":"2016","author":"Freeman C. D.","key":"atypb17"},{"volume-title":"Learning One-Hidden-Layer Neural Networks under General Input Distributions, preprint, https:\/\/arxiv.org\/abs\/1810.04133","year":"2018","author":"Gao W.","key":"atypb18"},{"key":"atypb19","first-page":"797","volume-title":"Proceedings of the Conference on Learning Theory","author":"Ge R.","year":"2015"},{"key":"atypb20","first-page":"1233","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Ge R.","year":"2017"},{"key":"atypb21","first-page":"2973","author":"Ge R.","year":"2016","journal-title":"Advances in Neural Information Processing Systems"},{"volume-title":"Learning One-Hidden-Layer Neural Networks with Landscape Design, preprint, https:\/\/arxiv.org\/abs\/1711.00501","year":"2017","author":"Ge R.","key":"atypb22"},{"key":"atypb23","first-page":"249","volume-title":"Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics","author":"Glorot X.","year":"2010"},{"key":"atypb24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.467"},{"key":"atypb25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"atypb26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"atypb27","first-page":"8571","author":"Jacot A.","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"volume-title":"Beating the Perils of Non-convexity: Guaranteed Training of Neural Networks Using Tensor Methods, preprint, https:\/\/arxiv.org\/abs\/1506.08473","year":"2015","author":"Janzamin M.","key":"atypb28"},{"volume-title":"Classics in Math. 132","year":"2013","author":"Kato T.","key":"atypb29"},{"key":"atypb30","first-page":"586","author":"Kawaguchi K.","year":"2016","journal-title":"Advances in Neural Information Processing Systems"},{"volume-title":"Elimination of All Bad Local Minima in Deep Learning, preprint, https:\/\/arxiv.org\/abs\/1901.00279","year":"2019","author":"Kawaguchi K.","key":"atypb31"},{"key":"atypb32","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2010.2046205"},{"key":"atypb33","first-page":"689","author":"Krizhevsky A.","year":"2012","journal-title":"Advances in Neural Information Processing Systems"},{"volume-title":"The Multilinear Structure of ReLU Networks, preprint, https:\/\/arxiv.org\/abs\/1712.10132","year":"2017","author":"Laurent T.","key":"atypb34"},{"volume-title":"Gradient Descent Converges to Minimizers, preprint, https:\/\/arxiv.org\/abs\/1602.04915","year":"2016","author":"Lee J. D.","key":"atypb35"},{"volume-title":"Over-Parameterized Deep Neural Networks Have No Strict Local Minima for Any Continuous Activations, preprint, https:\/\/arxiv.org\/abs\/1812.11039v1","year":"2018","author":"Li D.","key":"atypb36"},{"volume-title":"Algorithmic Regularization in Over-Parameterized Matrix Sensing and Neural Networks with Quadratic Activations, preprint, https:\/\/arxiv.org\/abs\/1712.09203","year":"2017","author":"Li Y.","key":"atypb37"},{"key":"atypb38","first-page":"597","author":"Li Y.","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"atypb39","first-page":"4355","author":"Liang S.","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"volume-title":"Understanding the Loss Surface of Neural Networks for Binary Classification, preprint, https:\/\/arxiv.org\/abs\/1803.00909","year":"2018","author":"Liang S.","key":"atypb40"},{"volume-title":"A Mean Field View of the Landscape of Two-Layers Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1804.06561","year":"2018","author":"Mei S.","key":"atypb41"},{"volume-title":"On the Connection between Learning Two-Layers Neural Networks and Tensor Decomposition, preprint, https:\/\/arxiv.org\/abs\/1802.07301","year":"2018","author":"Mondelli M.","key":"atypb42"},{"key":"atypb43","first-page":"5947","author":"Neyshabur B.","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"volume-title":"Towards Understanding the Role of Over-Parametrization in Generalization of Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1805.12076","year":"2018","author":"Neyshabur B.","key":"atypb44"},{"key":"atypb45","first-page":"2603","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Nguyen Q.","year":"2017"},{"volume-title":"On the Loss Landscape of a Class of Deep Neural Networks with No Bad Local Valleys, preprint, https:\/\/arxiv.org\/abs\/1809.10749","year":"2018","author":"Nguyen Q.","key":"atypb46"},{"volume-title":"Towards Moderate Overparameterization: Global Convergence Guarantees for Training Shallow Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1902.04674","year":"2019","author":"Oymak S.","key":"atypb47"},{"volume-title":"Convergence Results for Neural Networks via Electrodynamics, preprint, https:\/\/arxiv.org\/abs\/1702.00458v3","year":"2017","author":"Panigrahy R.","key":"atypb48"},{"volume-title":"Neural Networks as Interacting Particle Systems: Asymptotic Convexity of the Loss Landscape and Universal Scaling of the Approximation Error, preprint, https:\/\/arxiv.org\/abs\/1805.00915v1","year":"2018","author":"Rotskoff G. M.","key":"atypb49"},{"volume-title":"2nd rev. ed.","year":"1937","author":"Saks S.","key":"atypb50"},{"volume-title":"Mean Field Analysis of Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1805.01053v1","year":"2018","author":"Sirignano J.","key":"atypb51"},{"key":"atypb52","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2018.2854560"},{"volume-title":"Exponentially Vanishing Sub-optimal Local Minima in Multilayer Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1702.05777","year":"2017","author":"Soudry D.","key":"atypb53"},{"key":"atypb54","doi-asserted-by":"publisher","DOI":"10.1007\/s10208-017-9365-9"},{"volume-title":"Guaranteed Matrix Completion via Non-convex Factorization, preprint, https:\/\/arxiv.org\/abs\/1411.8003","year":"2014","author":"Sun R.","key":"atypb55"},{"key":"atypb56","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2014.2368984"},{"key":"atypb57","first-page":"3404","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Tian Y.","year":"2017"},{"key":"atypb58","doi-asserted-by":"publisher","DOI":"10.1023\/A:1017501703105"},{"volume-title":"Learning ReLU Networks on Linearly Separable Data: Algorithm, Optimality, and Generalization, preprint, https:\/\/arxiv.org\/abs\/1808.04685","year":"2018","author":"Wang G.","key":"atypb59"},{"volume-title":"On the Margin Theory of Feedforward Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1810.05369v1","year":"2018","author":"Wei C.","key":"atypb60"},{"key":"atypb61","first-page":"4140","volume-title":"Proceedings of the 34th International Conference on Machine Learning","volume":"70","author":"Zhong K.","year":"2017"},{"volume-title":"On the Convergence of Adaptive Gradient Methods for Nonconvex Optimization, preprint, https:\/\/arxiv.org\/abs\/1808.05671v1","year":"2018","author":"Zhou D.","key":"atypb62"},{"volume-title":"Stochastic Gradient Descent Optimizes Over-parameterized Deep ReLU Networks, preprint, https:\/\/arxiv.org\/abs\/1811.08888","year":"2018","author":"Zou D.","key":"atypb63"},{"volume-title":"On the Convergence of AdaGrad with Momentum for Training Deep Neural Networks, preprint, https:\/\/arxiv.org\/abs\/1808.03408v1","year":"2018","author":"Zou F.","key":"atypb64"}],"container-title":["SIAM Journal on Optimization"],"original-title":[],"language":"en","deposited":{"date-parts":[[2022,12,22]],"date-time":"2022-12-22T16:17:58Z","timestamp":1671725878000},"score":1,"resource":{"primary":{"URL":"https:\/\/epubs.siam.org\/doi\/10.1137\/19M1299074"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,17]]},"references-count":64,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["10.1137\/19M1299074"],"URL":"https:\/\/doi.org\/10.1137\/19m1299074","relation":{},"ISSN":["1052-6234","1095-7189"],"issn-type":[{"type":"print","value":"1052-6234"},{"type":"electronic","value":"1095-7189"}],"subject":[],"published":{"date-parts":[[2022,11,17]]}}}