{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T16:29:15Z","timestamp":1781368155840,"version":"3.54.1"},"reference-count":52,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2020,9,1]],"date-time":"2020-09-01T00:00:00Z","timestamp":1598918400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,9,1]],"date-time":"2020-09-01T00:00:00Z","timestamp":1598918400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,9,1]],"date-time":"2020-09-01T00:00:00Z","timestamp":1598918400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Signal Process. Mag."],"published-print":{"date-parts":[[2020,9]]},"DOI":"10.1109\/msp.2020.3004124","type":"journal-article","created":{"date-parts":[[2020,9,10]],"date-time":"2020-09-10T20:56:34Z","timestamp":1599771394000},"page":"95-108","source":"Crossref","is-referenced-by-count":54,"title":["The Global Landscape of Neural Networks: An Overview"],"prefix":"10.1109","volume":"37","author":[{"given":"Ruoyu","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0374-3101","authenticated-orcid":false,"given":"Dawei","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiyu","family":"Liang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tian","family":"Ding","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rayadurgam","family":"Srikant","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","article-title":"On the loss landscape of a class of deep neural networks with no bad local valleys","author":"nguyen","year":"0","journal-title":"Proc Int Conf Learning Representations"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.5555\/3305890.3305950"},{"key":"ref33","first-page":"2835","article-title":"Understanding the loss surface of neural networks for binary classification","author":"liang","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref32","first-page":"4355","article-title":"Adding one neuron can eliminate all bad local minima","author":"liang","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref31","first-page":"6391","article-title":"Learning overparameterized neural networks via stochastic gradient descent on structured data","author":"li","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref30","first-page":"6391","article-title":"Visualizing the loss landscape of neural nets","author":"li","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref37","first-page":"4790","article-title":"On connected sublevel sets in deep learning","author":"nguyen","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/s10208-015-9296-2"},{"key":"ref35","article-title":"Depth creates no bad local minima","author":"lu","year":"2017"},{"key":"ref34","article-title":"Revisiting landscape analysis in deep neural networks: Eliminating decreasing paths to infinity","author":"liang","year":"2019"},{"key":"ref28","first-page":"1246","article-title":"Gradient descent only converges to minimizers","author":"lee","year":"0","journal-title":"Proc Conf Learning Theory"},{"key":"ref27","first-page":"14,574","article-title":"Explaining landscape connectivity of low-cost solutions for multilayer nets","author":"kuditipudi","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref29","article-title":"On the benefit of width for neural networks: Disappearance of bad basins","author":"li","year":"2018"},{"key":"ref2","first-page":"6673","article-title":"On the convergence rate of training recurrent neural networks","author":"allen-zhu","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref1","first-page":"242","article-title":"A convergence theory for deep learning via over-parameterization","author":"allen-zhu","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.467"},{"key":"ref22","first-page":"8571","article-title":"Neural tangent kernel: Convergence and generalization in neural networks","author":"jacot","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref21","article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and Huffman coding","author":"han","year":"2015"},{"key":"ref24","first-page":"586","article-title":"Deep learning without poor local minima","author":"kawaguchi","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref23","first-page":"1724","article-title":"How to escape saddle points efficiently","author":"jin","year":"0","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref26","first-page":"2698","article-title":"An alternative view: When does SGD escape local minima?","author":"kleinberg","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref25","article-title":"Elimination of all bad local minima in deep learning","author":"kawaguchi","year":"2019"},{"key":"ref50","article-title":"On large-batch training for deep learning: Generalization gap and sharp minima","author":"keskar","year":"2016"},{"key":"ref51","article-title":"Entropic gradient descent algorithms and wide flat minima","author":"pittorino","year":"2020"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1908636117"},{"key":"ref10","first-page":"2933","article-title":"On lazy training in differentiable programming","author":"chizat","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref11","first-page":"192","article-title":"The loss surfaces of multilayer networks","author":"choromanska","year":"0","journal-title":"Proc Artificial Intelligence and Statistics"},{"key":"ref40","first-page":"4433","article-title":"Spurious local minima are common in two-layer relu neural networks","author":"safran","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.5555\/2969033.2969154"},{"key":"ref13","article-title":"Sub-optimal local minima exist for almost all over-parameterized neural networks","author":"ding","year":"2019"},{"key":"ref14","first-page":"1309","article-title":"Essentially no barriers in neural network energy landscape","author":"draxler","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref15","first-page":"1675","article-title":"Gradient descent finds global minima of deep neural networks","author":"du","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref16","first-page":"1339","article-title":"Gradient descent learns one-hidden-layer CNN: Don&#x2019;t be afraid of spurious local minima","author":"du","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref17","article-title":"Topology and geometry of half-rectified network optimization","author":"freeman","year":"0","journal-title":"Proc Int Conf Learning Representations"},{"key":"ref18","first-page":"8789","article-title":"Loss surfaces, mode connectivity, and fast ensembling of DNNs","author":"garipov","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref19","article-title":"Qualitatively characterizing neural network optimization problems","author":"goodfellow","year":"2014"},{"key":"ref4","first-page":"322","article-title":"Fine-grained analysis of optimization and generalization for overparameterized two-layer neural networks","author":"arora","year":"0","journal-title":"Proc Int Conf Machine Learning"},{"key":"ref3","first-page":"81","article-title":"Efficient approaches for escaping higher order saddle points in non-convex optimization","author":"anandkumar","year":"0","journal-title":"Proc Conf Learning Theory"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90014-2"},{"key":"ref5","first-page":"8139","article-title":"On exact computation with an infinitely wide neural net","author":"arora","year":"0","journal-title":"Proc Advances Neural Information Processing Systems"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-002-0352-8"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/0925-2312(95)00032-1"},{"key":"ref49","first-page":"2","article-title":"Algorithmic regularization in over-parameterized matrix sensing and neural networks with quadratic activations","author":"li","year":"0","journal-title":"Proc 31st Conf on Learning Theory 75"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2019.2937282"},{"key":"ref46","article-title":"Weighted AdaGrad with unified momentum","author":"zou","year":"2018"},{"key":"ref45","article-title":"Critical points of neural networks: Analytical forms and landscape properties","author":"zhou","year":"0","journal-title":"Proc Int Conf Learning Representations"},{"key":"ref48","article-title":"Towards understanding the role of over-parametrization in generalization of neural networks","author":"neyshabur","year":"2019"},{"key":"ref47","first-page":"157","article-title":"Perceptrons","author":"minsky","year":"1988","journal-title":"Neurocomputing Foundations of Research"},{"key":"ref42","article-title":"Neural networks with finite intrinsic dimension have no spurious valleys","author":"venturi","year":"2018"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/s40305-020-00309-6"},{"key":"ref44","article-title":"Global optimality conditions for deep neural networks","author":"yun","year":"0","journal-title":"Proc Int Conf Learning Representations"},{"key":"ref43","doi-asserted-by":"crossref","first-page":"1300","DOI":"10.1109\/72.410380","article-title":"On the local minima free condition of backpropagation learning","volume":"6","author":"yu","year":"1995","journal-title":"IEEE Trans Neural Netw"}],"container-title":["IEEE Signal Processing Magazine"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/79\/9186128\/09194023.pdf?arnumber=9194023","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T15:35:53Z","timestamp":1651073753000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9194023\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,9]]},"references-count":52,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/msp.2020.3004124","relation":{},"ISSN":["1053-5888","1558-0792"],"issn-type":[{"value":"1053-5888","type":"print"},{"value":"1558-0792","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,9]]}}}