{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T17:09:16Z","timestamp":1780765756260,"version":"3.54.1"},"reference-count":87,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61806013"],"award-info":[{"award-number":["61806013"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1109\/tnnls.2022.3204319","type":"journal-article","created":{"date-parts":[[2022,9,20]],"date-time":"2022-09-20T19:27:57Z","timestamp":1663702077000},"page":"5382-5394","source":"Crossref","is-referenced-by-count":8,"title":["Spurious Local Minima are Common for Deep Neural Networks With Piecewise Linear Activations"],"prefix":"10.1109","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3482-6930","authenticated-orcid":false,"given":"Bo","family":"Liu","sequence":"first","affiliation":[{"name":"Faculty of Information Technology, College of Computer Science, Beijing University of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1097","article-title":"ImageNet classification with deep convolutional neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Krizhevsky"},{"key":"ref2","first-page":"1","article-title":"Very deep convolutional networks for large-scale image recognition","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Simonyan"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2572683"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2020.09.003"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2021.02.005"},{"key":"ref7","article-title":"Recent advances in deep learning theory","author":"He","year":"2020","journal-title":"arXiv:2012.10931"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2020.3004124"},{"key":"ref9","first-page":"4433","article-title":"Spurious local minima are common in two-layer ReLU neural networks","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Safran"},{"key":"ref10","article-title":"Local minima in training of neural networks","author":"Swirszcz","year":"2016","journal-title":"arXiv:1611.06310"},{"key":"ref11","first-page":"1","article-title":"Small nonlinearities in activation functions create bad local minima in neural networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Yun"},{"key":"ref12","first-page":"1","article-title":"Critical points of neural networks: Analytical forms and landscape properties","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Zhou"},{"key":"ref13","first-page":"1","article-title":"Bounds on over-parameterization for guaranteed existence of descent paths in shallow ReLU networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Sharifnassab"},{"key":"ref14","first-page":"1","article-title":"Piecewise linear activations substantially shape the loss surfaces of neural networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"He"},{"key":"ref15","article-title":"Spurious local minima exist for almost all over-parameterized neural networks","author":"Ding","year":"2019","journal-title":"Optim. Online"},{"key":"ref16","first-page":"1","article-title":"Truth or backpropaganda? An empirical investigation of deep learning theory","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Goldblum"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2021.08.005"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90014-2"},{"key":"ref19","first-page":"1","article-title":"Deep learning without poor local minima","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kawaguchi"},{"key":"ref20","article-title":"Depth creates no bad local minima","author":"Lu","year":"2017","journal-title":"arXiv:1702.08580"},{"key":"ref21","first-page":"2902","article-title":"Deep linear networks with arbitrary loss: All local minima are global","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Laurent"},{"key":"ref22","first-page":"1","article-title":"Global optimality conditions for deep neural networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Yun"},{"key":"ref23","article-title":"Learning deep models: Critical points and local openness","author":"Nouiehed","year":"2018","journal-title":"arXiv:1803.02968"},{"key":"ref24","article-title":"Depth creates no more spurious local minima","author":"Zhang","year":"2019","journal-title":"arXiv:1901.09827"},{"key":"ref25","first-page":"1","article-title":"Matrix completion has no spurious local minimum","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ge"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2018.2854560"},{"key":"ref27","first-page":"1329","article-title":"On the power of over-parametrization in neural networks with quadratic activation","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Du"},{"key":"ref28","first-page":"1","article-title":"Identity matters in deep learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Hardt"},{"key":"ref29","article-title":"Avoiding spurious local minima in deep quadratic networks","author":"Kazemipour","year":"2020","journal-title":"arXiv:2001.00098"},{"key":"ref30","article-title":"No bad local minima: Data independent training error guarantees for multilayer neural networks","author":"Soudry","year":"2016","journal-title":"arXiv:1605.08361"},{"key":"ref31","first-page":"2908","article-title":"The multilinear structure of ReLU networks","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Laurent"},{"key":"ref32","first-page":"774","article-title":"On the quality of the initial basin in overspecified neural networks","volume-title":"Proc. 33rd Int. Conf. Mach. Learn.","author":"Safran"},{"key":"ref33","article-title":"Exponentially vanishing sub-optimal local minima in multilayer neural networks","author":"Soudry","year":"2017","journal-title":"arXiv:1702.05777"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.106923"},{"key":"ref35","article-title":"Spurious valleys in two-layer neural network optimization landscapes","author":"Venturi","year":"2018","journal-title":"arXiv:1802.06384"},{"key":"ref36","first-page":"4790","article-title":"On connected sublevel sets in deep learning","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Nguyen"},{"key":"ref37","article-title":"On the benefit of width for neural networks: Disappearance of bad basins","author":"Li","year":"2018","journal-title":"arXiv:1812.11039"},{"key":"ref38","article-title":"On the loss landscape of a class of deep neural networks with no bad local valleys","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Nguyen"},{"key":"ref39","first-page":"1","article-title":"Understanding the loss surface of neural networks for binary classification","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Liang"},{"key":"ref40","first-page":"1","article-title":"Adding one neuron can eliminate all bad local minima","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liang"},{"key":"ref41","article-title":"Elimination of all bad local minima in deep learning","author":"Kawaguchi","year":"2019","journal-title":"arXiv:1901.00279"},{"key":"ref42","article-title":"Learning one-hidden-layer neural networks with landscape design","author":"Ge","year":"2017","journal-title":"arXiv:1711.00501"},{"key":"ref43","article-title":"Learning one-hidden-layer neural networks under general input distributions","author":"Gao","year":"2018","journal-title":"arXiv:1810.04133"},{"key":"ref44","article-title":"Porcupine neural networks: (Almost) all local optima are global","author":"Feizi","year":"2017","journal-title":"arXiv:1710.02196"},{"key":"ref45","first-page":"1","article-title":"Deforming the loss surface to affect the behaviour of the optimizer","volume-title":"Proc. AAAI Conf. Artif. Intell.","author":"Chen"},{"key":"ref46","first-page":"1","article-title":"Are ResNets provably better than linear predictors?","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shamir"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2019.06.009"},{"key":"ref48","first-page":"1","article-title":"Piecewise strong convexity of neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Milne"},{"key":"ref49","first-page":"192","article-title":"The loss surfaces of multilayer networks","volume-title":"Proc. 18th Int. Conf. Artif. Intell. Statist.","author":"Choromanska"},{"key":"ref50","first-page":"1","article-title":"The spectrum of the Fisher information matrix of a single-hidden-layer neural network","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Pennington"},{"key":"ref51","first-page":"2798","article-title":"Geometry of neural network loss surfaces via random matrix theory","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Pennington"},{"key":"ref52","article-title":"The landscape of empirical risk for non-convex losses","author":"Mei","year":"2016","journal-title":"arXiv:1607.06534"},{"key":"ref53","first-page":"1","article-title":"Empirical risk landscape analysis for understanding deep neural networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Zhou"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2013.2293637"},{"key":"ref55","first-page":"1","article-title":"Identifying and attacking the saddle point problem in high-dimensional non-convex optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dauphin"},{"key":"ref56","first-page":"1","article-title":"Topology and geometry of half-rectified network optimization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Freeman"},{"key":"ref57","first-page":"1","article-title":"Qualitatively characterizing neural network optimization problems","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Goodfellow"},{"key":"ref58","article-title":"Theory II: Landscape of the empirical risk in deep learning","author":"Liao","year":"2017","journal-title":"arXiv:1703.09833"},{"key":"ref59","first-page":"1","article-title":"Visualizing the loss landscape of neural nets","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref60","first-page":"1","article-title":"Large scale structure of neural network loss landscapes","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Fort"},{"key":"ref61","first-page":"1","article-title":"Ringing ReLUs: Harmonic distortion analysis of nonlinear feedforward networks","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Mehmeti-Gopel"},{"key":"ref62","first-page":"1309","article-title":"Essentially no barriers in neural network energy landscape","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Draxler"},{"key":"ref63","first-page":"1","article-title":"Loss surfaces, mode connectivity, and fast ensembling of DNNs","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Garipov"},{"key":"ref64","first-page":"335","article-title":"Low-loss connection of weight vectors: Distribution-based approaches","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Anokhin"},{"key":"ref65","first-page":"3404","article-title":"An analytical formula of population gradient for two-layered ReLU network and its applications in convergence and critical point analysis","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Tian"},{"key":"ref66","first-page":"4140","article-title":"Recovery guarantees for one-hidden-layer neural networks","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Zhong"},{"key":"ref67","first-page":"1","article-title":"Convergence analysis of two-layer neural networks with ReLU activation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref68","first-page":"1","article-title":"SGD converges to global minimum in deep learning via star-convex path","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Zhou"},{"key":"ref69","article-title":"Learning one-hidden-layer ReLU networks via gradient descent","author":"Zhang","year":"2018","journal-title":"arXiv:1806.07808"},{"key":"ref70","first-page":"1675","article-title":"Gradient descent finds global minima of deep neural networks","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Du"},{"key":"ref71","first-page":"1","article-title":"A convergence theory for deep learning via over-parameterization","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Allen-Zhu"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-019-05839-6"},{"key":"ref73","first-page":"8056","article-title":"On the proof of global convergence of gradient descent for deep ReLU networks with linear widths","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Nguyen"},{"key":"ref74","first-page":"1","article-title":"Neural tangent kernel: Convergence and generalization in neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jacot"},{"key":"ref75","first-page":"1","article-title":"On the number of linear regions of deep neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Montufar"},{"key":"ref76","first-page":"4558","article-title":"Bounding and counting linear regions of deep neural networks","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Serra"},{"key":"ref77","first-page":"2596","article-title":"Complexity of linear regions in deep networks","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Hanin"},{"key":"ref78","article-title":"Deep ReLU networks have surprisingly few activation patterns","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Hanin"},{"key":"ref79","first-page":"10514","article-title":"On the number of linear regions of convolutional neural networks","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Xiong"},{"key":"ref80","first-page":"1","article-title":"Rectifier nonlinearities improve neural network acoustic models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Maas"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1016\/S0895-7177(99)00195-8"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2016.08.006"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/TCSII.2015.2436131"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1016\/j.amc.2019.03.026"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2015.01.007"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2015.07.009"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10492491\/09896818.pdf?arnumber=9896818","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,16]],"date-time":"2025-04-16T17:51:40Z","timestamp":1744825900000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9896818\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4]]},"references-count":87,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2022.3204319","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4]]}}}