{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T11:14:31Z","timestamp":1730200471717,"version":"3.28.0"},"reference-count":36,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,12,10]],"date-time":"2020-12-10T00:00:00Z","timestamp":1607558400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,12,10]],"date-time":"2020-12-10T00:00:00Z","timestamp":1607558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,12,10]],"date-time":"2020-12-10T00:00:00Z","timestamp":1607558400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,12,10]]},"DOI":"10.1109\/bigdata50022.2020.9378359","type":"proceedings-article","created":{"date-parts":[[2021,3,19]],"date-time":"2021-03-19T17:10:21Z","timestamp":1616173821000},"page":"141-146","source":"Crossref","is-referenced-by-count":0,"title":["Neural Network Training Techniques Regularize Optimization Trajectory: An Empirical Study"],"prefix":"10.1109","author":[{"given":"Cheng","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junjie","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"article-title":"Characterization of gradient dominance and regularity conditions for neural networks","year":"2017","author":"zhou","key":"ref33"},{"key":"ref32","first-page":"4140","article-title":"Recovery guarantees for one-hidden-layer neural networks","volume":"70","author":"zhong","year":"2017","journal-title":"Proc of the International Conference on Machine Learning (ICML)"},{"key":"ref31","first-page":"1524","article-title":"Learning one-hidden-layer relu networks via gradient descent","volume":"89","author":"zhang","year":"2019","journal-title":"Proc International Conference on Artificial Intelligence and Statistics (AISTATS)"},{"key":"ref30","first-page":"1","article-title":"A nonconvex approach for phase retrieval: reshaped Wirtinger flow and incremental algorithms","volume":"18","author":"zhang","year":"2017","journal-title":"Journal of Machine Learning Research (JMLR)"},{"journal-title":"Stochastic gradient descent optimizes over-parameterized deep RELU networks","year":"2018","author":"zou","key":"ref36"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ALLERTON.2016.7852249"},{"key":"ref34","article-title":"SGD converges to global minimum in deep learning via star-convex path","author":"zhou","year":"2019","journal-title":"Proc International Conference on Learning Representations(ICLR)"},{"article-title":"Identity matters in deep learning","year":"2016","author":"hardt","key":"ref10"},{"key":"ref11","first-page":"770","article-title":"Deep residual learning for image recognition","author":"he","year":"2015","journal-title":"Proc IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"ref12","first-page":"2261","article-title":"Densely connected convolutional networks","author":"huang","year":"2016","journal-title":"Proc IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"ref13","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"2015","journal-title":"Proc International Conference on Machine Learning (ICML)"},{"key":"ref14","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2015","journal-title":"Proc International Conference on Learning Representations (ICLR)"},{"key":"ref15","article-title":"Learning multiple layers of features from tiny images","author":"krizhevsky","year":"2009","journal-title":"Technical Report"},{"key":"ref16","article-title":"Rapid, robust, and reliable blind deconvolution via nonconvex optimization","author":"li","year":"2018","journal-title":"Applied and Computational Harmonic Analysis"},{"key":"ref17","first-page":"597","article-title":"Convergence analysis of two-layer neural networks with relu activation","author":"li","year":"2017","journal-title":"Proc Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref18","article-title":"Rectifier nonlinearities improve neural network acoustic models","author":"maas","year":"2013","journal-title":"Proc ICML Workshop on Deep Learning for Audio Speech and Language Processing"},{"key":"ref19","first-page":"807","article-title":"Rectified linear units improve restricted boltzmann machines","author":"nair","year":"2010","journal-title":"Proc International Conference on Machine Learning (ICML)"},{"article-title":"Highway networks","year":"2015","author":"srivastava","key":"ref28"},{"key":"ref4","article-title":"Fast and accurate deep network learning by exponential linear units (elus)","author":"clevert","year":"2015","journal-title":"Proc International Conference on Learning Representations (ICLR)"},{"key":"ref27","article-title":"How does batch normalization help optimization?","author":"santurkar","year":"2018","journal-title":"Proc Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref3","article-title":"On the convergence of a class of Adam-type algorithms for non-convex optimization","author":"chen","year":"2019","journal-title":"Proc International Conference on Learning Representations (ICLR)"},{"key":"ref6","first-page":"1675","article-title":"Gradient descent finds global minima of deep neural networks","volume":"97","author":"du","year":"2019","journal-title":"Proc International Conference on Machine Learning (ICML)"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/BF02551274"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-015-0871-8"},{"key":"ref7","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"duchi","year":"2011","journal-title":"Journal of Machine Learning Research"},{"key":"ref2","first-page":"7694","article-title":"Understanding batch normalization","author":"bjorck","year":"2018","journal-title":"Proc Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref9","first-page":"315","article-title":"Deep sparse rectifier neural networks","author":"glorot","year":"2011","journal-title":"Proc International Conference on Artificial Intelligence and Statistics (AISTATS)"},{"key":"ref1","article-title":"A convergence analysis of gradient descent for deep linear neural networks","author":"arora","year":"2019","journal-title":"Proc International Conference on Learning Representations (ICLR)"},{"journal-title":"Introductory Lectures on Convex Optimization A Basic Course","year":"2014","author":"nesterov","key":"ref20"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/S0893-6080(98)00116-6"},{"key":"ref21","article-title":"Skip connections eliminate singularities","author":"orhan","year":"2018","journal-title":"Proc International Conference on Learning Representations (ICLR)"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729586"},{"key":"ref23","article-title":"On the convergence of Adam and beyond","author":"reddi","year":"2018","journal-title":"Proc International Conference on Learning Representations (ICLR)"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1038\/323533a0"},{"key":"ref25","first-page":"234","article-title":"U-net: Convolutional networks for biomedical image segmentation","author":"ronneberger","year":"2015","journal-title":"Proc Medical Image Computing and Computer-Assisted Intervention (MICCAI)"}],"event":{"name":"2020 IEEE International Conference on Big Data (Big Data)","start":{"date-parts":[[2020,12,10]]},"location":"Atlanta, GA, USA","end":{"date-parts":[[2020,12,13]]}},"container-title":["2020 IEEE International Conference on Big Data (Big Data)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9377717\/9377728\/09378359.pdf?arnumber=9378359","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,27]],"date-time":"2022-06-27T12:12:47Z","timestamp":1656331967000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9378359\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,12,10]]},"references-count":36,"URL":"https:\/\/doi.org\/10.1109\/bigdata50022.2020.9378359","relation":{},"subject":[],"published":{"date-parts":[[2020,12,10]]}}}