{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T09:52:23Z","timestamp":1742982743821,"version":"3.40.3"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030551797"},{"type":"electronic","value":"9783030551803"}],"license":[{"start":{"date-parts":[[2020,8,25]],"date-time":"2020-08-25T00:00:00Z","timestamp":1598313600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,8,25]],"date-time":"2020-08-25T00:00:00Z","timestamp":1598313600000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-55180-3_27","type":"book-chapter","created":{"date-parts":[[2020,8,24]],"date-time":"2020-08-24T23:04:00Z","timestamp":1598310240000},"page":"360-374","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Convergence of a Relaxed Variable Splitting Method for Learning Sparse Neural Networks via $$\\ell _1, \\ell _0$$, and Transformed-$$\\ell _1$$ Penalties"],"prefix":"10.1007","author":[{"given":"Thu","family":"Dinh","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jack","family":"Xin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,8,25]]},"reference":[{"key":"27_CR1","unstructured":"Brutzkus, A., Globerson, A.: Globally optimal gradient descent for a convnet with gaussian inputs (2017). ArXiv preprint 1702.07966"},{"key":"27_CR2","unstructured":"Cho, Y., Saul, L.K.: Kernel methods for deep learning. In: Advances in neural information processing systems, pp. 342\u2013350 (2009)"},{"key":"27_CR3","unstructured":"Dauphin, Y.N., Fan, A., Auli, M., Grangier, D.: Language modeling with gated convolutional networks (2016). ArXiv preprint 1612.08083"},{"key":"27_CR4","unstructured":"Du, S., Lee, J., Tian, Y.: When is a convolutional filter easy to learn? (2017). ArXiv 1709.06129"},{"key":"27_CR5","unstructured":"Du, S., Lee, J., Tian, Y., Poczos, B., Singh, A.: Gradient descent learns one-hidden-layer CNN: don\u2019t be afraid of spurious local minima. In: International Conference on Machine Learning, ICML (2018)"},{"key":"27_CR6","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi, J., Hazan, E., Singer, Y.: Adaptive subgradient methods for online learning and stochastic optimization. J. Mach. Learn. Res. 12, 2121\u20132159 (2011)","journal-title":"J. Mach. Learn. Res."},{"key":"27_CR7","unstructured":"Han, S., Mao, H., Dally, W.J.: Deep compression: compressing deep neural networks with pruning, trained quantization and Huffman coding (2015). ArXiv preprint 1510.00149"},{"issue":"6","key":"27_CR8","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G.E., Mohamed, A., Jaitly, N., Senior, A., Vanhoucke, V., Nguyen, P., Sainath, T.N., Kingsbury, B.: Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"27_CR9","unstructured":"Kingma, D., Ba, J.: Adam: a method for stochastic optimization (2014). ArXiv preprint 1412.6980"},{"key":"27_CR10","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in neural information processing systems, pp. 1097\u20131105 (2012)"},{"key":"27_CR11","unstructured":"LeCun, Y., Denker, J., Solla, S.: Optimal brain damage. In: NIPS, vol. 2, pp. 598\u2013605 (1989)"},{"key":"27_CR12","unstructured":"Louizos, C., Welling, M., Kingma, D.: Learning sparse neural networks through $$\\ell _0$$ regularization (2018). ArXiv preprint 1712.01312v2"},{"issue":"3","key":"27_CR13","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1080\/10556788.2014.936438","volume":"30","author":"Z Lu","year":"2014","unstructured":"Lu, Z., Zhang, Y.: Penalty decomposition methods for rank minimization. Optim. Methods Softw. 30(3), 531\u2013558 (2014). \nhttps:\/\/doi.org\/10.1080\/10556788.2014.936438","journal-title":"Optim. Methods Softw."},{"key":"27_CR14","unstructured":"Molchanov, D., Ashukha, A., Vetrov, D.: Variational dropout sparsifies deep neural networks (2017). ArXiv preprint 1701.05369"},{"issue":"2","key":"27_CR15","doi-asserted-by":"publisher","first-page":"633","DOI":"10.1137\/S0036139997327794","volume":"61","author":"M Nikolova","year":"2000","unstructured":"Nikolova, M.: Local strong homogeneity of a regularized estimator. SIAM J. Appl. Math. 61(2), 633\u2013658 (2000)","journal-title":"SIAM J. Appl. Math."},{"issue":"5","key":"27_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/0041-5553(64)90137-5","volume":"4","author":"B Polyak","year":"1964","unstructured":"Polyak, B.: Some methods of speeding up the convergence of iteration methods. USSR Comput. Math. Math. Phys. 4(5), 1\u201317 (1964)","journal-title":"USSR Comput. Math. Math. Phys."},{"key":"27_CR17","unstructured":"Reddi, S., Kale, S., Kumar, S.: On the convergence of adam and beyond. In: International Conference on Learning Representations (2018)"},{"key":"27_CR18","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins, H., Monro, S.: A stochastic approximation method. Ann. Math. Stat. 22, 400\u2013407 (1951)","journal-title":"Ann. Math. Stat."},{"key":"27_CR19","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1038\/323533a0","volume":"323","author":"D Rumelhart","year":"1986","unstructured":"Rumelhart, D., Hinton, G., Williams, R.: Learning representations by back-propagating errors. Nature 323, 533\u2013536 (1986)","journal-title":"Nature"},{"key":"27_CR20","unstructured":"Shamir, O.: Distribution-specific hardness of learning neural networks (2016). ArXiv preprint 1609.01037"},{"key":"27_CR21","unstructured":"Taylor, G., Burmeister, R., Xu, Z., Singh, B., Patel, A., Goldstein, T.: Training neural networks without gradients: a scalable admm approach. In: International Conference on Machine Learning, pp. 2722\u20132731 (2016)"},{"key":"27_CR22","unstructured":"Tian, Y.: An analytical formula of population gradient for two-layered relu network and its applications in convergence and critical point analysis (2017). ArXiv preprint 1703.00560"},{"key":"27_CR23","unstructured":"Tieleman, T., Hinton, G.: Divide the gradient by a running average of its recent magnitude. coursera: neural networks for machine learning. Technical report (2017)"},{"key":"27_CR24","unstructured":"Ullrich, K., Meeds, E., Welling, M.: Soft weight-sharing for neural network compression. In: ICLR (2017)"},{"issue":"1","key":"27_CR25","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1007\/s10915-018-0757-z","volume":"78","author":"Y Wang","year":"2018","unstructured":"Wang, Y., Zeng, J., Yin, W.: Global convergence of ADMM in nonconvex nonsmooth optimization. J. Sci. Comput. 78(1), 29\u201363 (2018). \nhttps:\/\/doi.org\/10.1007\/s10915-018-0757-z","journal-title":"J. Sci. Comput."},{"key":"27_CR26","unstructured":"Zhang, C., Bengio, S., Hardt, M., Recht, B., Vinyals, O.: Understanding deep learning requires rethinking generalization (2016). ArXiv preprint 1611.03530"},{"issue":"2","key":"27_CR27","doi-asserted-by":"publisher","first-page":"511","DOI":"10.4310\/cms.2017.v15.n2.a9","volume":"15","author":"S Zhang","year":"2017","unstructured":"Zhang, S., Xin, J.: Minimization of transformed $$l_1$$ penalty: closed form representation and iterative thresholding algorithms. Commun. Math. Sci. 15(2), 511\u2013537 (2017). \nhttps:\/\/doi.org\/10.4310\/cms.2017.v15.n2.a9","journal-title":"Commun. Math. Sci."},{"key":"27_CR28","unstructured":"Zhang, T., Ye, S., Zhang, K., Tang, J., Wen, W., Fardad, M., Wang, Y.: A systematic DNN weight pruning framework using alternating direction method of multipliers. arXiv preprint 1804.03294 (2018). \nhttps:\/\/arxiv.org\/abs\/1804.03294"}],"container-title":["Advances in Intelligent Systems and Computing","Intelligent Systems and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-55180-3_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,8,24]],"date-time":"2020-08-24T23:04:29Z","timestamp":1598310269000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-55180-3_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,8,25]]},"ISBN":["9783030551797","9783030551803"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-55180-3_27","relation":{},"ISSN":["2194-5357","2194-5365"],"issn-type":[{"type":"print","value":"2194-5357"},{"type":"electronic","value":"2194-5365"}],"subject":[],"published":{"date-parts":[[2020,8,25]]},"assertion":[{"value":"25 August 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"IntelliSys","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Proceedings of SAI Intelligent Systems Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"London","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 September 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 September 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"intellisys2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/saiconference.com\/IntelliSys","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}