{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:09:31Z","timestamp":1778080171476,"version":"3.51.4"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2019,11,15]],"date-time":"2019-11-15T00:00:00Z","timestamp":1573776000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,11,15]],"date-time":"2019-11-15T00:00:00Z","timestamp":1573776000000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Glob Optim"],"published-print":{"date-parts":[[2020,5]]},"DOI":"10.1007\/s10898-019-00856-0","type":"journal-article","created":{"date-parts":[[2019,11,15]],"date-time":"2019-11-15T13:02:46Z","timestamp":1573822966000},"page":"97-124","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Block layer decomposition schemes for training deep neural networks"],"prefix":"10.1007","volume":"77","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9496-6097","authenticated-orcid":false,"given":"Laura","family":"Palagi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5292-1774","authenticated-orcid":false,"given":"Ruggiero","family":"Seccia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,11,15]]},"reference":[{"issue":"4","key":"856_CR1","doi-asserted-by":"publisher","first-page":"2037","DOI":"10.1137\/120887679","volume":"23","author":"A Beck","year":"2013","unstructured":"Beck, A., Tetruashvili, L.: On the convergence of block coordinate descent type methods. SIAM J. Optim. 23(4), 2037\u20132060 (2013)","journal-title":"SIAM J. Optim."},{"issue":"3","key":"856_CR2","doi-asserted-by":"publisher","first-page":"807","DOI":"10.1137\/S1052623494268522","volume":"6","author":"DP Bertsekas","year":"1996","unstructured":"Bertsekas, D.P.: Incremental least squares methods and the extended Kalman filter. SIAM J. Optim. 6(3), 807\u2013822 (1996)","journal-title":"SIAM J. Optim."},{"issue":"3","key":"856_CR3","doi-asserted-by":"publisher","first-page":"334","DOI":"10.1057\/palgrave.jors.2600425","volume":"48","author":"DP Bertsekas","year":"1997","unstructured":"Bertsekas, D.P.: Nonlinear programming. J. Oper. Res. Soc. 48(3), 334\u2013334 (1997)","journal-title":"J. Oper. Res. Soc."},{"key":"856_CR4","unstructured":"Bertsekas, D.P.: Incremental gradient, subgradient, and proximal methods for convex optimization: a survey. CoRR, arXiv:abs\/1507.01030 (2015)"},{"issue":"3","key":"856_CR5","doi-asserted-by":"publisher","first-page":"627","DOI":"10.1137\/S1052623497331063","volume":"10","author":"DP Bertsekas","year":"2000","unstructured":"Bertsekas, D.P., Tsitsiklis, J.N.: Gradient convergence in gradient methods with errors. SIAM J. Optim. 10(3), 627\u2013642 (2000)","journal-title":"SIAM J. Optim."},{"key":"856_CR6","doi-asserted-by":"crossref","unstructured":"Bottou, L.: Large-scale machine learning with stochastic gradient descent. In: COMPSTAT (2010)","DOI":"10.1007\/978-3-7908-2604-3_16"},{"issue":"2","key":"856_CR7","doi-asserted-by":"publisher","first-page":"223","DOI":"10.1137\/16M1080173","volume":"60","author":"L Bottou","year":"2018","unstructured":"Bottou, L., Curtis, F.E., Nocedal, J.: Optimization methods for large-scale machine learning. SIAM Rev. 60(2), 223\u2013311 (2018)","journal-title":"SIAM Rev."},{"key":"856_CR8","first-page":"80","volume":"235","author":"L Bravi","year":"2014","unstructured":"Bravi, L., Sciandrone, M.: An incremental decomposition method for unconstrained optimization. Appl. Math. Comput. 235, 80\u201386 (2014)","journal-title":"Appl. Math. Comput."},{"issue":"8","key":"856_CR9","doi-asserted-by":"publisher","first-page":"1891","DOI":"10.1162\/08997660152469396","volume":"13","author":"C Buzzi","year":"2001","unstructured":"Buzzi, C., Grippo, L., Sciandrone, M.: Convergent decomposition techniques for training RBF neural networks. Neural Comput. 13(8), 1891\u20131920 (2001)","journal-title":"Neural Comput."},{"key":"856_CR10","unstructured":"Chauhan, V.K., Dahiya, K., Sharma, A.: Mini-batch block-coordinate based stochastic average adjusted gradient methods to solve big data problems. In: Proceedings of the Ninth Asian Conference on Machine Learning, volume\u00a077 of Proceedings of Machine Learning Research, pp. 49\u201364. PMLR, 15\u201317 Nov 2017"},{"key":"856_CR11","unstructured":"Chollet, F., et\u00a0al.: Keras (2015)"},{"key":"856_CR12","first-page":"2933","volume":"27","author":"YN Dauphin","year":"2014","unstructured":"Dauphin, Y.N., Pascanu, R., Gulcehre, C., Cho, K., Ganguli, S., Bengio, Y.: Identifying and attacking the saddle point problem in high-dimensional non-convex optimization. Adv. Neural Inf. Process. Syst. 27, 2933\u20132941 (2014)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"856_CR13","first-page":"1646","volume":"27","author":"A Defazio","year":"2014","unstructured":"Defazio, A., Bach, F., Lacoste-Julien, S.: SAGA: a fast incremental gradient method with support for non-strongly convex composite objectives. Adv. Neural Inf. Process. Syst. 27, 1646\u20131654 (2014)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"856_CR14","first-page":"2121","volume":"12","author":"J Duchi","year":"2011","unstructured":"Duchi, J., Hazan, E., Singer, Y.: Adaptive subgradient methods for online learning and stochastic optimization. J. Mach. Learn. Res. 12, 2121\u20132159 (2011)","journal-title":"J. Mach. Learn. Res."},{"key":"856_CR15","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1007\/978-1-4612-4380-9_6","volume-title":"Breakthroughs in Statistics","author":"RA Fisher","year":"1992","unstructured":"Fisher, R.A.: Statistical methods for research workers. In: Johnson, N.L., Kotz, S. (eds.) Breakthroughs in Statistics, pp. 66\u201370. Springer, Berlin (1992)"},{"key":"856_CR16","unstructured":"Glorot, X., Bengio, Y.: Understanding the difficulty of training deep feedforward neural networks. In: Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics, pp. 249\u2013256 (2010)"},{"key":"856_CR17","volume-title":"Deep Learning","author":"I Goodfellow","year":"2016","unstructured":"Goodfellow, I., Bengio, Y., Courville, A.: Deep Learning. MIT Press, Cambridge (2016)"},{"issue":"11","key":"856_CR18","doi-asserted-by":"publisher","first-page":"2146","DOI":"10.1109\/TNNLS.2015.2475621","volume":"27","author":"L Grippo","year":"2016","unstructured":"Grippo, L., Manno, A., Sciandrone, M.: Decomposition techniques for multilayer perceptron training. IEEE Trans. Neural Netw. Learn. Syst. 27(11), 2146\u20132159 (2016)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"4","key":"856_CR19","doi-asserted-by":"publisher","first-page":"587","DOI":"10.1080\/10556789908805730","volume":"10","author":"L Grippo","year":"1999","unstructured":"Grippo, L., Sciandrone, M.: Globally convergent block-coordinate techniques for unconstrained optimization. Optim. Methods Softw. 10(4), 587\u2013637 (1999)","journal-title":"Optim. Methods Softw."},{"key":"856_CR20","doi-asserted-by":"publisher","first-page":"489","DOI":"10.1016\/j.neucom.2005.12.126","volume":"70","author":"G Huang","year":"2006","unstructured":"Huang, G., Zhu, Q., Siew, C.: Extreme learning machine: theory and applications. Neurocomputing 70, 489\u2013501 (2006)","journal-title":"Neurocomputing"},{"issue":"2","key":"856_CR21","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1007\/s13042-011-0019-y","volume":"2","author":"G-B Huang","year":"2011","unstructured":"Huang, G.-B., Wang, D.H., Lan, Y.: Extreme learning machines: a survey. Int. J. Mach. Learn. Cybern. 2(2), 107\u2013122 (2011)","journal-title":"Int. J. Mach. Learn. Cybern."},{"key":"856_CR22","first-page":"315","volume":"26","author":"R Johnson","year":"2013","unstructured":"Johnson, R., Zhang, T.: Accelerating stochastic gradient descent using predictive variance reduction. Adv. Neural Inf. Process. Syst. 26, 315\u2013323 (2013)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"856_CR23","unstructured":"Jones, E., Oliphant, T., Peterson, P., et\u00a0al.: SciPy: open source scientific tools for Python. [Online; Accessed $$<$$today$$>$$] (2001)"},{"key":"856_CR24","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. CoRR, arXiv:abs\/1412.6980 (2014)"},{"issue":"2","key":"856_CR25","doi-asserted-by":"publisher","first-page":"341","DOI":"10.1137\/100802001","volume":"22","author":"Y Nesterov","year":"2012","unstructured":"Nesterov, Y.: Efficiency of coordinate descent methods on huge-scale optimization problems. SIAM J. Optim. 22(2), 341\u2013362 (2012)","journal-title":"SIAM J. Optim."},{"key":"856_CR26","first-page":"543","volume":"269","author":"YE Nesterov","year":"1983","unstructured":"Nesterov, Y.E.: A method for solving the convex programming problem with convergence rate o $$(1\/{\\rm k}\\hat{}2)$$. Dokl. Akad. Nauk SSSR 269, 543\u2013547 (1983)","journal-title":"Dokl. Akad. Nauk SSSR"},{"key":"856_CR27","volume-title":"Numerical Optimization","author":"J Nocedal","year":"2006","unstructured":"Nocedal, J., Wright, S.J.: Numerical Optimization, 2nd edn. Springer, New York (2006)","edition":"2"},{"key":"856_CR28","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1007\/s10898-018-0701-7","volume":"73","author":"L Palagi","year":"2018","unstructured":"Palagi, L.: Global optimization issues in deep network regression: an overview. J. Glob. Optim. 73, 239\u2013277 (2018)","journal-title":"J. Glob. Optim."},{"issue":"6","key":"856_CR29","doi-asserted-by":"publisher","first-page":"143","DOI":"10.1007\/s12532-013-0051-x","volume":"5","author":"T Qin","year":"2013","unstructured":"Qin, T., Scheinberg, K., Goldfarb, D.: Efficient block-coordinate descent algorithms for the group lasso. Math. Program. Comput. 5(6), 143\u2013169 (2013)","journal-title":"Math. Program. Comput."},{"key":"856_CR30","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1214\/aoms\/1177729586","volume":"22","author":"H Robbins","year":"1951","unstructured":"Robbins, H., Monro, S.: A stochastic approximation method. Ann. Math. Stat. 22, 400\u2013407 (1951)","journal-title":"Ann. Math. Stat."},{"issue":"2","key":"856_CR31","first-page":"26","volume":"4","author":"T Tieleman","year":"2012","unstructured":"Tieleman, T., Hinton, G.: Lecture 6.5-RMSProp: divide the gradient by a running average of its recent magnitude. COURSERA Neural Netw. Mach. Learn. 4(2), 26\u201331 (2012)","journal-title":"COURSERA Neural Netw. Mach. Learn."},{"key":"856_CR32","unstructured":"Wang, H., Banerjee, A.: Randomized block coordinate descent for online and stochastic optimization. arXiv preprint arXiv:1407.0107 (2014)"},{"issue":"1","key":"856_CR33","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/s10107-015-0892-3","volume":"151","author":"SJ Wright","year":"2015","unstructured":"Wright, S.J.: Coordinate descent algorithms. Math. Program. 151(1), 3\u201334 (2015)","journal-title":"Math. Program."},{"key":"856_CR34","unstructured":"Yu, A.W., Huang, L., Lin, Q., Salakhutdinov, R., Carbonell, J.: Normalized gradient with adaptive stepsize method for deep neural network training. CoRR arXiv:abs\/1707.04822 (2017)"},{"key":"856_CR35","unstructured":"Zhao, T., Yu, M., Wang, Y., Arora, R., Liu, H.: Accelerated mini-batch randomized block coordinate descent method. In: Advances in Neural Information Processing Systems, pp. 3329\u20133337 (2014)"}],"container-title":["Journal of Global Optimization"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10898-019-00856-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10898-019-00856-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10898-019-00856-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,14]],"date-time":"2020-11-14T00:30:21Z","timestamp":1605313821000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10898-019-00856-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,15]]},"references-count":35,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2020,5]]}},"alternative-id":["856"],"URL":"https:\/\/doi.org\/10.1007\/s10898-019-00856-0","relation":{},"ISSN":["0925-5001","1573-2916"],"issn-type":[{"value":"0925-5001","type":"print"},{"value":"1573-2916","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,11,15]]},"assertion":[{"value":"3 December 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 November 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 November 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}