{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T10:15:17Z","timestamp":1783678517140,"version":"3.55.0"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sci Comput"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1007\/s10915-026-03357-x","type":"journal-article","created":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T13:41:13Z","timestamp":1782394873000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Clipped Stochastic Gradient Tracking For Locally Smooth Functions"],"prefix":"10.1007","volume":"108","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1254-5390","authenticated-orcid":false,"given":"Leilei","family":"Mei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junyu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,25]]},"reference":[{"key":"3357_CR1","unstructured":"Assran, M., Loizou, N., Ballas, N., Rabbat, M.: Stochastic gradient push for distributed deep learning. In International Conference on Machine Learning, pages 344\u2013353. PMLR, (2019)"},{"issue":"2","key":"3357_CR2","doi-asserted-by":"publisher","first-page":"330","DOI":"10.1287\/moor.2016.0817","volume":"42","author":"HH Bauschke","year":"2017","unstructured":"Bauschke, H.H., Bolte, J., Teboulle, M.: A descent lemma beyond lipschitz gradient continuity: first-order methods revisited and applications. Math. Oper. Res. 42(2), 330\u2013348 (2017)","journal-title":"Math. Oper. Res."},{"key":"3357_CR3","doi-asserted-by":"crossref","unstructured":"Birsan, T., Tiba, D.: One hundred years since the introduction of the set distance by dimitrie pompeiu. In System Modeling and Optimization: Proceedings of the 22nd IFIP TC7 Conference held from July 18\u201322, 2005, in Turin, Italy 22, pages 35\u201339. Springer, (2006)","DOI":"10.1007\/0-387-33006-2_4"},{"issue":"3","key":"3357_CR4","doi-asserted-by":"publisher","first-page":"2131","DOI":"10.1137\/17M1138558","volume":"28","author":"J Bolte","year":"2018","unstructured":"Bolte, J., Sabach, S., Teboulle, M., Vaisbourd, Y.: First order methods beyond convexity and lipschitz gradient continuity with applications to quadratic inverse problems. SIAM J. Optim. 28(3), 2131\u20132151 (2018)","journal-title":"SIAM J. Optim."},{"issue":"4","key":"3357_CR5","doi-asserted-by":"publisher","first-page":"1985","DOI":"10.1109\/TIT.2015.2399924","volume":"61","author":"EJ Candes","year":"2015","unstructured":"Candes, E.J., Li, X., Soltanolkotabi, M.: Phase retrieval via wirtinger flow: Theory and algorithms. IEEE Trans. Inf. Theory 61(4), 1985\u20132007 (2015)","journal-title":"IEEE Trans. Inf. Theory"},{"key":"3357_CR6","unstructured":"Chen, Z., Zhou, Y., Liang, Y., Lu, Z.: Generalized-smooth nonconvex optimization is as efficient as smooth nonconvex optimization. In International Conference on Machine Learning, pages 5396\u20135427. PMLR, (2023)"},{"issue":"8","key":"3357_CR7","doi-asserted-by":"publisher","first-page":"4289","DOI":"10.1109\/TSP.2012.2198470","volume":"60","author":"J Chen","year":"2012","unstructured":"Chen, J., Sayed, A.H.: Diffusion adaptation strategies for distributed optimization and learning over networks. IEEE Trans. Signal Process. 60(8), 4289\u20134305 (2012)","journal-title":"IEEE Trans. Signal Process."},{"key":"3357_CR8","unstructured":"Cutkosky, A., Orabona, F.: Momentum-based variance reduction in non-convex sgd. In: Advances in neural information processing systems, vol. 32, (2019)"},{"key":"3357_CR9","unstructured":"Defazio, A., Bach, F., Lacoste-Julien, S.: Saga: A fast incremental gradient method with support for non-strongly convex composite objectives. In: Advances in neural information processing systems, vol. 27, (2014)"},{"issue":"3","key":"3357_CR10","first-page":"1245","volume":"5","author":"Q Guannan","year":"2017","unstructured":"Guannan, Q., Li, N.: Harnessing smoothness to accelerate distributed optimization. IEEE Transactions on Control of Network Systems 5(3), 1245\u20131260 (2017)","journal-title":"IEEE Transactions on Control of Network Systems"},{"key":"3357_CR11","unstructured":"Hong, M., Hajinezhad, D., Zhao, M.-M.: Prox-pda: The proximal primal-dual algorithm for fast distributed nonconvex optimization and learning over networks. In International Conference on Machine Learning, pages 1529\u20131538. PMLR, (2017)"},{"key":"3357_CR12","unstructured":"Hong, M., Zeng, S., Zhang, J., Sun, H.: On the divergence of decentralized non-convex optimization. arXiv:2006.11662, (2020)"},{"issue":"9","key":"3357_CR13","doi-asserted-by":"publisher","first-page":"5310","DOI":"10.1109\/TNNLS.2022.3170944","volume":"34","author":"X Jiang","year":"2022","unstructured":"Jiang, X., Zeng, X., Sun, J., Chen, J.: Distributed stochastic gradient tracking algorithm with variance reduction for non-convex optimization. IEEE Transactions on Neural Networks and Learning Systems 34(9), 5310\u20135321 (2022)","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"3357_CR14","first-page":"2771","volume":"34","author":"J Jin","year":"2021","unstructured":"Jin, J., Zhang, B., Wang, H., Wang, L.: Non-convex distributionally robust optimization: Non-asymptotic analysis. Adv. Neural. Inf. Process. Syst. 34, 2771\u20132782 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3357_CR15","unstructured":"Johnson, R., Zhang, T.: Accelerating stochastic gradient descent using predictive variance reduction. In: Advances in neural information processing systems, vol. 26, (2013)"},{"key":"3357_CR16","unstructured":"Koloskova, A., Hendrikx, H., Stich, S.U.: Revisiting gradient clipping: Stochastic bias and tight convergence guarantees. In International Conference on Machine Learning, pages 17343\u201317363. PMLR, (2023)"},{"key":"3357_CR17","first-page":"11422","volume":"34","author":"A Koloskova","year":"2021","unstructured":"Koloskova, A., Lin, T., Stich, S.U.: An improved analysis of gradient tracking for decentralized machine learning. Adv. Neural. Inf. Process. Syst. 34, 11422\u201311435 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"17","key":"3357_CR18","doi-asserted-by":"publisher","first-page":"4494","DOI":"10.1109\/TSP.2019.2926022","volume":"67","author":"Z Li","year":"2019","unstructured":"Li, Z., Shi, W., Yan, M.: A decentralized proximal-gradient method with network independent step-sizes and separated convergence rates. IEEE Trans. Signal Process. 67(17), 4494\u20134506 (2019)","journal-title":"IEEE Trans. Signal Process."},{"issue":"180","key":"3357_CR19","first-page":"1","volume":"21","author":"B Li","year":"2020","unstructured":"Li, B., Cen, S., Chen, Y., Chi, Y.: Communication-efficient distributed optimization in networks with gradient tracking and variance reduction. J. Mach. Learn. Res. 21(180), 1\u201351 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"3357_CR20","doi-asserted-by":"publisher","first-page":"40238","DOI":"10.52202\/075280-1749","volume":"36","author":"H Li","year":"2023","unstructured":"Li, H., Qian, J., Tian, Y., Rakhlin, A., Jadbabaie, A.: Convex and non-convex optimization under generalized smoothness. Adv. Neural. Inf. Process. Syst. 36, 40238\u201340271 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3357_CR21","unstructured":"Lian, X., Zhang, C., Zhang, H., Hsieh, C.-J., Zhang, W., Liu, J.: Can decentralized algorithms outperform centralized algorithms? a case study for decentralized parallel stochastic gradient descent. Adv. Neural. Inf. Process. Syst. 30, (2017)"},{"issue":"5","key":"3357_CR22","doi-asserted-by":"publisher","first-page":"1165","DOI":"10.1109\/TSP.2012.2236830","volume":"61","author":"Q Ling","year":"2012","unstructured":"Ling, Q., Wen, Z., Yin, W.: Decentralized jointly sparse optimization by reweighted lq minimization. IEEE Trans. Signal Process. 61(5), 1165\u20131170 (2012)","journal-title":"IEEE Trans. Signal Process."},{"key":"3357_CR23","doi-asserted-by":"crossref","unstructured":"Lu, S., Zhang, X., Sun, H., Hong, M.: Gnsd: A gradient-tracking based nonconvex stochastic algorithm for decentralized optimization. In 2019 IEEE Data Science Workshop (DSW), pages 315\u2013321. IEEE, (2019)","DOI":"10.1109\/DSW.2019.8755807"},{"issue":"4","key":"3357_CR24","doi-asserted-by":"publisher","first-page":"2577","DOI":"10.1287\/moor.2024.0407","volume":"50","author":"Z Lu","year":"2024","unstructured":"Lu, Z., Mei, S.: Primal-dual extrapolation methods for monotone inclusions under local lipschitz continuity. Math. Oper. Res. 50(4), 2577\u20132599 (2024)","journal-title":"Math. Oper. Res."},{"issue":"1","key":"3357_CR25","doi-asserted-by":"publisher","first-page":"333","DOI":"10.1137\/16M1099546","volume":"28","author":"H Lu","year":"2018","unstructured":"Lu, H., Freund, R.M., Nesterov, Y.: Relatively smooth convex optimization by first-order methods, and applications. SIAM J. Optim. 28(1), 333\u2013354 (2018)","journal-title":"SIAM J. Optim."},{"issue":"3","key":"3357_CR26","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1109\/MSP.2020.2975210","volume":"37","author":"A Nedic","year":"2020","unstructured":"Nedic, A.: Distributed gradient methods for convex machine learning problems in networks. IEEE Signal Process. Mag. 37(3), 92\u2013101 (2020)","journal-title":"IEEE Signal Process. Mag."},{"issue":"1","key":"3357_CR27","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1109\/TAC.2008.2009515","volume":"54","author":"A Nedic","year":"2009","unstructured":"Nedic, A., Ozdaglar, A.: Distributed subgradient methods for multi-agent optimization. IEEE Trans. Autom. Control 54(1), 48\u201361 (2009)","journal-title":"IEEE Trans. Autom. Control"},{"issue":"4","key":"3357_CR28","doi-asserted-by":"publisher","first-page":"2597","DOI":"10.1137\/16M1084316","volume":"27","author":"A Nedic","year":"2017","unstructured":"Nedic, A., Olshevsky, A., Shi, W.: Achieving geometric convergence for distributed optimization over time-varying graphs. SIAM J. Optim. 27(4), 2597\u20132633 (2017)","journal-title":"SIAM J. Optim."},{"key":"3357_CR29","unstructured":"Nesterov, Y.: Introductory lectures on convex optimization: A basic course, volume\u00a087. Springer Science & Business Media, (2013)"},{"key":"3357_CR30","unstructured":"Nhan, H., Pham, L., M., Nguyen, Dzung, T., Phan, Q., Tran-Dinh.: Proxsarah: An efficient algorithmic framework for stochastic composite nonconvex optimization. The Journal of Machine Learning Research 21(1), 4455\u20134502 (2020)"},{"key":"3357_CR31","unstructured":"Pesquet, J. Repetti, C. A.: A class of randomized primal-dual algorithms for distributed optimization, (2014). arXiv:1406.6404 arXiv preprint"},{"key":"3357_CR32","first-page":"290","volume":"6","author":"E Renyi","year":"1959","unstructured":"Renyi, E.: On random graph. Publicationes Mathematicate 6, 290\u2013297 (1959)","journal-title":"On random graph. Publicationes Mathematicate"},{"issue":"1","key":"3357_CR33","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1109\/JPROC.2006.887293","volume":"95","author":"J Reza Olfati-Saber","year":"2007","unstructured":"Reza Olfati-Saber, J., Fax, A., Murray, R.M.: Consensus and cooperation in networked multi-agent systems. Proc. IEEE 95(1), 215\u2013233 (2007)","journal-title":"Proc. IEEE"},{"issue":"1","key":"3357_CR34","first-page":"409","volume":"187","author":"P Shi","year":"2021","unstructured":"Shi, P., Nedi\u0107, A.: Distributed stochastic gradient tracking methods. Math. Program. 187(1), 409\u2013457 (2021)","journal-title":"Math. Program."},{"issue":"2","key":"3357_CR35","doi-asserted-by":"publisher","first-page":"944","DOI":"10.1137\/14096668X","volume":"25","author":"W Shi","year":"2015","unstructured":"Shi, W., Ling, Q., Gang, W., Yin, W.: Extra: An exact first-order algorithm for decentralized consensus optimization. SIAM J. Optim. 25(2), 944\u2013966 (2015)","journal-title":"SIAM J. Optim."},{"key":"3357_CR36","unstructured":"Sun, H., Lu, S., Hong, M.: Improving the sample and communication complexity for decentralized non-convex optimization: Joint gradient estimation and tracking. In International conference on machine learning, pages 9217\u20139228. PMLR, (2020)"},{"issue":"22","key":"3357_CR37","doi-asserted-by":"publisher","first-page":"5912","DOI":"10.1109\/TSP.2019.2943230","volume":"67","author":"H Sun","year":"2019","unstructured":"Sun, H., Hong, M.: Distributed non-convex first-order optimization and information processing: Lower complexity bounds and rate optimal algorithms. IEEE Trans. Signal Process. 67(22), 5912\u20135928 (2019)","journal-title":"IEEE Trans. Signal Process."},{"issue":"2","key":"3357_CR38","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1137\/19M1259973","volume":"32","author":"Y Sun","year":"2022","unstructured":"Sun, Y., Scutari, G., Daneshmand, A.: Distributed optimization based on gradient tracking revisited: Enhancing convergence rate via surrogation. SIAM J. Optim. 32(2), 354\u2013385 (2022)","journal-title":"SIAM J. Optim."},{"key":"3357_CR39","unstructured":"Tang, H., Lian, X., Yan, M., Zhang, C., Liu, J.: $$ d^2$$: Decentralized training over decentralized data. In International Conference on Machine Learning, pages 4848\u20134856. PMLR, (2018)"},{"issue":"1","key":"3357_CR40","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1007\/s10107-018-1284-2","volume":"170","author":"M Teboulle","year":"2018","unstructured":"Teboulle, M.: A simplified view of first order methods for optimization. Math. Program. 170(1), 67\u201396 (2018)","journal-title":"Math. Program."},{"key":"3357_CR41","doi-asserted-by":"crossref","unstructured":"Xin, R., Usman, A., Khan, S., Kar: A near-optimal stochastic gradient method for decentralized non-convex finite-sum optimization, (2020). arXiv:2008.07428 arXiv preprint","DOI":"10.1109\/CDC42340.2020.9304007"},{"key":"3357_CR42","doi-asserted-by":"publisher","first-page":"6255","DOI":"10.1109\/TSP.2020.3031071","volume":"68","author":"R Xin","year":"2020","unstructured":"Xin, R., Usman, A., Khan, S.: Kar: Variance-reduced decentralized stochastic optimization with accelerated convergence. IEEE Trans. Signal Process. 68, 6255\u20136271 (2020)","journal-title":"IEEE Trans. Signal Process."},{"issue":"11","key":"3357_CR43","doi-asserted-by":"publisher","first-page":"1869","DOI":"10.1109\/JPROC.2020.3024266","volume":"108","author":"R Xin","year":"2020","unstructured":"Xin, R., Pu, S., Nedi\u0107, A., Khan, U.A.: A general framework for decentralized optimization with first-order methods. Proc. IEEE 108(11), 1869\u20131889 (2020)","journal-title":"Proc. IEEE"},{"issue":"10","key":"3357_CR44","doi-asserted-by":"publisher","first-page":"5150","DOI":"10.1109\/TAC.2021.3122586","volume":"67","author":"R Xin","year":"2021","unstructured":"Xin, R., Usman, A., Khan, S.: Kar: A fast randomized incremental gradient method for decentralized nonconvex optimization. IEEE Trans. Autom. Control 67(10), 5150\u20135165 (2021)","journal-title":"IEEE Trans. Autom. Control"},{"issue":"1","key":"3357_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1137\/20M1361158","volume":"32","author":"R Xin","year":"2022","unstructured":"Xin, R., Khan, U.A., Kar, S.: Fast decentralized nonconvex finite-sum optimization with recursive variance reduction. SIAM J. Optim. 32(1), 1\u201328 (2022)","journal-title":"SIAM J. Optim."},{"key":"3357_CR46","doi-asserted-by":"crossref","unstructured":"Yeh, J.: Real analysis: theory of measure and integration second edition, World Scientific Publishing Company (2006)","DOI":"10.1142\/6023"},{"issue":"3","key":"3357_CR47","doi-asserted-by":"publisher","first-page":"1835","DOI":"10.1137\/130943170","volume":"26","author":"K Yuan","year":"2016","unstructured":"Yuan, K., Ling, Q., Yin, W.: On the convergence of decentralized gradient descent. SIAM J. Optim. 26(3), 1835\u20131854 (2016)","journal-title":"SIAM J. Optim."},{"issue":"11","key":"3357_CR48","doi-asserted-by":"publisher","first-page":"2834","DOI":"10.1109\/TSP.2018.2818081","volume":"66","author":"J Zeng","year":"2018","unstructured":"Zeng, J., Yin, W.: On nonconvex decentralized gradient descent. IEEE Trans. Signal Process. 66(11), 2834\u20132848 (2018)","journal-title":"IEEE Trans. Signal Process."},{"key":"3357_CR49","doi-asserted-by":"crossref","unstructured":"Zhang, J.: Stochastic bregman proximal gradient method revisited: Kernel conditioning and painless variance reduction. Mathematical Programming, 1\u201360 (2025)","DOI":"10.1007\/s10107-025-02285-2"},{"issue":"1","key":"3357_CR50","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1109\/TSIPN.2017.2672403","volume":"4","author":"G Zhang","year":"2017","unstructured":"Zhang, G., Heusdens, R.: Distributed optimization using the primal-dual method of multipliers. IEEE Transactions on Signal and Information Processing over Networks 4(1), 173\u2013187 (2017)","journal-title":"IEEE Transactions on Signal and Information Processing over Networks"},{"issue":"2","key":"3357_CR51","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1287\/ijoo.2021.0029","volume":"6","author":"J Zhang","year":"2024","unstructured":"Zhang, J., Hong, M.: First-order algorithms without lipschitz gradient: A sequential local optimization approach. INFORMS Journal on Optimization 6(2), 118\u2013136 (2024)","journal-title":"INFORMS Journal on Optimization"},{"key":"3357_CR52","unstructured":"Zhang, J., You, K.: Decentralized stochastic gradient tracking for non-convex empirical risk minimization, (2019). arXiv:1909.02712 arXiv preprint"},{"key":"3357_CR53","unstructured":"Zhang, J., He, T., Sra, S., Jadbabaie, A.: Why gradient clipping accelerates training: A theoretical justification for adaptivity. In: International Conference on Learning Representations, (2019)"},{"key":"3357_CR54","first-page":"2228","volume":"34","author":"J Zhang","year":"2021","unstructured":"Zhang, J., Ni, C., Szepesvari, C., Wang, M., et al.: On the convergence and sample efficiency of variance-reduced policy gradient method. Adv. Neural. Inf. Process. Syst. 34, 2228\u20132240 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3357_CR55","doi-asserted-by":"crossref","unstructured":"Zhao, Z., Lu, S., Hong, M., Palomar, D.P.: Distributed optimization for generalized phase retrieval over networks. In 2018 52nd Asilomar Conference on Signals, Systems, and Computers, pages 48\u201352. IEEE, (2018)","DOI":"10.1109\/ACSSC.2018.8645496"},{"issue":"3","key":"3357_CR56","doi-asserted-by":"publisher","first-page":"2275","DOI":"10.1137\/22M1500496","volume":"33","author":"L Zhaosong","year":"2023","unstructured":"Zhaosong, L., Mei, S.: Accelerated first-order methods for convex optimization with locally lipschitz continuous gradient. SIAM J. Optim. 33(3), 2275\u20132310 (2023)","journal-title":"SIAM J. Optim."}],"container-title":["Journal of Scientific Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10915-026-03357-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10915-026-03357-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10915-026-03357-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T09:38:19Z","timestamp":1783676299000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10915-026-03357-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,25]]},"references-count":56,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,8]]}},"alternative-id":["3357"],"URL":"https:\/\/doi.org\/10.1007\/s10915-026-03357-x","relation":{},"ISSN":["0885-7474","1573-7691"],"issn-type":[{"value":"0885-7474","type":"print"},{"value":"1573-7691","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,25]]},"assertion":[{"value":"28 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 May 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 June 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}}],"article-number":"55"}}