{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T22:28:36Z","timestamp":1743114516300,"version":"3.40.3"},"publisher-location":"Cham","reference-count":37,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030948757"},{"type":"electronic","value":"9783030948764"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-030-94876-4_4","type":"book-chapter","created":{"date-parts":[[2022,1,18]],"date-time":"2022-01-18T08:20:02Z","timestamp":1642494002000},"page":"60-75","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["The Impact of Synchronization in Parallel Stochastic Gradient Descent"],"prefix":"10.1007","author":[{"given":"Karl","family":"B\u00e4ckstr\u00f6m","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marina","family":"Papatriantafilou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Philippas","family":"Tsigas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,1,17]]},"reference":[{"key":"4_CR1","unstructured":"Agarwal, A., Duchi, J.C.: Distributed delayed stochastic optimization. In: Advances in Neural Information Processing Systems, pp. 873\u2013881 (2011)"},{"key":"4_CR2","unstructured":"Alistarh, D., Chatterjee, B., Kungurtsev, V.: Elastic consistency: a general consistency model for distributed stochastic gradient descent. arXiv preprint arXiv:2001.05918 (2020)"},{"key":"4_CR3","doi-asserted-by":"publisher","unstructured":"Alistarh, D., De Sa, C., Konstantinov, N.: The convergence of stochastic gradient descent in asynchronous shared memory. In: ACM Symposium on Principles of Distributed Computing, PODC 2018, pp. 169\u2013178. ACM, New York (2018). https:\/\/doi.org\/10.1145\/3212734.3212763","DOI":"10.1145\/3212734.3212763"},{"key":"4_CR4","unstructured":"Alistarh, D., Grubic, D., Li, J., Tomioka, R., Vojnovic, M.: QSGD: communication-efficient SGD via gradient quantization and encoding. In: Advances in Neural Information Processing Systems, pp. 1709\u20131720 (2017)"},{"key":"4_CR5","doi-asserted-by":"crossref","unstructured":"Amdahl, G.M.: Validity of the single processor approach to achieving large scale computing capabilities. In: Proceedings of the April 18\u201320, 1967, Spring Joint Computer Conference, pp. 483\u2013485 (1967)","DOI":"10.1145\/1465482.1465560"},{"key":"4_CR6","doi-asserted-by":"crossref","unstructured":"B\u00e4ckstr\u00f6m, K., Papatriantafilou, M., Tsigas, P.: MindTheStep-AsyncPSGD: adaptive asynchronous parallel stochastic gradient descent. In: 2019 IEEE International Conference on Big Data (Big Data), pp. 16\u201325. IEEE (2019)","DOI":"10.1109\/BigData47090.2019.9006054"},{"key":"4_CR7","doi-asserted-by":"crossref","unstructured":"Ben-Nun, T., Besta, M., Huber, S., Ziogas, A.N., Peter, D., Hoefler, T.: A modular benchmarking infrastructure for high-performance and reproducible deep learning. In: 2019 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 66\u201377. IEEE (2019)","DOI":"10.1109\/IPDPS.2019.00018"},{"issue":"4","key":"4_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3320060","volume":"52","author":"T Ben-Nun","year":"2019","unstructured":"Ben-Nun, T., Hoefler, T.: Demystifying parallel and distributed deep learning: an in-depth concurrency analysis. ACM Comput. Surv. (CSUR) 52(4), 1\u201343 (2019)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"4_CR9","volume-title":"Parallel and Distributed Computation: Numerical Methods","author":"DP Bertsekas","year":"1989","unstructured":"Bertsekas, D.P., Tsitsiklis, J.N.: Parallel and Distributed Computation: Numerical Methods, vol. 23. Prentice Hall, Upper Saddle River (1989)"},{"key":"4_CR10","doi-asserted-by":"publisher","unstructured":"B\u00e4ckstr\u00f6m, K., Walulya, I., Papatriantafilou, M., Tsigas, P.: Consistent lock-free parallel stochastic gradient descent for fast and stable convergence. In: 2021 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 423\u2013432 (2021). https:\/\/doi.org\/10.1109\/IPDPS49936.2021.00051","DOI":"10.1109\/IPDPS49936.2021.00051"},{"key":"4_CR11","unstructured":"Chaturapruek, S., Duchi, J.C., R\u00e9, C.: Asynchronous stochastic convex optimization: the noise is in the noise and SGD don\u2019t care. In: Advances in Neural Information Processing Systems, pp. 1531\u20131539 (2015)"},{"key":"4_CR12","unstructured":"De Sa, C.M., Zhang, C., Olukotun, K., R\u00e9, C., R\u00e9, C.: Taming the wild: a unified analysis of Hogwild-style algorithms. In: Advances in Neural Information Processing Systems, vol. 28, pp. 2674\u20132682. Curran Associates, Inc. (2015). http:\/\/papers.nips.cc\/paper\/5717-taming-the-wild-a-unified-analysis-of-hogwild-style-algorithms.pdf"},{"key":"4_CR13","doi-asserted-by":"crossref","unstructured":"Gupta, S., Zhang, W., Wang, F.: Model accuracy and runtime tradeoff in distributed deep learning: a systematic study. In: 2016 IEEE 16th International Conference on Data Mining (ICDM), pp. 171\u2013180. IEEE (2016)","DOI":"10.1109\/ICDM.2016.0028"},{"key":"4_CR14","unstructured":"Ho, Q., et al.: More effective distributed ml via a stale synchronous parallel parameter server. In: Advances in Neural Information Processing Systems, pp. 1223\u20131231 (2013)"},{"key":"4_CR15","unstructured":"Jiang, Z., Balu, A., Hegde, C., Sarkar, S.: Collaborative deep learning in fixed topology networks. In: Advances in Neural Information Processing Systems, pp. 5904\u20135914 (2017)"},{"key":"4_CR16","unstructured":"Keskar, N.S., Mudigere, D., Nocedal, J., Smelyanskiy, M., Tang, P.T.P.: On large-batch training for deep learning: generalization gap and sharp minima. arXiv:1609.04836 (2016)"},{"key":"4_CR17","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: Scaling distributed machine learning with the parameter server. In: 11th Symposium on Operating Systems Design and Implementation, pp. 583\u2013598 (2014)","DOI":"10.1145\/2640087.2644155"},{"key":"4_CR18","doi-asserted-by":"crossref","unstructured":"Li, S., Ben-Nun, T., Girolamo, S.D., Alistarh, D., Hoefler, T.: Taming unbalanced training workloads in deep learning with partial collective operations. In: 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 45\u201361 (2020)","DOI":"10.1145\/3332466.3374528"},{"key":"4_CR19","unstructured":"Lian, X., Huang, Y., Li, Y., Liu, J.: Asynchronous parallel stochastic gradient for nonconvex optimization. In: Advances in Neural Information Processing Systems, pp. 2737\u20132745 (2015)"},{"key":"4_CR20","unstructured":"Lian, X., Zhang, W., Zhang, C., Liu, J.: Asynchronous decentralized parallel stochastic gradient descent. In: International Conference on Machine Learning, pp. 3043\u20133052. PMLR (2018)"},{"key":"4_CR21","doi-asserted-by":"crossref","unstructured":"Lopez, F., Chow, E., Tomov, S., Dongarra, J.: Asynchronous SGD for DNN training on shared-memory parallel architectures. In: International Parallel and Distributed Processing Symposium Workshops (IPDPSW), pp. 1\u20134. IEEE (2020)","DOI":"10.1109\/IPDPSW50202.2020.00168"},{"key":"4_CR22","doi-asserted-by":"crossref","unstructured":"Ma, Y., Rusu, F., Torres, M.: Stochastic gradient descent on modern hardware: multi-core CPU or GPU? Synchronous or asynchronous? In: 2019 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 1063\u20131072. IEEE (2019)","DOI":"10.1109\/IPDPS.2019.00113"},{"issue":"4","key":"4_CR23","doi-asserted-by":"publisher","first-page":"2202","DOI":"10.1137\/16M1057000","volume":"27","author":"H Mania","year":"2017","unstructured":"Mania, H., Pan, X., Papailiopoulos, D., Recht, B., Ramchandran, K., Jordan, M.I.: Perturbed iterate analysis for asynchronous stochastic optimization. SIAM J. Optim. 27(4), 2202\u20132229 (2017)","journal-title":"SIAM J. Optim."},{"key":"4_CR24","unstructured":"McMahan, B., Streeter, M.: Delay-tolerant algorithms for asynchronous distributed online learning. In: Advances in Neural Information Processing Systems, vol. 27, pp. 2915\u20132923. Curran Associates, Inc. (2014). http:\/\/papers.nips.cc\/paper\/5242-delay-tolerant-algorithms-for-asynchronous-distributed-online-learning.pdf"},{"key":"4_CR25","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1016\/j.cviu.2017.05.007","volume":"161","author":"D Mishkin","year":"2017","unstructured":"Mishkin, D., Sergievskiy, N., Matas, J.: Systematic evaluation of convolution neural network advances on the ImageNet. Comput. Vis. Image Underst. 161, 11\u201319 (2017)","journal-title":"Comput. Vis. Image Underst."},{"key":"4_CR26","doi-asserted-by":"crossref","unstructured":"Mitliagkas, I., Zhang, C., Hadjis, S., R\u00e9, C.: Asynchrony begets momentum, with an application to deep learning. In: 54th Annual Allerton Conference on Communication, Control, and Computing, pp. 997\u20131004. IEEE (2016)","DOI":"10.1109\/ALLERTON.2016.7852343"},{"key":"4_CR27","unstructured":"Nguyen, L.M., Nguyen, P.H., van Dijk, M., Richt\u00e1rik, P., Scheinberg, K., Tak\u00e1\u010d, M.: SGD and Hogwild! convergence without the bounded gradients assumption. arXiv preprint arXiv:1802.03801 (2018)"},{"key":"4_CR28","unstructured":"Recht, B., Re, C., Wright, S., Niu, F.: Hogwild: a lock-free approach to parallelizing stochastic gradient descent. In: Advances in Neural Information Processing Systems (NIPS), vol. 24, pp. 693\u2013701. Curran Associates, Inc. (2011)"},{"key":"4_CR29","doi-asserted-by":"crossref","unstructured":"Sallinen, S., Satish, N., Smelyanskiy, M., Sury, S.S., R\u00e9, C.: High performance parallel stochastic gradient descent in shared memory. In: IEEE International Parallel and Distributed Processing Symposium, pp. 873\u2013882. IEEE (2016)","DOI":"10.1109\/IPDPS.2016.107"},{"key":"4_CR30","unstructured":"Sra, S., Yu, A.W., Li, M., Smola, A.J.: AdaDelay: delay adaptive distributed stochastic convex optimization. arXiv preprint arXiv:1508.05003 (2015)"},{"key":"4_CR31","unstructured":"Stich, S.U.: Local SGD converges fast and communicates little. In: International Conference on Learning Representations (ICLR) (2019)"},{"key":"4_CR32","unstructured":"Sutskever, I., Martens, J., Dahl, G., Hinton, G.: On the importance of initialization and momentum in deep learning. In: International Conference on Machine Learning, pp. 1139\u20131147 (2013)"},{"key":"4_CR33","doi-asserted-by":"crossref","unstructured":"Wei, J., Gibson, G.A., Gibbons, P.B., Xing, E.P.: Automating dependence-aware parallelization of machine learning training on distributed shared memory. In: 14th EuroSys Conference 2019, pp. 1\u201317 (2019)","DOI":"10.1145\/3302424.3303954"},{"key":"4_CR34","unstructured":"Zhang, W., Gupta, S., Lian, X., Liu, J.: Staleness-aware Async-SGD for distributed deep learning. arXiv preprint arXiv:1511.05950 (2015)"},{"key":"4_CR35","unstructured":"Zhang, W., Gupta, S., Lian, X., Liu, J.: Staleness-aware Async-SGD for distributed deep learning. In: Proceedings of the Twenty-Fifth International Joint Conference on Artificial Intelligence, IJCAI 2016, pp. 2350\u20132356. AAAI Press (2016)"},{"key":"4_CR36","doi-asserted-by":"publisher","unstructured":"Zhao, X., An, A., Liu, J., Chen, B.X.: Dynamic stale synchronous parallel distributed training for deep learning. In: 2019 IEEE 39th International Conference on Distributed Computing Systems (ICDCS), pp. 1507\u20131517, July 2019. https:\/\/doi.org\/10.1109\/ICDCS.2019.00150","DOI":"10.1109\/ICDCS.2019.00150"},{"key":"4_CR37","unstructured":"Zinkevich, M., Weimer, M., Li, L., Smola, A.J.: Parallelized stochastic gradient descent. In: Advances in Neural Information Processing Systems, pp. 2595\u20132603 (2010)"}],"container-title":["Lecture Notes in Computer Science","Distributed Computing and Intelligent Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-94876-4_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,23]],"date-time":"2023-01-23T10:24:19Z","timestamp":1674469459000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-94876-4_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783030948757","9783030948764"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-94876-4_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"17 January 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICDCIT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Distributed Computing and Internet Technology","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Bhubaneswar","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 January 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 January 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdcit2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.icdcit.ac.in\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"50","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"11","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"22% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.2","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.7","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Additionally, 4 invited papers are included.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}