{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T19:17:14Z","timestamp":1777058234363,"version":"3.51.4"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819708338","type":"print"},{"value":"9789819708345","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-97-0834-5_20","type":"book-chapter","created":{"date-parts":[[2024,3,12]],"date-time":"2024-03-12T02:02:48Z","timestamp":1710208968000},"page":"340-359","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Optimizing the\u00a0Parallelism of\u00a0Communication and\u00a0Computation in\u00a0Distributed Training Platform"],"prefix":"10.1007","author":[{"given":"Xiang","family":"Hou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sheng","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rui","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tiejun","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lizhou","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianmin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,3,12]]},"reference":[{"key":"20_CR1","unstructured":"Abadi, M., et al.: Tensorflow: large-scale machine learning on heterogeneous distributed systems. arXiv preprint arXiv:1603.04467 (2016)"},{"key":"20_CR2","doi-asserted-by":"crossref","unstructured":"Agarwal, N., Krishna, T., Peh, L.S., Jha, N.K.: Garnet: a detailed on-chip network model inside a full-system simulator. In IEEE International Symposium on Performance Analysis of Systems and Software, ISPASS 2009, April 26\u201328, 2009, Boston, Massachusetts, USA, Proceedings (2009)","DOI":"10.1109\/ISPASS.2009.4919636"},{"key":"20_CR3","doi-asserted-by":"crossref","unstructured":"Akata, Z., Reed, S., Walter, D., Lee, H., Schiele, B.: Evaluation of output embeddings for fine-grained image classification. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2927\u20132936 (2015)","DOI":"10.1109\/CVPR.2015.7298911"},{"key":"20_CR4","unstructured":"Chao, C., Saeta, B.: Cloud tpu: codesigning architecture and infrastructure. In: Hot Chips, volume 31 (2019)"},{"key":"20_CR5","unstructured":"Chaturapruek, S., Duchi, J.C., R\u00e9, C.: Asynchronous stochastic convex optimization: the noise is in the noise and sgd don\u2019t care. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"20_CR6","unstructured":"Chen, J., Pan, X., Monga, R., Bengio, S., Jozefowicz, R.: Revisiting distributed synchronous sgd. arXiv preprint arXiv:1604.00981 (2016)"},{"key":"20_CR7","doi-asserted-by":"crossref","unstructured":"Cho, M., Finkler, U., Kung, D., Hunter, H., et al.: Blueconnect: decomposing all-reduce for deep learning on heterogeneous network hierarchy. Ibm Journal of Research and Development, PP(99), 1\u20131 (2019)","DOI":"10.1147\/JRD.2019.2947013"},{"issue":"4","key":"20_CR8","first-page":"429","volume":"10","author":"J Chorowski","year":"2015","unstructured":"Chorowski, J., Bahdanau, D., Serdyuk, D., Cho, K., Bengio, Y.: Attention-based models for speech recognition. Computer ence 10(4), 429\u2013439 (2015)","journal-title":"Computer ence"},{"key":"20_CR9","unstructured":"Dean, J., et al.: Large scale distributed deep networks. In: Advances in Neural Information Processing Systems, 25 (2012)"},{"key":"20_CR10","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"20_CR11","doi-asserted-by":"crossref","unstructured":"Graves, A., Jaitly, N., Mohamed, A.R.: Hybrid speech recognition with deep bidirectional lstm. In: Automatic Speech Recognition and Understanding (ASRU), 2013 IEEE Workshop (2013)","DOI":"10.1109\/ASRU.2013.6707742"},{"key":"20_CR12","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S. and Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"20_CR13","doi-asserted-by":"crossref","unstructured":"Hou, X., Xu, R., Ma, S., Wang, Q., Jiang, W., Lu, H.: Co-designing the topology\/algorithm to accelerate distributed training. In: 2021 IEEE Intl Conf on Parallel & Distributed Processing with Applications, Big Data & Cloud Computing, Sustainable Computing & Communications, Social Computing & Networking (ISPA\/BDCloud\/SocialCom\/SustainCom), pp. 1010\u20131018, 2021","DOI":"10.1109\/ISPA-BDCloud-SocialCom-SustainCom52081.2021.00141"},{"key":"20_CR14","doi-asserted-by":"crossref","unstructured":"Jouppi, N.P., Yoon, D.H., Ashcraft, M., Gottscho, M., Patterson, D.: Ten lessons from three generations shaped google\u2019s tpuv4i : Industrial product. In: 2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture (ISCA) (2021)","DOI":"10.1109\/ISCA52012.2021.00010"},{"key":"20_CR15","doi-asserted-by":"crossref","unstructured":"Kielmann, T., Hofman, R.F., Bal, H.E., Plaat, A., Bhoedjang, R.A.. Magpie: Mpi\u2019s collective communication operations for clustered wide area systems. In: Proceedings of the seventh ACM SIGPLAN symposium on Principles and practice of parallel programming, pp. 131\u2013140 (1999)","DOI":"10.1145\/329366.301116"},{"key":"20_CR16","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: NIPS (2012)"},{"issue":"11","key":"20_CR17","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86(11), 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"20_CR18","unstructured":"Li, M., et al.: Scaling distributed machine learning with the parameter server. In: 11th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14), pp. 583\u2013598 (2014)"},{"key":"20_CR19","unstructured":"Li, M., Andersen, D.G., Smola, A.J., Yu, K.: Communication efficient distributed machine learning with the parameter server. In: Advances in Neural Information Processing Systems, 27 19\u201327 (2014)"},{"key":"20_CR20","unstructured":"Naumov, M., et al.: Deep learning recommendation model for personalization and recommendation systems. arXiv preprint arXiv:1906.00091 (2019)"},{"issue":"2","key":"20_CR21","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1016\/j.jpdc.2008.09.002","volume":"69","author":"P Patarasuk","year":"2009","unstructured":"Patarasuk, P., Yuan, X.: Bandwidth optimal all-reduce algorithms for clusters of workstations. J. Parall. Distrib. Comput. 69(2), 117\u2013124 (2009)","journal-title":"J. Parall. Distrib. Comput."},{"key":"20_CR22","unstructured":"Perez, L., Wang, J.: The effectiveness of data augmentation in image classification using deep learning. arXiv preprint arXiv:1712.04621 (2017)"},{"key":"20_CR23","doi-asserted-by":"crossref","unstructured":"Rashidi, S., Shurpali, P., Sridharan, S., Hassani, N., Krishna, T.: Scalable distributed training of recommendation models: An astra-sim + ns3 case-study with tcp\/ip transport. In: 2020 IEEE Symposium on High-Performance Interconnects (HOTI) (2020)","DOI":"10.1109\/HOTI51249.2020.00020"},{"key":"20_CR24","doi-asserted-by":"crossref","unstructured":"Rashidi, S., Sridharan, S., Srinivasan, S., Krishna, T.: Astra-sim: enabling sw\/hw co-design exploration for distributed dl training platforms. In: 2020 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS) (2020)","DOI":"10.1109\/ISPASS48437.2020.00018"},{"key":"20_CR25","unstructured":"Rashidi, S., Sridharan, S., Srinivasan, S., Denton, M., Krishna, T.: Efficient communication acceleration for next-gen scale-up deep learning training platforms. arXiv preprint (2020)"},{"key":"20_CR26","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1007\/978-3-642-12331-3_2","volume-title":"Modeling and Tools for Network Simulation","author":"GF Riley","year":"2010","unstructured":"Riley, G.F., Henderson, T.R.: The ns-3 network simulator. In: Wehrle, K., G\u00fcne\u015f, M., Gross, J. (eds.) Modeling and Tools for Network Simulation, pp. 15\u201334. Springer Berlin Heidelberg, Berlin, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-12331-3_2"},{"key":"20_CR27","unstructured":"Schlkopf, B., Platt, J., Hofmann, T.: Map-reduce for machine learning on multicore. In: Advances in Neural Information Processing Systems 19: Proceedings of the 2006 Conference"},{"key":"20_CR28","doi-asserted-by":"crossref","unstructured":"Sepulchre, R., Paley, D.A. Leonard, N.E.: Stabilization of planar collective motion: All-to-all communication. IEEE Trans. Autom. Contr. 52(5), 811\u2013824 (2007)","DOI":"10.1109\/TAC.2007.898077"},{"key":"20_CR29","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, Q., Chu, X., Li, B.: dag model of synchronous stochastic gradient descent in distributed deep learning. In: 2018 IEEE 24th International Conference on Parallel and Distributed Systems (ICPADS), pp. 425\u2013432. IEEE (2018)","DOI":"10.1109\/PADSW.2018.8644932"},{"key":"20_CR30","unstructured":"Tr\u00e4ff, J.L.: Efficient all-gather communication on parallel systems with hierarchical communication structure. preparation (2003)"},{"key":"20_CR31","unstructured":"Vaswani, A.: Attention is all you need. In: Advances In Neural Information Processing Systems, 30 (2017)"},{"key":"20_CR32","doi-asserted-by":"crossref","unstructured":"Xu, R., Ma, S., Guo, Y., Li, D.: A survey of design and optimization for systolic array based dnn accelerators. In: ACM Computing Surveys (2023)","DOI":"10.1145\/3604802"},{"key":"20_CR33","doi-asserted-by":"crossref","unstructured":"Rui, X., Ma, S., Wang, Y., Chen, X., Guo, Y.: Configurable multi-directional systolic array architecture for convolutional neural networks. ACM Trans. Architect. Code Optim. (TACO) 18(4), 1\u201324 (2021)","DOI":"10.1145\/3460776"},{"issue":"11","key":"20_CR34","first-page":"2860","volume":"33","author":"X Rui","year":"2021","unstructured":"Rui, X., Ma, S., Wang, Y., Guo, Y., Li, D., Qiao, Y.: Heterogeneous systolic array architecture for compact cnns hardware accelerators. IEEE Trans. Parallel Distrib. Syst. 33(11), 2860\u20132871 (2021)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"20_CR35","unstructured":"Zhang, H., et al.: Poseidon: an efficient communication architecture for distributed deep learning on $$\\{$$GPU$$\\}$$ clusters. In: 2017 USENIX Annual Technical Conference (USENIX ATC 17), pp. 181\u2013193 (2017)"}],"container-title":["Lecture Notes in Computer Science","Algorithms and Architectures for Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-0834-5_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,12]],"date-time":"2024-03-12T02:07:06Z","timestamp":1710209226000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-0834-5_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9789819708338","9789819708345"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-0834-5_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"12 March 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICA3PP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Algorithms and Architectures for Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tianjin","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 October 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ica3pp2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/tjutanklab.com\/ica3pp2023\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Online submission system","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"439","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"145","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"33% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}