{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T23:42:15Z","timestamp":1740181335945,"version":"3.37.3"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2021,3,2]],"date-time":"2021-03-02T00:00:00Z","timestamp":1614643200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,3,2]],"date-time":"2021-03-02T00:00:00Z","timestamp":1614643200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61972158"],"award-info":[{"award-number":["61972158"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["CCF Trans. HPC"],"published-print":{"date-parts":[[2021,6]]},"DOI":"10.1007\/s42514-021-00064-x","type":"journal-article","created":{"date-parts":[[2021,3,2]],"date-time":"2021-03-02T17:02:47Z","timestamp":1614704567000},"page":"171-185","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Rationing bandwidth resources for mitigating network resource contention in distributed DNN training clusters"],"prefix":"10.1007","volume":"3","author":[{"given":"Qiang","family":"Qi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fei","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,3,2]]},"reference":[{"key":"64_CR1","first-page":"265","volume":"2016","author":"M Abadi","year":"2016","unstructured":"Abadi, M., Barham, P., Chen, J., Chen, Z., Davis, A., Dean, J., Devin, M., Ghemawat, S., Irving, G., Isard, M., et al.: Tensorflow: a system for large-scale machine learning. Proc. USENIX OSDI 2016, 265\u2013283 (2016)","journal-title":"Proc. USENIX OSDI"},{"key":"64_CR2","unstructured":"Berral JL, Wang C, Youssef A (2020) AI4DL: mining behaviors of deep learning workloads for resource management. In: Proceedings of USENIX HotCloud (2020)"},{"key":"64_CR3","first-page":"532","volume":"2019","author":"C Chen","year":"2019","unstructured":"Chen, C., Wang, W., Li, B.: Round-Robin synchronization: mitigating communication Bottlenecks in parameter servers. Proc IEEE INFOCOM 2019, 532\u2013540 (2019)","journal-title":"Proc IEEE INFOCOM"},{"key":"64_CR4","unstructured":"Chen, T., Li, M., Li, Y., Lin, M., Wang, N., Wang, M., Xiao, T., Xu, B., Zhang, C., Zhang, Z.: Mxnet: a flexible and efficient machine learning library for heterogeneous distributed systems (2015). arXiv preprint arXiv:151201274"},{"key":"64_CR5","first-page":"485","volume":"2019","author":"J Gu","year":"2019","unstructured":"Gu, J., Chowdhury, M., Shin, K.G., Zhu, Y., Jeon, M., Qian, J., Liu, H., Guo, C.: Tiresias: a GPU cluster manager for distributed deep learning. Proc. USENIX NSDI 2019, 485\u2013500 (2019)","journal-title":"Proc. USENIX NSDI"},{"key":"64_CR6","doi-asserted-by":"publisher","first-page":"873","DOI":"10.1109\/TNET.2015.2389270","volume":"24","author":"J Guo","year":"2015","unstructured":"Guo, J., Liu, F., Lui, J.C.S., Jin, H.: Fair network bandwidth allocation in iaas datacenters via a cooperative game approach. IEEE\/ACM Trans. Netw. 24, 873\u2013886 (2015)","journal-title":"IEEE\/ACM Trans. Netw."},{"key":"64_CR7","first-page":"1737","volume":"2015","author":"S Guptaand","year":"2015","unstructured":"Guptaand, S., Agrawal, A., Gopalakrishnan, K., Narayanan, P.: Deep learning with limited numerical precision. Proc. ICML 2015, 1737\u20131746 (2015)","journal-title":"Proc. ICML"},{"key":"64_CR8","first-page":"430","volume":"2019","author":"XS Huang","year":"2019","unstructured":"Huang, X.S., Chen, A., Ng, T.: Green, yellow, yield: end-host traffic scheduling for distributed deep learning with tensorlights. Proc. IEEE IPDPSW 2019, 430\u2013437 (2019)","journal-title":"Proc. IEEE IPDPSW"},{"key":"64_CR9","unstructured":"Jayarajan, A., Wei, J., Gibson, G., Fedorova, A., Pekhimenko, G.: Priority-based parameter propagation for distributed DNN training. In: Talwalkar, A., Smith, V., Zaharia, M. (eds.) Proceedings of Machine Learning and Systems 2019, vol. 3, pp. 132\u2013145 (2019)"},{"key":"64_CR10","first-page":"947","volume":"2019","author":"M Jeon","year":"2019","unstructured":"Jeon, M., Venkataraman, S., Phanishayee, A., Qian, J., Xiao, W., Yang, F.: Analysis of large-scale multi-tenant GPU clusters for DNN training workloads. Proc. USENIX ATC 2019, 947\u2013960 (2019)","journal-title":"Proc. USENIX ATC"},{"key":"64_CR11","unstructured":"Jiang, Y., Zhu, Y., Lan, C., Yi, B., Cui, Y., Guo, C.: A unified architecture for accelerating distributed DNN training in heterogeneous GPU\/CPU clusters. In: Proceedings of USENIX OSDI, pp 463\u2013479 (2020)"},{"key":"64_CR12","first-page":"1097","volume":"2012","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Proc. NIPS 2012, 1097\u20131105 (2012)","journal-title":"Proc. NIPS"},{"key":"64_CR13","first-page":"289","volume":"2020","author":"M Kshiteej","year":"2020","unstructured":"Kshiteej, M., Arjun, B., Arjun, S., Shivaram, V., Aditya, A., Amar, P., Shuchi, C.: Themis: fair and efficient GPU cluster scheduling. Proc. USENIX NSDI 2020, 289\u2013304 (2020)","journal-title":"Proc. USENIX NSDI"},{"key":"64_CR14","unstructured":"Lin, Y., Han, S., Mao, H., Wang, Y., Dally, W.J.: Deep Gradient Compression: Reducing the Communication Bandwidth for Distributed Training. (2017) arXiv preprint arXiv:171201887"},{"key":"64_CR15","first-page":"41","volume":"2018","author":"L Luo","year":"2018","unstructured":"Luo, L., Nelson, J., Ceze, L., Phanishayee, A., Krishnamurthy, A.: Parameter hub: a rack-scale parameter server for distributed deep neural network training. Proc. ACM SOCC 2018, 41\u201354 (2018)","journal-title":"Proc. ACM SOCC"},{"key":"64_CR16","unstructured":"Luo L, West P, Krishnamurthy A, Ceze L, Nelson J (2020) PLink: Discovering and Exploiting Datacenter Network Locality for Efficient Cloud-based Distributed Training. Proc.\u00a0of MLSys 2020"},{"key":"64_CR17","unstructured":"Mai, L., Hong, C., Costa, P. (2015) Optimizing network performance in distributed machine learning. In: Proceedings of USENIX HotCloud 2015"},{"key":"64_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3363554","volume":"53","author":"R Mayer","year":"2020","unstructured":"Mayer, R., Jacobsen, H.A.: Scalable deep learning on distributed infrastructures: challenges, techniques, and tools. ACM Comput Surv (CSUR) 53, 1\u201337 (2020)","journal-title":"ACM Comput Surv (CSUR)"},{"key":"64_CR19","first-page":"2430","volume":"2017","author":"A Mirhoseini","year":"2017","unstructured":"Mirhoseini, A., Pham, H., Le, Q.V., Steiner, B., Larsen, R., Zhou, Y., Kumar, N., Norouzi, M., Bengio, S., Dean, J.: Device placement optimization with reinforcement learning. Proc. ICML 2017, 2430\u20132439 (2017)","journal-title":"Proc. ICML"},{"key":"64_CR20","first-page":"1","volume":"2019","author":"D Narayanan","year":"2019","unstructured":"Narayanan, D., Harlap, A., Phanishayee, A., Seshadri, V., Devanur, N.R., Ganger, G.R., Gibbons, P.B., Zaharia, M.: PipeDream: generalized pipeline parallelism for DNN training. Proc. ACM SOSP 2019, 1\u201315 (2019)","journal-title":"Proc. ACM SOSP"},{"key":"64_CR21","doi-asserted-by":"publisher","first-page":"1853","DOI":"10.1109\/JLT.2019.2894179","volume":"37","author":"T Panayiotou","year":"2019","unstructured":"Panayiotou, T., Manousakis, K., Chatzis, S.P., Ellinas, G.: A data-driven bandwidth allocation framework With QoS considerations for EONs. J Lightwave Technol 37, 1853\u20131864 (2019)","journal-title":"J Lightwave Technol"},{"key":"64_CR22","first-page":"1","volume":"2018","author":"Y Peng","year":"2018","unstructured":"Peng, Y., Bao, Y., Chen, Y., Wu, C., Guo, C.: Optimus: an efficient dynamic resource scheduler for deep learning clusters. Proc. EuroSys 2018, 1\u201314 (2018)","journal-title":"Proc. EuroSys"},{"key":"64_CR23","first-page":"16","volume":"2019","author":"Y Peng","year":"2019","unstructured":"Peng, Y., Zhu, Y., Chen, Y., Bao, Y., Yi, B., Lan, C., Wu, C., Guo, C.: A generic communication scheduler for distributed DNN training acceleration. Proc. ACM SOSP 2019, 16\u201329 (2019)","journal-title":"Proc. ACM SOSP"},{"key":"64_CR24","volume-title":"Artificial Intelligence: A Modern Approach","author":"S Russell","year":"2020","unstructured":"Russell, S., Norvig, P.: Artificial Intelligence: A Modern Approach. Prentice Hall, New York (2020)"},{"key":"64_CR25","first-page":"381","volume":"13","author":"D Shen","year":"2020","unstructured":"Shen, D., Luo, J., Dong, F., Jin, J., Zhang, J., Shen, J.: Facilitating Application-Aware Bandwidth Allocation in the Cloud with One-Step-Ahead Traffic Information. IEEE Trans. Serv. Comput. 13, 381\u2013394 (2020)","journal-title":"IEEE Trans. Serv. Comput."},{"key":"64_CR26","first-page":"172","volume":"2019","author":"S Shi","year":"2019","unstructured":"Shi, S., Chu, X., Li, B.: MG-WFBP: efficient data communication for distributed synchronous SGD algorithms. Proc. IEEE INFOCOM 2019, 172\u2013180 (2019)","journal-title":"Proc. IEEE INFOCOM"},{"key":"64_CR27","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, Q., Chu, X., Li, B., Qin, Y., Liu, R., Zhao, X.: Communication-efficient distributed deep learning with merged gradient sparsification on gpus. In: Proceedings of IEEE INFOCOM 2020 (2020)","DOI":"10.1109\/INFOCOM41043.2020.9155269"},{"key":"64_CR28","first-page":"353","volume":"2016","author":"Y Ukidave","year":"2016","unstructured":"Ukidave, Y., Li, X., Kaeli, D.: Mystic: predictive scheduling for Gpu based cloud servers using machine learning. Proc. IEEE IPDPS 2016, 353\u2013362 (2016)","journal-title":"Proc. IEEE IPDPS"},{"key":"64_CR29","first-page":"1","volume":"2020","author":"C Wang","year":"2020","unstructured":"Wang, C., Zhang, S., Chen, Y., Qian, Z., Wu, J., Xiao, M.: Joint configuration adaptation and bandwidth allocation for edge-based real-time video analytics. Proc. IEEE INFOCOM 2020, 1\u201310 (2020)","journal-title":"Proc. IEEE INFOCOM"},{"key":"64_CR30","unstructured":"Wang, Q., Shi, S., Wang, C., Chu, X. Communication Contention Aware Scheduling of Multiple Deep Learning Training Jobs. (2020b). arXiv preprint arXiv:200210105"},{"key":"64_CR31","first-page":"1678","volume":"2020","author":"S Wang","year":"2020","unstructured":"Wang, S., Li, D., Geng, J.: Geryon: accelerating distributed CNN training by network-level flow scheduling. Proc. IEEE INFOCOM 2020, 1678\u20131687 (2020)","journal-title":"Proc. IEEE INFOCOM"},{"key":"64_CR32","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1007\/s11036-016-0739-z","volume":"22","author":"F Xu","year":"2017","unstructured":"Xu, F., Ye, W., Liu, Y., Zhang, W.: Ufalloc: towards utility max-min fairness of bandwidth allocation for applications in datacenter networks. Mobile Netw. Appl. 22, 161\u2013173 (2017)","journal-title":"Mobile Netw. Appl."},{"key":"64_CR33","first-page":"181","volume":"2017","author":"H Zhang","year":"2017","unstructured":"Zhang, H., Zheng, Z., Xu, S., Dai, W., Ho, Q., Liang, X., Hu, Z., Wei, J., Xie, P., Xing, E.P.: Poseidon: an efficient communication architecture for distributed deep learning on GPU clusters. Proc. USENIX ATC 2017, 181\u2013193 (2017)","journal-title":"Proc. USENIX ATC"}],"container-title":["CCF Transactions on High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-021-00064-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42514-021-00064-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-021-00064-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,6,30]],"date-time":"2021-06-30T08:40:05Z","timestamp":1625042405000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42514-021-00064-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,3,2]]},"references-count":33,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2021,6]]}},"alternative-id":["64"],"URL":"https:\/\/doi.org\/10.1007\/s42514-021-00064-x","relation":{},"ISSN":["2524-4922","2524-4930"],"issn-type":[{"type":"print","value":"2524-4922"},{"type":"electronic","value":"2524-4930"}],"subject":[],"published":{"date-parts":[[2021,3,2]]},"assertion":[{"value":"31 August 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 February 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 March 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}