{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T05:22:02Z","timestamp":1743139322175,"version":"3.40.3"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031697654"},{"type":"electronic","value":"9783031697661"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-69766-1_20","type":"book-chapter","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T19:02:05Z","timestamp":1724612525000},"page":"288-301","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["CSIMD: Cross-Search Algorithm with\u00a0Improved Multi-dimensional Dichotomy for\u00a0Micro-Batch-Based Pipeline Parallel Training in\u00a0DNN"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0809-5799","authenticated-orcid":false,"given":"Guangyao","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haocheng","family":"Lan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuanlun","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenhong","family":"Tian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiahong","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Teng","family":"Su","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,8,26]]},"reference":[{"key":"20_CR1","doi-asserted-by":"crossref","unstructured":"Bai, Y., Li, C., Zhou, Q., et\u00a0al.: Gradient compression supercharged high-performance data parallel DNN training. In: Proceedings of the ACM SIGOPS 28th Symposium on Operating Systems Principles, pp. 359\u2013375 (2021)","DOI":"10.1145\/3477132.3483553"},{"key":"20_CR2","doi-asserted-by":"crossref","unstructured":"Beaumont, O., Eyraud-Dubois, L., Shilova, A.: Madpipe: memory aware dynamic programming algorithm for pipelined model parallelism. In: International Parallel and Distributed Processing Symposium Workshops, pp. 1063\u20131073. IEEE (2022)","DOI":"10.1109\/IPDPSW55747.2022.00174"},{"key":"20_CR3","doi-asserted-by":"crossref","unstructured":"Dong, J., Cao, Z., Zhang, T., et\u00a0al.: Eflops: algorithm and system co-design for a high performance distributed training platform. In: International Symposium on High Performance Computer Architecture (HPCA), pp. 610\u2013622. IEEE (2020)","DOI":"10.1109\/HPCA47549.2020.00056"},{"key":"20_CR4","doi-asserted-by":"crossref","unstructured":"Dryden, N., Maruyama, N., Moon, T., et\u00a0al.: Channel and filter parallelism for large-scale CNN training. In: Taufer, M., Balaji, P., Pe\u00f1a, A.J. (eds.) Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 10:1\u201310:20. ACM (2019)","DOI":"10.1145\/3295500.3356207"},{"key":"20_CR5","doi-asserted-by":"crossref","unstructured":"Elango, V.: Pase: parallelization strategies for efficient DNN training. In: 35th IEEE International Parallel and Distributed Processing Symposium, IPDPS, pp. 1025\u20131034. IEEE (2021)","DOI":"10.1109\/IPDPS49936.2021.00111"},{"key":"20_CR6","doi-asserted-by":"crossref","unstructured":"Fan, S., Rong, Y., Meng, C., et\u00a0al.: DAPPLE: a pipelined data parallel approach for training large models. In: Proceedings of the 26th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 431\u2013445. ACM (2021)","DOI":"10.1145\/3437801.3441593"},{"issue":"11","key":"20_CR7","doi-asserted-by":"publisher","first-page":"12741","DOI":"10.1007\/s11227-021-03746-z","volume":"77","author":"H Fu","year":"2021","unstructured":"Fu, H., Tang, S., He, B., et al.: HGP4CNN: an efficient parallelization framework for training convolutional neural networks on modern GPUs. J. Supercomput. 77(11), 12741\u201312770 (2021)","journal-title":"J. Supercomput."},{"issue":"11","key":"20_CR8","first-page":"2808","volume":"33","author":"R Gu","year":"2022","unstructured":"Gu, R., Chen, Y., Liu, S., et al.: Liquid: intelligent resource estimation and network-efficient scheduling for deep learning jobs on distributed GPU clusters. IEEE Trans. Parallel Distrib. Syst. 33(11), 2808\u20132820 (2022)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"20_CR9","doi-asserted-by":"publisher","first-page":"382","DOI":"10.1016\/j.ins.2021.10.072","volume":"585","author":"Z Han","year":"2022","unstructured":"Han, Z., Qu, G., Liu, B., Zhang, F.: Exploit the data level parallelism and schedule dependent tasks on the multi-core processors. Inf. Sci. 585, 382\u2013394 (2022)","journal-title":"Inf. Sci."},{"key":"20_CR10","doi-asserted-by":"crossref","unstructured":"Lee, Y., Chung, J., Rhu, M.: Smartsage: training large-scale graph neural networks using in-storage processing architectures. In: Proceedings of the 49th Annual International Symposium on Computer Architecture, pp. 932\u2013945. ACM (2022)","DOI":"10.1145\/3470496.3527391"},{"key":"20_CR11","doi-asserted-by":"crossref","unstructured":"Li, J., Wang, Y., Zhang, J., et\u00a0al.: Pipepar: a pipelined hybrid parallel approach for accelerating distributed DNN training. In: 2021 IEEE 24th International Conference on Computer Supported Cooperative Work in Design (CSCWD), pp. 470\u2013475. IEEE (2021)","DOI":"10.1109\/CSCWD49262.2021.9437625"},{"issue":"4","key":"20_CR12","doi-asserted-by":"publisher","first-page":"1241","DOI":"10.1109\/LCOMM.2020.3041453","volume":"25","author":"Y Li","year":"2021","unstructured":"Li, Y., Zeng, Z., Li, J., et al.: Distributed model training based on data parallelism in edge computing-enabled elastic optical networks. IEEE Commun. Lett. 25(4), 1241\u20131244 (2021)","journal-title":"IEEE Commun. Lett."},{"key":"20_CR13","doi-asserted-by":"publisher","first-page":"206","DOI":"10.1016\/j.future.2021.06.021","volume":"125","author":"Z Li","year":"2021","unstructured":"Li, Z., Chang, V., Hu, H., et al.: Optimizing makespan and resource utilization for multi-DNN training in GPU cluster. Future Gener. Comput. Syst. 125, 206\u2013220 (2021)","journal-title":"Future Gener. Comput. Syst."},{"key":"20_CR14","unstructured":"Li, Z., Zhuang, S., Guo, S., et\u00a0al.: Terapipe: token-level pipeline parallelism for training large-scale language models. In: International Conference on Machine Learning, pp. 6543\u20136552. PMLR (2021)"},{"key":"20_CR15","doi-asserted-by":"crossref","unstructured":"Md, V., Misra, S., Ma, G., et\u00a0al.: Distgnn: scalable distributed training for large-scale graph neural networks. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201314 (2021)","DOI":"10.1145\/3458817.3480856"},{"key":"20_CR16","doi-asserted-by":"crossref","unstructured":"Narayanan, D., Harlap, A., Phanishayee, A., et\u00a0al.: Pipedream: generalized pipeline parallelism for DNN training. In: Proceedings of the 27th ACM Symposium on Operating Systems Principles, pp. 1\u201315 (2019)","DOI":"10.1145\/3341301.3359646"},{"key":"20_CR17","unstructured":"Narayanan, D., Phanishayee, A., Shi, K., et\u00a0al.: Memory-efficient pipeline-parallel DNN training. In: International Conference on Machine Learning, pp. 7937\u20137947. PMLR (2021)"},{"key":"20_CR18","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1016\/j.jpdc.2020.11.005","volume":"149","author":"S Ouyang","year":"2021","unstructured":"Ouyang, S., Dong, D., Xu, Y., Xiao, L.: Communication optimization strategies for distributed deep neural network training: a survey. J. Parallel Distrib. Comput. 149, 52\u201365 (2021)","journal-title":"J. Parallel Distrib. Comput."},{"key":"20_CR19","unstructured":"Romero, J., Yin, J., Laanait, N., et\u00a0al.: Accelerating collective communication in data parallel training across deep learning frameworks. In: 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 2022), pp. 1027\u20131040 (2022)"},{"issue":"2","key":"20_CR20","first-page":"495","volume":"8","author":"P Sun","year":"2022","unstructured":"Sun, P., Wen, Y., Han, R., et al.: Gradientflow: optimizing network performance for large-scale distributed DNN training. IEEE Trans. Big Data 8(2), 495\u2013507 (2022)","journal-title":"IEEE Trans. Big Data"},{"key":"20_CR21","unstructured":"Wallach, H.M., Larochelle, H., Beygelzimer, A., et\u00a0al.: GPipe: efficient training of giant neural networks using pipeline parallelism. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"issue":"2","key":"20_CR22","doi-asserted-by":"publisher","first-page":"939","DOI":"10.1109\/JIOT.2021.3111624","volume":"9","author":"H Wang","year":"2022","unstructured":"Wang, H., Qu, Z., Zhou, Q., et al.: A comprehensive survey on training acceleration for large machine learning models in IoT. IEEE Internet Things J. 9(2), 939\u2013963 (2022)","journal-title":"IEEE Internet Things J."},{"issue":"1","key":"20_CR23","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1007\/s10723-021-09550-6","volume":"19","author":"J Xu","year":"2021","unstructured":"Xu, J., Wang, J., Qi, Q., et al.: Effective scheduler for distributed DNN training based on mapreduce and GPU cluster. J. Grid Comput. 19(1), 8 (2021)","journal-title":"J. Grid Comput."},{"key":"20_CR24","doi-asserted-by":"crossref","unstructured":"Ye, X., Lai, Z., Li, S., et\u00a0al.: Hippie: a data-paralleled pipeline approach to improve memory-efficiency and scalability for large DNN training. In: Proceedings of the 50th International Conference on Parallel Processing, pp. 71:1\u201371:10. ACM (2021)","DOI":"10.1145\/3472456.3472497"},{"key":"20_CR25","doi-asserted-by":"crossref","unstructured":"Zeng, Z., Liu, C., Tang, Z., et\u00a0al.: Training acceleration for deep neural networks: a hybrid parallelization strategy. In: 2021 58th ACM\/IEEE Design Automation Conference (DAC), pp. 1165\u20131170. IEEE (2021)","DOI":"10.1109\/DAC18074.2021.9586300"},{"key":"20_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Chen, J., Hu, B.: The optimization of model parallelization strategies for multi-GPU training. In: IEEE Global Communications Conference, GLOBECOM 2021, Madrid, Spain, 70\u201311 December 2021, pp.\u00a01\u20136. IEEE (2021)","DOI":"10.1109\/GLOBECOM46510.2021.9685964"},{"key":"20_CR27","doi-asserted-by":"crossref","unstructured":"Zhao, S., Li, F., Chen, X., et\u00a0al.: Naspipe: high performance and reproducible pipeline parallel supernet training via causal synchronous parallelism. In: Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 374\u2013387 (2022)","DOI":"10.1145\/3503222.3507735"},{"issue":"3","key":"20_CR28","doi-asserted-by":"publisher","first-page":"489","DOI":"10.1109\/TPDS.2021.3094364","volume":"33","author":"S Zhao","year":"2022","unstructured":"Zhao, S., Li, F., Chen, X., et al.: vPipe: a virtualized acceleration system for achieving efficient and scalable pipeline parallel DNN training. IEEE Trans. Parallel Distrib. Syst. 33(3), 489\u2013506 (2022)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"20_CR29","unstructured":"Zheng, L., Li, Z., Zhang, H., et\u00a0al.: Alpa: automating inter-and $$\\{$$Intra-Operator$$\\}$$ parallelism for distributed deep learning. In: 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 2022), pp. 559\u2013578 (2022)"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2024: Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-69766-1_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T19:10:31Z","timestamp":1724613031000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-69766-1_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031697654","9783031697661"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-69766-1_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"26 August 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Madrid","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Spain","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 August 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 August 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2024.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}