{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:29:24Z","timestamp":1787495364261,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_1","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:56Z","timestamp":1787492816000},"page":"3-17","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["ZIPPer: A Fine-Grained Co-design of ZeRO-DP and Pipeline Parallelism for LLM Training on Heterogeneous GPUs"],"prefix":"10.1007","author":[{"given":"Lang","family":"Yuan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linbo","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingxiang","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Shang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongsheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"1_CR1","doi-asserted-by":"crossref","unstructured":"Ansel, J., et al.: Pytorch 2: faster machine learning through dynamic python bytecode transformation and graph compilation. In: Proceedings of the 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 2, pp. 929\u2013947 (2024)","DOI":"10.1145\/3620665.3640366"},{"key":"1_CR2","unstructured":"Bai, J., et al.: Qwen technical report. arXiv:2309.16609 (2023)"},{"key":"1_CR3","unstructured":"Brown, T.B., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems (NeurIPS) (2020)"},{"key":"1_CR4","unstructured":"Chen, T., Xu, B., Zhang, C., Guestrin, C.: Training deep nets with sublinear memory cost. arXiv:1604.06174 (2016)"},{"issue":"240","key":"1_CR5","first-page":"1","volume":"24","author":"A Chowdhery","year":"2023","unstructured":"Chowdhery, A., et al.: PaLM: scaling language modeling with pathways. J. Mach. Learn. Res. 24(240), 1\u2013113 (2023)","journal-title":"J. Mach. Learn. Res."},{"key":"1_CR6","doi-asserted-by":"crossref","unstructured":"Dao, T., Fu, D.Y., Ermon, S., Rudra, A., R\u00e9, C.: FlashAttention: fast and memory-efficient exact attention with IO-awareness. In: Advances in Neural Information Processing Systems (NeurIPS) (2022)","DOI":"10.52202\/068431-1189"},{"key":"1_CR7","unstructured":"DeepSeek-AI: DeepSeek-R1: Incentivizing reasoning capability in LLMs via reinforcement learning. arXiv:2501.12948 (2025)"},{"key":"1_CR8","doi-asserted-by":"crossref","unstructured":"Fan, S., et al.: DAPPLE: a pipelined data parallel approach for training large models. In: PPoPP (2021)","DOI":"10.1145\/3437801.3441593"},{"key":"1_CR9","unstructured":"Guo, J., et al.: Adaptis: reducing pipeline bubbles with adaptive pipeline parallelism on heterogeneous models (2025). https:\/\/arxiv.org\/abs\/2509.23722"},{"key":"1_CR10","unstructured":"Harlap, A., et al.: PipeDream: fast and efficient pipeline parallel DNN training. arXiv:1806.03377 (2018)"},{"key":"1_CR11","unstructured":"Huang, Y., et al.: GPipe: efficient training of giant neural networks using pipeline parallelism. In: NeurIPS (2019)"},{"key":"1_CR12","unstructured":"Jiang, A.Q., et al.: Mistral 7B. arXiv:2310.06825 (2023)"},{"key":"1_CR13","unstructured":"Kim, T., Kim, H., Yu, G.I., Chun, B.G.: BPIPE: memory-balanced pipeline parallelism for training large language models. In: International Conference on Machine Learning (ICML) (2023)"},{"key":"1_CR14","unstructured":"Korthikanti, V.A., et al.: Reducing activation recomputation in large transformer models. In: Proceedings of Machine Learning and Systems (MLSys) (2023)"},{"key":"1_CR15","doi-asserted-by":"crossref","unstructured":"Li, S., Hoefler, T.: Chimera: efficiently training large-scale neural networks with bidirectional pipelines. In: SC (2021)","DOI":"10.1145\/3458817.3476145"},{"key":"1_CR16","doi-asserted-by":"publisher","unstructured":"Liang, P., Qiao, L., Lai, Z., Li, D.: Parallelsim: an accurate, generic, and efficient simulator for distributed deep learning. CCF Trans. High Perform. Comput. 8 (2026). https:\/\/doi.org\/10.1007\/s42514-025-00271-w","DOI":"10.1007\/s42514-025-00271-w"},{"key":"1_CR17","doi-asserted-by":"crossref","unstructured":"Liu, Z., Cheng, S., Zhou, H., You, Y.: Hanayo: harnessing wave-like pipeline parallelism for enhanced large model training efficiency. In: SC (2023)","DOI":"10.1145\/3581784.3607073"},{"key":"1_CR18","doi-asserted-by":"crossref","unstructured":"Narayanan, D., et al.: PipeDream: generalized pipeline parallelism for DNN training. In: SOSP (2019)","DOI":"10.1145\/3341301.3359646"},{"key":"1_CR19","unstructured":"Narayanan, D., Phanishayee, A., Shi, K., Chen, X., Zaharia, M.: Memory-efficient pipeline-parallel DNN training. In: ICML (2021)"},{"key":"1_CR20","doi-asserted-by":"crossref","unstructured":"Narayanan, D., et al.: Efficient large-scale language model training on GPU clusters using Megatron-LM. In: SC (2021)","DOI":"10.1145\/3458817.3476209"},{"key":"1_CR21","unstructured":"OpenAI: GPT-4 technical report. Technical report, OpenAI (2023). https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"1_CR22","unstructured":"Park, J.H., et al.: HetPipe: enabling large DNN training on (whimpy) heterogeneous GPU clusters through integration of pipelined model parallelism and data parallelism. In: USENIX ATC (2020)"},{"key":"1_CR23","unstructured":"Qi, P., Wan, X., Huang, G., Lin, M.: Zero bubble pipeline parallelism. In: ICLR (2024)"},{"key":"1_CR24","doi-asserted-by":"crossref","unstructured":"Rajbhandari, S., Rasley, J., Ruwase, O., He, Y.: ZeRO: memory optimizations toward training trillion parameter models. In: SC (2020)","DOI":"10.1109\/SC41405.2020.00024"},{"key":"1_CR25","doi-asserted-by":"crossref","unstructured":"Rajbhandari, S., Ruwase, O., Rasley, J., Smith, S., He, Y.: ZeRO-infinity: breaking the GPU memory wall for extreme scale deep learning. In: SC (2021)","DOI":"10.1145\/3458817.3476205"},{"key":"1_CR26","doi-asserted-by":"crossref","unstructured":"Rasley, J., Rajbhandari, S., Ruwase, O., He, Y.: DeepSpeed: system optimizations enable training deep learning models with over 100 billion parameters. In: KDD (2020)","DOI":"10.1145\/3394486.3406703"},{"key":"1_CR27","unstructured":"Ren, J., et al.: ZeRO-offload: democratizing billion-scale model training. In: USENIX ATC (2021)"},{"key":"1_CR28","unstructured":"Shoeybi, M., Patwary, M., Puri, R., LeGresley, P., Casper, J., Catanzaro, B.: Megatron-LM: training multi-billion parameter language models using model parallelism. arXiv:1909.08053 (2019)"},{"key":"1_CR29","doi-asserted-by":"publisher","unstructured":"Tang, Y., et al.: Koala: efficient pipeline training through automated schedule searching on domain-specific language. ACM Trans. Archit. Code Optim. 22(2) (2025). https:\/\/doi.org\/10.1145\/3722113","DOI":"10.1145\/3722113"},{"key":"1_CR30","unstructured":"Touvron, H., et al.: LLaMA: open and efficient foundation language models. arXiv:2302.13971 (2023)"},{"key":"1_CR31","unstructured":"Touvron, H., et al.: Llama 2: open foundation and fine-tuned chat models. arXiv:2307.09288 (2023)"},{"key":"1_CR32","unstructured":"Um, T., et al.: Metis: fast automatic distributed training on heterogeneous GPUs. In: USENIX ATC, pp. 563\u2013578 (2024)"},{"key":"1_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, S., Diao, L., Wu, C., Cao, Z., Wang, S., Lin, W.: HAP: SPMD DNN training on heterogeneous GPU clusters with automated program synthesis. In: EuroSys (2024)","DOI":"10.1145\/3627703.3629580"},{"key":"1_CR34","doi-asserted-by":"publisher","unstructured":"Zhang, W., Hu, Y., Shi, J., Bai, X.: Poplar: efficient scaling of distributed DNN training on heterogeneous GPU clusters. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, no. 21, pp. 22587\u201322595 (2025). https:\/\/doi.org\/10.1609\/aaai.v39i21.34417","DOI":"10.1609\/aaai.v39i21.34417"},{"key":"1_CR35","unstructured":"Zheng, L., et al.: Alpa: automating inter- and intra-operator parallelism for distributed deep learning. In: 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI) (2022)"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:47:00Z","timestamp":1787492820000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The technical solution and experimental design presented in this paper were independently completed by the authors. AI tools were used solely for language polishing and formatting optimization, and did not participate in the development of the research ideas or core content.","order":1,"name":"Ethics","label":"Statement on AI Usage","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}