{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:29:21Z","timestamp":1787495361543,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":28,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_18","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:20Z","timestamp":1787492780000},"page":"269-281","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["C$$^{3}$$FT: Computation-Centric Checkpointing for\u00a0Distributed Large Model Training"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2331-0108","authenticated-orcid":false,"given":"Yonghua","family":"Huang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baodong","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0022-7865","authenticated-orcid":false,"given":"Shigang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiahao","family":"Ding","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rongtian","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingping","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Boxun","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenhua","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"18_CR1","unstructured":"Bloom chronicles (2022). https:\/\/github.com\/bigscience-workshop\/bigscience\/blob\/master\/train\/tr11-176B-ml\/chronicles.md"},{"key":"18_CR2","unstructured":"Aaron\u00a0Gokaslan, Vanya\u00a0Cohen, E.P., Tellex, S.: Openwebtext corpus (2019). http:\/\/Skylion007.github.io\/OpenWebTextCorpus"},{"key":"18_CR3","doi-asserted-by":"crossref","unstructured":"Cai, W., Qin, L., Huang, J.: Moc-system: efficient fault tolerance for sparse mixture-of-experts model training. In: Proceedings of the 30th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 2, pp. 655\u2013671. Association for Computing Machinery (2025)","DOI":"10.1145\/3676641.3716006"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Chen, M., Hua, Y., Bai, R., Huang, J.: A cost-efficient failure-tolerant scheme for distributed DNN training. In: 2023 IEEE 41st International Conference on Computer Design, pp. 150\u2013157. IEEE (2023)","DOI":"10.1109\/ICCD58817.2023.00031"},{"issue":"240","key":"18_CR5","first-page":"1","volume":"24","author":"A Chowdhery","year":"2023","unstructured":"Chowdhery, A., et al.: Palm: Scaling language modeling with pathways. J. Mach. Learn. Res. 24(240), 1\u2013113 (2023)","journal-title":"J. Mach. Learn. Res."},{"key":"18_CR6","unstructured":"Dubey, A., et\u00a0al.: The llama 3 herd of models. arXiv e-prints pp. arXiv\u20132407 (2024)"},{"key":"18_CR7","unstructured":"Eisenman, A., et al.: $$\\{$$Check-N-Run$$\\}$$: a checkpointing system for training deep learning recommendation models. In: 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22), pp. 929\u2013943 (2022)"},{"key":"18_CR8","unstructured":"Huang, Y., et\u00a0al.: Gpipe: efficient training of giant neural networks using pipeline parallelism. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"18_CR9","unstructured":"Jia, X., et\u00a0al.: Whale: efficient giant model training over heterogeneous $$\\{$$GPUs$$\\}$$. In: 2022 USENIX Annual Technical Conference (USENIX ATC 22), pp. 673\u2013688 (2022)"},{"key":"18_CR10","unstructured":"Jiang, A.Q., et\u00a0al.: Mixtral of experts. arXiv preprint arXiv:2401.04088 (2024)"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Kokolis, A., et al.: Revisiting reliability in large-scale machine learning research clusters. In: 2025 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 1259\u20131274. IEEE (2025)","DOI":"10.1109\/HPCA61900.2025.00096"},{"key":"18_CR12","unstructured":"Li, S., Xue, F., Baranwal, C., Li, Y., You, Y.: Sequence parallelism: long sequence training from system perspective. arXiv preprint arXiv:2105.13120 (2021)"},{"key":"18_CR13","unstructured":"Liu, A., et al.: Deepseek-v3 technical report. arXiv:2412.19437 arXiv preprint (2024)"},{"key":"18_CR14","first-page":"637","volume":"3","author":"K Maeng","year":"2021","unstructured":"Maeng, K., et al.: Understanding and improving failure tolerant training for deep learning recommendation with partial recovery. Proc. Mach. Learn. Syst. 3, 637\u2013651 (2021)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"18_CR15","unstructured":"Mai, L., Li, G., Wagenl\u00e4nder, M., Fertakis, K., Brabete, A.O., Pietzuch, P.: $$\\{$$KungFu$$\\}$$: making training in distributed machine learning adaptive. In: 14th USENIX Symposium on Operating Systems Design and Implementation, pp. 937\u2013954 (2020)"},{"key":"18_CR16","unstructured":"Micikevicius, P., et al.: Mixed precision training. arXiv:1710.03740 arXiv preprint (2017)"},{"key":"18_CR17","unstructured":"Mnih, V., et al.: Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning, pp. 1928\u20131937 (2016)"},{"key":"18_CR18","unstructured":"Mohan, J., Phanishayee, A., Chidambaram, V.: $$\\{$$CheckFreq$$\\}$$: frequent,$$\\{$$Fine-Grained$$\\}$$$$\\{$$DNN$$\\}$$ checkpointing. In: 19th USENIX Conference on File and Storage Technologies, pp. 203\u2013216 (2021)"},{"issue":"8","key":"18_CR19","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI blog 1(8), 9 (2019)","journal-title":"OpenAI blog"},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Rajbhandari, S., Rasley, J., Ruwase, O., He, Y.: Zero: memory optimizations toward training trillion parameter models. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201316 (2020)","DOI":"10.1109\/SC41405.2020.00024"},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Rasley, J., Rajbhandari, S., Ruwase, O., He, Y.: Deepspeed: system optimizations enable training deep learning models with over 100 billion parameters. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 3505\u20133506 (2020)","DOI":"10.1145\/3394486.3406703"},{"key":"18_CR22","unstructured":"Ren, J., et al.: $$\\{$$Zero-offload$$\\}$$: democratizing $$\\{$$billion-scale$$\\}$$ model training. In: 2021 USENIX Annual Technical Conference (USENIX ATC 21), pp. 551\u2013564 (2021)"},{"key":"18_CR23","unstructured":"Shoeybi, M., Patwary, M., Puri, R., LeGresley, P., Casper, J., Catanzaro, B.: Megatron-LM: training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)"},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Wang, Z., et al.: Gemini: fast failure recovery in distributed training with in-memory checkpoints. In: Proceedings of the 29th Symposium on Operating Systems Principles, pp. 364\u2013381 (2023)","DOI":"10.1145\/3600006.3613145"},{"key":"18_CR25","unstructured":"Weng, Q., et al.: $$\\{$$MLaaS$$\\}$$ in the wild: workload analysis and scheduling in $$\\{$$Large-Scale$$\\}$$ heterogeneous $$\\{$$GPU$$\\}$$ clusters. In: 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22), pp. 945\u2013960 (2022)"},{"key":"18_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, S., Zhang, C., You, Z., Zheng, R., Xu, B.: Asynchronous stochastic gradient descent for DNN training. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing. pp. 6660\u20136663. IEEE (2013)","DOI":"10.1109\/ICASSP.2013.6638950"},{"key":"18_CR27","unstructured":"Zhang, S., et\u00a0al.: Opt: open pre-trained transformer language models. arXiv preprint arXiv:2205.01068 (2022)"},{"key":"18_CR28","doi-asserted-by":"crossref","unstructured":"Zhong, Y., Sheng, G., Liu, J., Yuan, J., Wu, C.: Swift: expedited failure recovery for large-scale DNN training. In: Proceedings of the 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming, pp. 447\u2013449 (2023)","DOI":"10.1145\/3572848.3577510"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:25Z","timestamp":1787492785000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}