{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T18:40:01Z","timestamp":1756492801970,"version":"3.44.0"},"publisher-location":"Singapore","reference-count":25,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819628292"},{"type":"electronic","value":"9789819628308"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-2830-8_12","type":"book-chapter","created":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T19:21:00Z","timestamp":1743362460000},"page":"146-158","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SpecInF: Exploiting Idle GPU Resources in\u00a0Distributed DL Training via\u00a0Speculative Inference Filling"],"prefix":"10.1007","author":[{"given":"Cunchi","family":"Lv","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiao","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenting","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaofang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,29]]},"reference":[{"key":"12_CR1","unstructured":"Clossal AI. Pytorch DDP (2024). https:\/\/github.com\/hpcaitech\/ColossalAI"},{"key":"12_CR2","unstructured":"Stability AI. Stable diffusion (2024). https:\/\/stability.ai\/"},{"key":"12_CR3","unstructured":"DeepSpeed. Hybrid parallelism (2024). https:\/\/www.deepspeed.ai\/tutorials\/pipeline\/"},{"key":"12_CR4","unstructured":"Tsinghua University\u00a0Knowledge Engineering and Data\u00a0Mining Group. Thudm chatglm3 (2024). https:\/\/github.com\/THUDM\/ChatGLM3"},{"key":"12_CR5","doi-asserted-by":"crossref","unstructured":"Gu, J., Zhu, Y., Wang, P., Chadha, M., Gerndt, M.: FaST-GShare: enabling efficient spatio-temporal GPU sharing in serverless computing for deep learning inference. In: Proceedings of the 52nd International Conference on Parallel Processing, pp. 635\u2013644 (2023)","DOI":"10.1145\/3605573.3605638"},{"key":"12_CR6","unstructured":"Gu, J., et al.: Tiresias: a $$\\{$$GPU$$\\}$$ cluster manager for distributed deep learning. In: 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 2019), pp. 485\u2013500 (2019)"},{"key":"12_CR7","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, H., Zhu, Y., Liu, Z., Guo, C., Wang, C.: Lyra: elastic scheduling for deep learning clusters. In: Proceedings of the Eighteenth European Conference on Computer Systems, pp. 835\u2013850 (2023)","DOI":"10.1145\/3552326.3587445"},{"key":"12_CR8","doi-asserted-by":"crossref","unstructured":"Li, S., et al.: PyTorch distributed: experiences on accelerating data parallel training. Proc. VLDB Endow. 13(12) (2020)","DOI":"10.14778\/3415478.3415530"},{"key":"12_CR9","unstructured":"Masters, D., Luschi, C.: Revisiting small batch training for deep neural networks. arXiv preprint arXiv:1804.07612 (2018)"},{"key":"12_CR10","unstructured":"Medium. All GPT-4 details (2024). https:\/\/openai.com\/chatgpt\/"},{"key":"12_CR11","unstructured":"Meta. Meta LLaMA2 (2024). https:\/\/llama.meta.com\/llama2\/"},{"key":"12_CR12","unstructured":"Meta. Pytorch DDP (2024). https:\/\/pytorch.org\/tutorials\/intermediate\/ddp_tutorial.html"},{"key":"12_CR13","unstructured":"Microsoft. Microsoft deepspeed (2024). https:\/\/github.com\/microsoft\/DeepSpeed"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Narayanan, D., et al.: PipeDream: generalized pipeline parallelism for DNN training. In: Proceedings of the 27th ACM Symposium on Operating Systems Principles, pp. 1\u201315 (2019)","DOI":"10.1145\/3341301.3359646"},{"key":"12_CR15","unstructured":"Nvidia. Industry AI (2024). https:\/\/www.nvidia.cn\/industries\/industrial\/"},{"key":"12_CR16","unstructured":"NVIDIA. Nvidia MIG (2024). https:\/\/www.nvidia.com\/en-us\/technologies\/multi-instance-gpu\/"},{"key":"12_CR17","unstructured":"NVIDIA. Nvidia MPS (2024). https:\/\/docs.nvidia.com\/deploy\/mps\/"},{"key":"12_CR18","unstructured":"NVIDIA. NVML library (2024). https:\/\/developer.nvidia.com\/management-library-nvml"},{"key":"12_CR19","unstructured":"OpenAI. OpenAI ChatGPT (2024). https:\/\/openai.com\/chatgpt\/"},{"key":"12_CR20","unstructured":"Park, S.J., Fried, J., Kim, S., Alizadeh, M., Belay, A.: Efficient strong scaling through burst parallel training. Proc. Mach. Learn. Syst. 4, 748\u2013761 (2022)"},{"key":"12_CR21","unstructured":"Shoeybi, M., Patwary, M., Puri, R., LeGresley, P., Casper, J., Catanzaro, B.: Megatron-LM: training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)"},{"key":"12_CR22","doi-asserted-by":"crossref","unstructured":"Strati, F., Ma, X., Klimovic, A.: Orion: interference-aware, fine-grained GPU sharing for ml applications. In: Proceedings of the Nineteenth European Conference on Computer Systems, pp. 1075\u20131092 (2024)","DOI":"10.1145\/3627703.3629578"},{"key":"12_CR23","unstructured":"Wu, B., Zhang, Z., Bai, Z., Liu, X., Jin, X.: Transparent $$\\{$$GPU$$\\}$$ sharing in container clouds for deep learning workloads. In: 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 2023), pp. 69\u201385 (2023)"},{"key":"12_CR24","unstructured":"Xiao, W., et al.: $$\\{$$AntMan$$\\}$$: dynamic scaling on $$\\{$$GPU$$\\}$$ clusters for deep learning. In: 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 2020), pp. 533\u2013548 (2020)"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Yang, Y., et al.: INFless: a native serverless system for low-latency, high-throughput inference. In: Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 768\u2013781 (2022)","DOI":"10.1145\/3503222.3507709"}],"container-title":["Lecture Notes in Computer Science","Network and Parallel Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-2830-8_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T18:14:29Z","timestamp":1756491269000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-2830-8_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819628292","9789819628308"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-2830-8_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"29 March 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NPC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"IFIP International Conference on Network and Parallel Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Haikou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"npc2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}