{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,12]],"date-time":"2026-01-12T21:51:21Z","timestamp":1768254681480,"version":"3.49.0"},"reference-count":39,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2025,2,1]],"date-time":"2025-02-01T00:00:00Z","timestamp":1738368000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,2,1]],"date-time":"2025-02-01T00:00:00Z","timestamp":1738368000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,2,1]],"date-time":"2025-02-01T00:00:00Z","timestamp":1738368000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2018AAA0103203"],"award-info":[{"award-number":["2018AAA0103203"]}]},{"name":"Project of Key Research and Development Program of Sichuan Province","award":["2021YFG0325"],"award-info":[{"award-number":["2021YFG0325"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Parallel Distrib. Syst."],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1109\/tpds.2024.3515804","type":"journal-article","created":{"date-parts":[[2024,12,11]],"date-time":"2024-12-11T22:44:55Z","timestamp":1733957095000},"page":"293-307","source":"Crossref","is-referenced-by-count":3,"title":["UMPIPE: Unequal Microbatches-Based Pipeline Parallelism for Deep Neural Network Training"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0809-5799","authenticated-orcid":false,"given":"Guangyao","family":"Zhou","sequence":"first","affiliation":[{"name":"School of Computing and Artificial Intelligence, Southwest Jiaotong University, Chengdu, Sichuan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5551-9796","authenticated-orcid":false,"given":"Wenhong","family":"Tian","sequence":"additional","affiliation":[{"name":"School of Information and Software Engineering, University of Electronic Science and Technology of China, Chengdu, Sichuan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9754-6496","authenticated-orcid":false,"given":"Rajkumar","family":"Buyya","sequence":"additional","affiliation":[{"name":"Department of Computing and Information Systems, Cloud Computing and Distributed Systems (CLOUDS) Laboratory, University of Melbourne, Melbourne, VIC, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2069-0032","authenticated-orcid":false,"given":"Kui","family":"Wu","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Victoria, Victoria, BC, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Very deep convolutional networks for large-scale image recognition","volume-title":"arXiv:1409.1556","author":"Simonyan"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00358"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3529755"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/s11227-021-03746-z"},{"key":"ref5","first-page":"58:1","article-title":"Efficient large-scale language model training on GPU clusters using Megatron-LM","volume-title":"Proc. Int. Conf. High Perform. Comput. Netw. Storage Anal.","author":"Narayanan"},{"key":"ref6","first-page":"6543","article-title":"TeraPipe: Token-level pipeline parallelism for training large-scale language models","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527391"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2021.3065985"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2021.3111624"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2021.06.021"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3472456.3472497"},{"key":"ref12","first-page":"1027","article-title":"Accelerating collective communication in data parallel training across deep learning frameworks","volume-title":"Proc. 19th USENIX Symp. Netw. Syst. Des. Implementation","author":"Romero"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2023.3247001"},{"key":"ref14","first-page":"103","article-title":"GPipe: Efficient training of giant neural networks using pipeline parallelism","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Huang"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441593"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3302424.3303953"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2023.3247883"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2020.11.005"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/GLOBECOM46510.2021.9685964"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS49936.2021.00111"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1186\/s13677-022-00382-7"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/s10723-021-09550-6"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00056"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-69766-1_20"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586300"},{"key":"ref27","first-page":"559","article-title":"Alpa: Automating inter- and intra-operator parallelism for distributed deep learning","volume-title":"Proc. 16th USENIX Symp. Operating Syst. Des. Implementation","author":"Zheng"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2021.10.072"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/LCOMM.2020.3041453"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2020.09.029"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/3431379.3460644"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.126661"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2021.09.007"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3138862"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3138825"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3094364"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1201\/9780203713402"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/4235.996017"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2023.110027"}],"container-title":["IEEE Transactions on Parallel and Distributed Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/71\/10795769\/10792656.pdf?arnumber=10792656","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,28]],"date-time":"2024-12-28T06:20:31Z","timestamp":1735366831000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10792656\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2]]},"references-count":39,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/tpds.2024.3515804","relation":{},"ISSN":["1045-9219","1558-2183","2161-9883"],"issn-type":[{"value":"1045-9219","type":"print"},{"value":"1558-2183","type":"electronic"},{"value":"2161-9883","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2]]}}}