{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T05:53:23Z","timestamp":1781848403475,"version":"3.54.5"},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,24]]},"DOI":"10.1109\/iscas66217.2026.11562076","type":"proceedings-article","created":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T20:06:41Z","timestamp":1781813201000},"page":"1749-1753","source":"Crossref","is-referenced-by-count":0,"title":["HOC: Hierarchical Overlapped Communication Optimization for Parallelism in Distributed Training"],"prefix":"10.1109","author":[{"given":"Zeyue","family":"Wang","sequence":"first","affiliation":[{"name":"Research Center for High Efficiency Computing Infrastructure,Zhejiang Lab,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuanyuan","family":"Wang","sequence":"additional","affiliation":[{"name":"Research Center for High Efficiency Computing Infrastructure,Zhejiang Lab,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shu","family":"Pan","sequence":"additional","affiliation":[{"name":"Research Center for High Efficiency Computing Infrastructure,Zhejiang Lab,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuyang","family":"Wang","sequence":"additional","affiliation":[{"name":"Research Center for High Efficiency Computing Infrastructure,Zhejiang Lab,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nana","family":"Tang","sequence":"additional","affiliation":[{"name":"Research Center for High Efficiency Computing Infrastructure,Zhejiang Lab,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fei","family":"Yang","sequence":"additional","affiliation":[{"name":"Research Center for High Efficiency Computing Infrastructure,Zhejiang Lab,Hangzhou,China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"265","article-title":"TensorFlow: A system for Large-Scale machine learning","volume-title":"12th USENIX Symposium on Operating Systems Design and Implementation (OSDI 16)","author":"Abadi"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2025.3554028"},{"key":"ref3","article-title":"Flux: Fast software-based communication overlap on gpus through kernel fusion","author":"Chang","year":"2024"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651379"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/s44336-026-00038-z"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS64566.2025.00089"},{"key":"ref7","first-page":"103","volume-title":"GPipe: efficient training of giant neural networks using pipeline parallelism","author":"Huang","year":"2019"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3663408.3663409"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"ref10","article-title":"Nvidia nvswitch: The world\u2019s highest-bandwidth on-node switch","year":"2018"},{"key":"ref11","volume-title":"PyTorch: an imperative style, high-performance deep learning library","author":"Paszke","year":"2019"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref13","article-title":"Horovod: fast and easy distributed deep learning in tensorflow","author":"Sergeev","year":"2018"},{"key":"ref14","article-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2020"},{"key":"ref15","article-title":"Pangu ultra moe: How to train your big moe on ascend npus","author":"Tang","year":"2025"},{"key":"ref16","article-title":"Revisiting the time cost model of allreduce","author":"Xiong","year":"2024"},{"key":"ref17","article-title":"Comet: Fine-grained computation-communication overlapping for mixture-of-experts","author":"Zhang","year":"2025"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.14778\/3561261.3561265"},{"key":"ref19","article-title":"Deepep: an efficient expert-parallel communication library","author":"Zhao","year":"2025"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.14778\/3611540.3611569"}],"event":{"name":"2026 IEEE International Symposium on Circuits and Systems (ISCAS)","location":"Shanghai, China","start":{"date-parts":[[2026,5,24]]},"end":{"date-parts":[[2026,5,28]]}},"container-title":["2026 IEEE International Symposium on Circuits and Systems (ISCAS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11561899\/11561804\/11562076.pdf?arnumber=11562076","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T05:33:18Z","timestamp":1781847198000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11562076\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,24]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/iscas66217.2026.11562076","relation":{},"subject":[],"published":{"date-parts":[[2026,5,24]]}}}