{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T06:02:33Z","timestamp":1730268153936,"version":"3.28.0"},"reference-count":43,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,20]],"date-time":"2024-05-20T00:00:00Z","timestamp":1716163200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,20]],"date-time":"2024-05-20T00:00:00Z","timestamp":1716163200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004735","name":"Natural Science Foundation of Hunan Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004735","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,20]]},"DOI":"10.1109\/infocom52122.2024.10621250","type":"proceedings-article","created":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T17:25:41Z","timestamp":1723483541000},"page":"1731-1740","source":"Crossref","is-referenced-by-count":1,"title":["Gsyn: Reducing Staleness and Communication Waiting via Grouping-based Synchronization for Distributed Deep Learning"],"prefix":"10.1109","author":[{"given":"Yijun","family":"Li","sequence":"first","affiliation":[{"name":"Central South University,School of Information Science and Engineering,Changsha,China,410083"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiawei","family":"Huang","sequence":"additional","affiliation":[{"name":"Central South University,School of Information Science and Engineering,Changsha,China,410083"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaoyi","family":"Li","sequence":"additional","affiliation":[{"name":"Central South University,School of Information Science and Engineering,Changsha,China,410083"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingling","family":"Liu","sequence":"additional","affiliation":[{"name":"Central South University,School of Information Science and Engineering,Changsha,China,410083"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengwen","family":"Zhou","sequence":"additional","affiliation":[{"name":"Central South University,School of Information Science and Engineering,Changsha,China,410083"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wanchun","family":"Jiang","sequence":"additional","affiliation":[{"name":"Central South University,School of Information Science and Engineering,Changsha,China,410083"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianxin","family":"Wang","sequence":"additional","affiliation":[{"name":"Central South University,School of Information Science and Engineering,Changsha,China,410083"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref3","first-page":"2493","article-title":"Natural language processing (almost) from scratch","volume":"12","author":"Collobert","year":"2011","journal-title":"Journal of machine learning research"},{"key":"ref4","first-page":"132","article-title":"Priority-based parameter propagation for distributed dnn training","volume-title":"Proc. MLSys","author":"Jayarajan"},{"key":"ref5","first-page":"1223","article-title":"Large scale distributed deep networks","volume":"25","author":"Dean","year":"2012","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.5555\/3026877.3026899"},{"key":"ref7","first-page":"741","article-title":"Atp: In-network aggregation for multi-tenant learning","volume-title":"Proc. USENIX NSDI","author":"Lao"},{"key":"ref8","first-page":"19","article-title":"Communication efficient distributed machine learning with the parameter server","volume":"27","author":"Li","year":"2014","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref9","first-page":"82","article-title":"Plink: Discovering and exploiting locality for accelerated distributed training on the public cloud","volume-title":"Proc. MLSys","author":"Luo"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2020.3040601"},{"article-title":"Accurate, large minibatch sgd: Training imagenet in 1 hour","year":"2017","author":"Goyal","key":"ref11"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/1327452.1327492"},{"key":"ref13","first-page":"785","article-title":"Scaling distributed machine learning with in-network aggregation","volume-title":"Proc. USENIX NSDI","author":"Sapio"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359642"},{"key":"ref15","first-page":"181","article-title":"Poseidon: An efficient communication architecture for distributed deep learning on gpu clusters","volume-title":"Proc. USENIX ATC","author":"Zhang"},{"article-title":"Hogwild!: A lock-free approach to parallelizing stochastic gradient descent","year":"2011","author":"Niu","key":"ref16"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.14778\/3503585.3503590"},{"key":"ref18","first-page":"571","article-title":"Project adam: Building an efficient and scalable deep learning training system","volume-title":"Proc. USENIX OSDI","author":"Chilimbi"},{"key":"ref19","first-page":"2331","article-title":"Slow learners are fast","volume-title":"Proc. NIPS","author":"Zinkevich"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/2901318.2901323"},{"issue":"1","key":"ref21","first-page":"1","article-title":"Gssp: Eliminating stragglers through grouping synchronous for distributed deep learning in heterogeneous cluster","author":"Sun","year":"2021","journal-title":"IEEE Transactions on Cloud Computing"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3411029.3411037"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/3229543.3229544"},{"article-title":"Fashion-mnist: a novel image dataset for benchmarking machine learning algorithms","year":"2017","author":"Xiao","key":"ref24"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref26","first-page":"7184","article-title":"On the linear speedup analysis of communication efficient momentum sgd for distributed non-convex optimization","volume-title":"Proc. ICML","author":"Yu"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1137\/16m1080173"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488815"},{"key":"ref29","first-page":"2350","article-title":"Staleness-aware async-sgd for distributed deep learning","volume-title":"Proc. IJCAI","author":"Zhang"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3511989"},{"key":"ref31","article-title":"Lipschitz regularity of deep neural networks: analysis and efficient estimation","volume":"31","author":"Virmaux","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref32","first-page":"8026","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume-title":"Proc. NIPS","author":"Paszke"},{"article-title":"Very deep convolutional networks for large-scale image recognition","year":"2014","author":"Simonyan","key":"ref33"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"article-title":"Learning multiple layers of features from tiny images","year":"2009","author":"Krizhevsky","key":"ref35"},{"key":"ref36","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","volume-title":"Proc. NIPS","author":"Sutskever"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W16-3210"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM48880.2022.9796688"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737587"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3544216.3544262"},{"key":"ref41","first-page":"463","article-title":"A unified architecture for accelerating distributed dnn training in heterogeneous gpu\/cpu clusters","volume-title":"Proc. USENIX OSDI","author":"Jiang"},{"key":"ref42","first-page":"1223","article-title":"More effective distributed ml via a stale synchronous parallel parameter server","volume-title":"Proc. NIPS","author":"Ho"},{"key":"ref43","first-page":"4243","article-title":"Bml: a high-performance, low-cost gradient synchronization algorithm for dml training","volume-title":"Proc. NIPS","author":"Wang"}],"event":{"name":"IEEE INFOCOM 2024 - IEEE Conference on Computer Communications","start":{"date-parts":[[2024,5,20]]},"location":"Vancouver, BC, Canada","end":{"date-parts":[[2024,5,23]]}},"container-title":["IEEE INFOCOM 2024 - IEEE Conference on Computer Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10621050\/10621073\/10621250.pdf?arnumber=10621250","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,13]],"date-time":"2024-08-13T05:42:08Z","timestamp":1723527728000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10621250\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,20]]},"references-count":43,"URL":"https:\/\/doi.org\/10.1109\/infocom52122.2024.10621250","relation":{},"subject":[],"published":{"date-parts":[[2024,5,20]]}}}