{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T13:24:42Z","timestamp":1730208282612,"version":"3.28.0"},"reference-count":43,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,9]]},"DOI":"10.1109\/cluster.2019.8891038","type":"proceedings-article","created":{"date-parts":[[2019,11,13]],"date-time":"2019-11-13T18:00:53Z","timestamp":1573668053000},"page":"1-12","source":"Crossref","is-referenced-by-count":2,"title":["FluentPS: A Parameter Server Design with Low-frequency Synchronization for Distributed Deep Learning"],"prefix":"10.1109","author":[{"given":"Xin","family":"Yao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueyu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cho-Li","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"journal-title":"Large batch training of convolutional networks","year":"2017","author":"you","key":"ref39"},{"journal-title":"Nvidia caffe (nvidia corporation &#x00A9;2017) is an nvidia-maintained fork of bvlc caffe tuned for nvidia gpus","year":"0","key":"ref38"},{"key":"ref33","first-page":"1097","article-title":"Imagenet classification with deep convolutional neural networks","author":"krizhevsky","year":"2012","journal-title":"Proceedings of the 25th International Conference on Neural Information Processing Systems - Volume 1 ser NIPS&#x2019;12"},{"journal-title":"Synchronized sgd in ps-lite","year":"0","author":"li","key":"ref32"},{"journal-title":"Pmls-caffe Distributed deep learning framework for parallel ml system","year":"0","key":"ref31"},{"key":"ref30","first-page":"4:1","article-title":"Geeps: Scalable deep learning on distributed gpus with a gpu-specialized parameter server","author":"cui","year":"2016","journal-title":"Proceedings of the Eleventh European Conference on Computer Systems ser EuroSys &#x2018;16"},{"key":"ref37","first-page":"629","article-title":"Gaia: Geo-distributed machine learning approaching LAN speeds","author":"hsieh","year":"2017","journal-title":"14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17)"},{"journal-title":"Various proofs of the cauchy-schwarz inequality","year":"0","author":"wu","key":"ref36"},{"journal-title":"A lightweight parameter server interface","year":"0","author":"li","key":"ref35"},{"journal-title":"Accurate large minibatch sgd Training imagenet in 1 hour","year":"2017","author":"goyal","key":"ref34"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654889"},{"journal-title":"Learning multiple layers of features from tiny images","year":"2009","author":"krizhevsky","key":"ref40"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1093\/nsr\/nwx018"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2015.2472014"},{"key":"ref13","first-page":"181","article-title":"Poseidon: An efficient communication architecture for distributed deep learning on GPU clusters","author":"zhang","year":"2017","journal-title":"2017 USENIX Annual Technical Conference (USENIX ATC 17)"},{"key":"ref14","first-page":"571","article-title":"Project adam: Building an efficient and scalable deep learning training system","author":"chilimbi","year":"2014","journal-title":"11th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14)"},{"journal-title":"Don&#x2019;t decay the learning rate increase the batch size","year":"2018","author":"smith","key":"ref15"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/79173.79181"},{"key":"ref17","first-page":"28","article-title":"Loose synchronization for large-scale networked systems","author":"albrecht","year":"2006","journal-title":"Proceedings of the Annual Conference on USENIX Annual Technical Conference ser ATEC &#x2018;05"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/2124295.2124312"},{"key":"ref19","article-title":"Revisiting distributed synchronous sgd","author":"chen","year":"2016","journal-title":"International Conference on Learning Representations (ICLR) Workshop Track"},{"key":"ref28","doi-asserted-by":"crossref","first-page":"566","DOI":"10.1145\/3187009.3177734","article-title":"Flexps: Flexible parallelism control in parameter server architecture","volume":"11","author":"huang","year":"2018","journal-title":"Proc VLDB Endow"},{"key":"ref4","first-page":"583","article-title":"Scaling distributed machine learning with the parameter server","author":"li","year":"2014","journal-title":"11th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14)"},{"key":"ref27","first-page":"37","article-title":"Exploiting bounded staleness to speed up big data analytics","author":"cui","year":"2014","journal-title":"Proceedings of the 2014 USENIX Conference on USENIX Annual Technical Conference ser USENIX ATC&#x2019;14"},{"key":"ref3","first-page":"1223","article-title":"Large scale distributed deep networks","author":"dean","year":"2012","journal-title":"Proceedings of the 25th International Conference on Neural Information Processing Systems - Volume 1 ser NIPS&#x2019;12"},{"key":"ref6","first-page":"2834","article-title":"On model parallelization and scheduling strategies for distributed machine learning","author":"lee","year":"2014","journal-title":"Proceedings of the 27th International Conference on Neural Information Processing Systems - Volume 2 ser NIPS&#x2019;14"},{"journal-title":"Parameter server framework for distributed machine learning","year":"0","key":"ref29"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/2806777.2806778"},{"key":"ref8","first-page":"265","article-title":"Tensorflow: A system for large-scale machine learning","author":"abadi","year":"2016","journal-title":"12th USENIX Symp Operating Systems Design and Implementation (OSDI 16)"},{"key":"ref7","first-page":"5:1","article-title":"Strads: A distributed framework for scheduled model parallel machine learning","author":"kim","year":"2016","journal-title":"Proceedings of the Eleventh European Conference on Computer Systems ser EuroSys &#x2018;16"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.534"},{"key":"ref9","article-title":"Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems","volume":"abs 1512 1274","author":"chen","year":"2015","journal-title":"CoRR"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref20","article-title":"Solving the straggler problem with bounded staleness","author":"cipar","year":"2013","journal-title":"Presented as part of the 14th Workshop on Hot Topics in Operating Systems"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3035918.3035933"},{"key":"ref21","article-title":"Adam: A method for stochastic optimization","volume":"abs 1412 6980","author":"kingma","year":"2014","journal-title":"CoRR"},{"journal-title":"Probabilistic Synchronous Parallel","year":"2017","author":"wang","key":"ref42"},{"key":"ref24","article-title":"Gradient energy matching for distributed asynchronous gradient descent","volume":"abs 1805 8469","author":"hermans","year":"2018","journal-title":"CoRR"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/2987550.2987554"},{"key":"ref23","first-page":"2350","article-title":"Staleness-aware async-sgd for distributed deep learning","author":"zhang","year":"2016","journal-title":"Proceedings of the Twenty-Fifth International Joint Conference on Artificial Intelligence ser IJCAI&#x2019;16"},{"key":"ref26","first-page":"1223","article-title":"More effective distributed ml via a stale synchronous parallel parameter server","author":"ho","year":"2013","journal-title":"Proceedings of the 26th International Conference on Neural Information Processing Systems - Volume 1 ser NIPS&#x2019;13"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2018.00020"},{"key":"ref25","article-title":"A parameter communication optimization strategy for distributed machine learning in sensors","volume":"17","author":"zhang","year":"2017","journal-title":"SENSORS"}],"event":{"name":"2019 IEEE International Conference on Cluster Computing (CLUSTER)","start":{"date-parts":[[2019,9,23]]},"location":"Albuquerque, NM, USA","end":{"date-parts":[[2019,9,26]]}},"container-title":["2019 IEEE International Conference on Cluster Computing (CLUSTER)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8884608\/8890988\/08891038.pdf?arnumber=8891038","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,17]],"date-time":"2022-07-17T21:55:13Z","timestamp":1658094913000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8891038\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,9]]},"references-count":43,"URL":"https:\/\/doi.org\/10.1109\/cluster.2019.8891038","relation":{},"subject":[],"published":{"date-parts":[[2019,9]]}}}