{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T08:22:18Z","timestamp":1759134138975,"version":"3.28.0"},"reference-count":28,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,12]]},"DOI":"10.1109\/padsw.2018.8644533","type":"proceedings-article","created":{"date-parts":[[2019,2,21]],"date-time":"2019-02-21T23:23:38Z","timestamp":1550791418000},"page":"126-133","source":"Crossref","is-referenced-by-count":5,"title":["Parallelizing Machine Learning Optimization Algorithms on Distributed Data-Parallel Platforms with Parameter Server"],"prefix":"10.1109","author":[{"given":"Rong","family":"Gu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shiqing","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiu","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunfeng","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yihua","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","article-title":"Adadelta: An adaptive learning rate method","volume":"abs 1212 5701","author":"zeiler","year":"2012","journal-title":"CoRR"},{"key":"ref11","article-title":"Adam: A method for stochastic optimization","volume":"abs 1412 6980","author":"kingma","year":"2014","journal-title":"CoRR"},{"key":"ref12","article-title":"An overview of gradient descent optimization algorithms","volume":"abs 1609 4747","author":"ruder","year":"2016","journal-title":"CoRR"},{"key":"ref13","first-page":"372","article-title":"A method of solving a convex programming problem with convergence rate O(l\/k2)","volume":"27","author":"nesterov","year":"1983","journal-title":"Soviet Mathematics Doklady"},{"key":"ref14","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"duchi","year":"2011","journal-title":"J Mach Learn Res"},{"year":"2012","author":"tieleman","journal-title":"Lecture 6 5 - RMSprop Divide the gradient by a runnin average of its recent magnitude","key":"ref15"},{"key":"ref16","article-title":"Accurate, Large Minibatch SGD: Training ImageNet in 1 Hour","author":"goyal","year":"2017","journal-title":"ArXiv e-prints"},{"year":"2009","author":"krizhevsky","journal-title":"Learning multiple layers of features from tiny images","key":"ref17"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/CVPR.2016.90"},{"year":"2016","author":"gross","journal-title":"Training and investigating residual nets","key":"ref19"},{"doi-asserted-by":"publisher","key":"ref28","DOI":"10.1145\/2647868.2654889"},{"key":"ref4","first-page":"265","article-title":"Ten-sorflow: A system for large-scale machine learning","author":"abadi","year":"2016","journal-title":"12th USENIX Symp Operating Systems Design and Implementation (OSDI 16)"},{"key":"ref27","doi-asserted-by":"crossref","DOI":"10.1093\/nsr\/nwx018","article-title":"Angel: a new large-scale machine learning system","author":"jiang","year":"2018","journal-title":"Nat Sci Rev"},{"year":"0","journal-title":"PyTorch","key":"ref3"},{"year":"0","journal-title":"Apache spark - lightning-fast cluster computing","key":"ref6"},{"key":"ref5","article-title":"Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems","volume":"abs 1512 1274","author":"chen","year":"2015","journal-title":"CoRR"},{"key":"ref8","first-page":"1","article-title":"Scaling distributed machine learning with the parameter server","author":"li","year":"2014","journal-title":"International Conference on Big Data Science and Computing"},{"year":"0","journal-title":"Apache Hadoop","key":"ref7"},{"key":"ref2","first-page":"1235","article-title":"Mllib: machine learning in apache spark","volume":"17","author":"meng","year":"2015","journal-title":"Journal of Machine Learning Research"},{"key":"ref9","article-title":"Deep learning with elastic averaging SGD","volume":"abs 1412 6651","author":"zhang","year":"2014","journal-title":"CoRR"},{"key":"ref1","first-page":"1097","article-title":"Imagenet classification with deep convolutional neural networks","author":"krizhevsky","year":"2012","journal-title":"Advances in Neural Information Processing Systems 25 Curran Associates Inc"},{"key":"ref20","article-title":"Why random reshuffling beats stochastic gradient descent","author":"g\u00fcrb\u00fczbalaban","year":"2015","journal-title":"arXiv preprint arXiv 1510 08560"},{"doi-asserted-by":"publisher","key":"ref22","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref21","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref24","first-page":"1223","article-title":"Large scale distributed deep networks","author":"dean","year":"2012","journal-title":"Proceedings of the 25th International Conference on Neural Information Processing Systems (NIPS)"},{"key":"ref23","first-page":"693","article-title":"Hogwild: A lock-free approach to parallelizing stochastic gradient descent","volume":"24","author":"recht","year":"2011","journal-title":"Advances in neural information processing systems"},{"key":"ref26","first-page":"167","article-title":"Intel Math Kernel Library","author":"wang","year":"2014","journal-title":"High Performance Computing and Communications"},{"year":"0","journal-title":"BigDL Distributed deep learning on Apache Spark","key":"ref25"}],"event":{"name":"2018 IEEE 24th International Conference on Parallel and Distributed Systems (ICPADS)","start":{"date-parts":[[2018,12,11]]},"location":"Singapore, Singapore","end":{"date-parts":[[2018,12,13]]}},"container-title":["2018 IEEE 24th International Conference on Parallel and Distributed Systems (ICPADS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8635632\/8644527\/08644533.pdf?arnumber=8644533","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,27]],"date-time":"2022-01-27T05:54:55Z","timestamp":1643262895000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8644533\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,12]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/padsw.2018.8644533","relation":{},"subject":[],"published":{"date-parts":[[2018,12]]}}}