{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:38:20Z","timestamp":1782833900053,"version":"3.54.5"},"reference-count":25,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016,9]]},"DOI":"10.1109\/allerton.2016.7852343","type":"proceedings-article","created":{"date-parts":[[2017,2,13]],"date-time":"2017-02-13T21:38:11Z","timestamp":1487021891000},"page":"997-1004","source":"Crossref","is-referenced-by-count":57,"title":["Asynchrony begets momentum, with an application to deep learning"],"prefix":"10.1109","author":[{"given":"Ioannis","family":"Mitliagkas","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ce","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stefan","family":"Hadjis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Christopher","family":"Re","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","first-page":"2674","article-title":"Taming the wild: A unified analysis of hogwild-style algorithms","author":"de sa","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref11","article-title":"Tensorflow: A system for large-scale machine learning","author":"abadi","year":"2016","journal-title":"arXiv preprint arXiv 1605 01584"},{"key":"ref12","article-title":"Revisiting distributed synchronous sgd","author":"chen","year":"2016","journal-title":"arXiv preprint arXiv 1604 00981"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/2901318.2901323"},{"key":"ref14","article-title":"Omnivore: An optimizer for multi-device deep learning on cpus and gpus","author":"hadjis","year":"2016","journal-title":"arXiv preprint arXiv 1606 04487"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/0041-5553(64)90137-5"},{"key":"ref16","first-page":"1097","article-title":"Imagenet classification with deep convolutional neural networks","author":"krizhevsky","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref17","article-title":"Caffe solver documentation","year":"0"},{"key":"ref18","article-title":"An overview of gradient descent optimization algorithms","author":"ruder","year":"0"},{"key":"ref19","first-page":"1139","article-title":"On the importance of initialization and momentum in deep learning","author":"sutskever","year":"2013","journal-title":"Proceedings of the 30th International Conference on Machine Learning (ICML-13)"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015332"},{"key":"ref6","first-page":"1223","article-title":"Large scale distributed deep networks","author":"dean","year":"2012","journal-title":"Advances in neural information processing systems"},{"key":"ref5","first-page":"693","article-title":"Hogwild: A lock-free approach to parallelizing stochastic gradient descent","author":"niu","year":"2011","journal-title":"Advances in neural information processing systems"},{"key":"ref8","article-title":"Perturbed iterate analysis for asynchronous stochastic optimization","author":"mania","year":"2015","journal-title":"arXiv preprint arXiv 1507 06970"},{"key":"ref7","first-page":"571","article-title":"Project adam: Building an efficient and scalable deep learning training system","author":"chilimbi","year":"2014","journal-title":"11 th USENIX Symposium on Operating Systems Design and Implementation (OSDI 14)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/0925-2312(93)90006-O"},{"key":"ref9","first-page":"1531","article-title":"Asyn-chronous stochastic convex optimization: the noise is in the noise and sgd don't care","author":"chaturapruek","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/0165-1684(84)90013-6"},{"key":"ref20","article-title":"Train faster, generalize better: Stability of stochastic gradient descent","author":"hardt","year":"2015","journal-title":"arXiv preprint arXiv 1509 01240"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.14778\/2732977.2733001"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ALLERTON.2016.7852343"},{"key":"ref24","article-title":"CaffeNet, solver for ImageNet","year":"0"},{"key":"ref23","article-title":"Caffe solver for CIFAR","year":"0"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"}],"event":{"name":"2016 54th Annual Allerton Conference on Communication, Control, and Computing (Allerton)","location":"Monticello, IL, USA","start":{"date-parts":[[2016,9,27]]},"end":{"date-parts":[[2016,9,30]]}},"container-title":["2016 54th Annual Allerton Conference on Communication, Control, and Computing (Allerton)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7819723\/7852197\/07852343.pdf?arnumber=7852343","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,10,3]],"date-time":"2017-10-03T02:05:21Z","timestamp":1506996321000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7852343\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,9]]},"references-count":25,"URL":"https:\/\/doi.org\/10.1109\/allerton.2016.7852343","relation":{},"subject":[],"published":{"date-parts":[[2016,9]]}}}