{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,13]],"date-time":"2025-11-13T12:34:47Z","timestamp":1763037287371,"version":"3.44.0"},"reference-count":28,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,9,1]],"date-time":"2019-09-01T00:00:00Z","timestamp":1567296000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Parallel Distrib. Syst."],"published-print":{"date-parts":[[2019,9,1]]},"DOI":"10.1109\/tpds.2019.2904058","type":"journal-article","created":{"date-parts":[[2019,3,8]],"date-time":"2019-03-08T14:36:24Z","timestamp":1552055784000},"page":"2090-2100","source":"Crossref","is-referenced-by-count":46,"title":["Parallelizing Word2Vec in Shared and Distributed Memory"],"prefix":"10.1109","volume":"30","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3573-5379","authenticated-orcid":false,"given":"Shihao","family":"Ji","sequence":"first","affiliation":[{"name":"Department of Computer Science, Georgia State University, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nadathur","family":"Satish","sequence":"additional","affiliation":[{"name":"Intel Labs, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sheng","family":"Li","sequence":"additional","affiliation":[{"name":"Intel Labs, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pradeep K.","family":"Dubey","sequence":"additional","affiliation":[{"name":"Intel Labs, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","first-page":"1412","article-title":"Effective approaches to attention-based neural machine translation","author":"luong hieupham","year":"2015","journal-title":"Proc Conf Empirical Methods Natural Language Process"},{"key":"ref11","article-title":"Memory networks","author":"weston","year":"2015","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref12","first-page":"1378","article-title":"Ask me anything: Dynamic memory networks for natural language processing","author":"kumar","year":"2016","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729586"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/567806.567807"},{"key":"ref15","first-page":"693","article-title":"HOGWILD: A lock-free approach to parallelizing stochastic gradient descent","author":"niu","year":"2011","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref16","first-page":"2121","article-title":"Adaptive subgradient methods for online learning and stochastic optimization","volume":"12","author":"duchi","year":"2011","journal-title":"J Mach Learn Res"},{"key":"ref17","first-page":"26","article-title":"Lecture 6.5-rmsprop: Divide the gradient by a running average of its recent magnitude","volume":"4","author":"hinton","year":"2012","journal-title":"Neural Netw Mach Learning"},{"key":"ref18","first-page":"2635","article-title":"One billion word benchmark for measuring progress in statistical language modeling","author":"chelba","year":"2014","journal-title":"Proc Annu Conf Int Speech Commun Assoc"},{"key":"ref19","article-title":"Parallelizing Word2Vec in multi-core and many-core architectures","author":"ji","year":"2016","journal-title":"Proc NIPS Workshop Efficient Methods Deep Neural Netw"},{"article-title":"Large batch training of convolutional networks","year":"2017","author":"you","key":"ref28"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390177"},{"article-title":"AdaBatch: Adaptive batch sizes for training deep neural networks","year":"2017","author":"devarakonda","key":"ref27"},{"journal-title":"Foundations of Statistical Natural Language Processing","year":"1999","author":"manning","key":"ref3"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00233"},{"key":"ref5","first-page":"513","article-title":"A unified architecture for natural language processing: Deep neural networks with multitask learning","author":"glorot","year":"2011","journal-title":"Proc 25th Int Conf Mach Learn"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"ref7","first-page":"2048","article-title":"Show, attend and tell: Neural image caption generation with visual attention","author":"xu","year":"2015","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref2","first-page":"3111","article-title":"Distributed representations of words and phrases and their compositionality","author":"mikolov","year":"2013","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref9","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","author":"sutskever","year":"2014","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref1","article-title":"Efficient estimation of word representations in vector space","author":"mikolov","year":"2013","journal-title":"Proc Workshop Int Conf Learn Represent"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1080\/01690969108406936"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/BigData.2015.7363760"},{"key":"ref21","first-page":"307","article-title":"Noise-contrastive estimation of unnormalized statistical models, with applications to natural image statistics","volume":"13","author":"gutmann","year":"2012","journal-title":"J Mach Learn Res"},{"article-title":"Splash: User-friendly programming interface for parallelizing stochastic algorithms","year":"2015","author":"zhang","key":"ref24"},{"key":"ref23","article-title":"On large-batch training for deep learning: Generalization gap and sharp minima","author":"keskar","year":"2017","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/2640087.2644155"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/503104.503110"}],"container-title":["IEEE Transactions on Parallel and Distributed Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/71\/8790955\/08663393.pdf?arnumber=8663393","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T01:03:47Z","timestamp":1755911027000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8663393\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,9,1]]},"references-count":28,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/tpds.2019.2904058","relation":{},"ISSN":["1045-9219","1558-2183","2161-9883"],"issn-type":[{"type":"print","value":"1045-9219"},{"type":"electronic","value":"1558-2183"},{"type":"electronic","value":"2161-9883"}],"subject":[],"published":{"date-parts":[[2019,9,1]]}}}