{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T14:46:22Z","timestamp":1764859582438,"version":"3.37.3"},"reference-count":58,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/access.2023.3315239","type":"journal-article","created":{"date-parts":[[2023,9,13]],"date-time":"2023-09-13T17:41:11Z","timestamp":1694626871000},"page":"100165-100179","source":"Crossref","is-referenced-by-count":1,"title":["ezLDA: Efficient and Scalable LDA on GPUs"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6232-2863","authenticated-orcid":false,"given":"Shilong","family":"Wang","sequence":"first","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of Massachusetts Lowell, Lowell, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6323-7388","authenticated-orcid":false,"given":"Hang","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, Stevens Institute of Technology, Hoboken, NJ, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4892-6799","authenticated-orcid":false,"given":"Anil","family":"Gaihre","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, Stevens Institute of Technology, Hoboken, NJ, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5852-0813","authenticated-orcid":false,"given":"Hengyong","family":"Yu","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of Massachusetts Lowell, Lowell, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","first-page":"1765","article-title":"End-to-end learning of LDA by mirror-descent back propagation over a deep architecture","author":"chen","year":"2015","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref57","first-page":"147","article-title":"Correlated topic models","volume":"18","author":"blei","year":"2006","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref12","article-title":"Syntax aware LSTM model for Chinese semantic role labeling","author":"qian","year":"2017","journal-title":"arXiv 1704 00405"},{"journal-title":"Nvidia Titan 1080","year":"2020","key":"ref56"},{"key":"ref15","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2018","journal-title":"arXiv 1810 04805"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1202"},{"key":"ref58","first-page":"1","article-title":"Sharing clusters among related groups: Hierarchical Dirichlet processes","volume":"17","author":"teh","year":"2004","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref53","first-page":"1","article-title":"UMBC-EBIQUITY-CORE: Semantic textual similarity systems","author":"han","year":"2013","journal-title":"Proc 2nd Joint Conf Lexical Comput Semantics"},{"journal-title":"Bag of Words","year":"2008","author":"newman","key":"ref52"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.14778\/3236187.3236203"},{"journal-title":"Nvidia Nvprof","year":"2020","key":"ref55"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1016\/j.dss.2017.11.001"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.21"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2019.00057"},{"key":"ref16","article-title":"Q-BERT: Hessian based ultra low precision quantization of BERT","author":"shen","year":"2019","journal-title":"arXiv 1909 05840"},{"key":"ref19","article-title":"Deep speech: Scaling up end-to-end speech recognition","author":"hannun","year":"2014","journal-title":"arXiv 1412 5567"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3146347.3146358"},{"journal-title":"Cuda Atomics Change Flag","year":"2018","author":"crovella","key":"ref51"},{"key":"ref50","doi-asserted-by":"crossref","first-page":"154","DOI":"10.14778\/3282495.3282501","article-title":"Start late or finish early: A distributed graph processing system with redundancy reduction","volume":"12","author":"song","year":"2018","journal-title":"Proc VLDB Endowment"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2017.41"},{"key":"ref45","first-page":"413","article-title":"Locality-aware software throttling for sparse matrix operation on GPUs","author":"chen","year":"2018","journal-title":"Proc USENIX Conf USENIX Annu Tech Conf"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2013.235"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3302424.3303949"},{"key":"ref42","first-page":"966","article-title":"Exponential stochastic cellular automata for massively parallel inference","author":"zaheer","year":"2016","journal-title":"Proc Int Conf Artif Intell Statist"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1002\/widm.1232"},{"journal-title":"Numerical Recipes 3rd Edition The Art of Scientific Computing","year":"2007","author":"press","key":"ref44"},{"key":"ref43","first-page":"97","article-title":"Scan primitives for GPU computing","author":"sengupta","year":"2007","journal-title":"Proc Graphics hardware"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-68530-8_33"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.14778\/3137628.3137649"},{"key":"ref7","first-page":"81","article-title":"Relational topic models for document networks","author":"chang","year":"2009","journal-title":"Proc Artif Intell Statist"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2018.2877363"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2007.4408965"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3205289.3205325"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/1639714.1639726"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/1526709.1526801"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2008.917757"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/1401890.1401960"},{"key":"ref34","first-page":"2134","article-title":"Parallel inference for latent Dirichlet allocation on graphics processing units","author":"yan","year":"2009","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.2307\/2290005"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21236\/ADA629956"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/2736277.2741682"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2783416"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2018.00052"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939821"},{"key":"ref2","first-page":"1","article-title":"A topic model for word sense disambiguation","author":"boyd-graber","year":"2007","journal-title":"Proc EMNLP-CoNLL"},{"key":"ref1","first-page":"993","article-title":"Latent Dirichlet allocation","volume":"3","author":"blei","year":"2003","journal-title":"J Mach Learn Res"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/3307681.3326606"},{"journal-title":"V100 GPU","year":"2019","key":"ref38"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1214\/aos\/1176325750"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/1557019.1557121"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/2736277.2741115"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/2623330.2623756"},{"key":"ref20","article-title":"ChainerMN: Scalable distributed deep learning framework","author":"akiba","year":"2017","journal-title":"arXiv 1710 11351"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/2688500.2688538"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/MLHPC.2018.8638639"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3093315.3037740"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.14778\/2977797.2977801"},{"key":"ref29","article-title":"CuLDA_CGS: Solving large-scale LDA problems on GPUs","author":"xie","year":"2018","journal-title":"arXiv 1803 04631"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10005208\/10250424.pdf?arnumber=10250424","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,9]],"date-time":"2023-10-09T19:32:31Z","timestamp":1696879951000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10250424\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":58,"URL":"https:\/\/doi.org\/10.1109\/access.2023.3315239","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2023]]}}}