{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T15:50:28Z","timestamp":1784217028561,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,6,3]],"date-time":"2021-06-03T00:00:00Z","timestamp":1622678400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2020YFB1506703"],"award-info":[{"award-number":["2020YFB1506703"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072018,61732002"],"award-info":[{"award-number":["62072018,61732002"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,6,3]]},"DOI":"10.1145\/3447818.3460692","type":"proceedings-article","created":{"date-parts":[[2021,6,4]],"date-time":"2021-06-04T15:09:36Z","timestamp":1622819376000},"page":"417-430","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["An optimized tensor completion library for multiple GPUs"],"prefix":"10.1145","author":[{"given":"Ming","family":"Dun","sequence":"first","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunchun","family":"Li","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingxiao","family":"Sun","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongzhi","family":"Luan","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2021,6,4]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"GPU performance analysis and optimisation","author":"Bradley Thomas","year":"2012","unstructured":"Thomas Bradley . 2012. GPU performance analysis and optimisation . NVIDIA Corporation ( 2012 ). Thomas Bradley. 2012. GPU performance analysis and optimisation. NVIDIA Corporation (2012)."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the Twenty-Fourth AAAI Conference on Artificial Intelligence. AAAI Press","author":"Carlson Andrew","unstructured":"Andrew Carlson , Justin Betteridge , Bryan Kisiel , Burr Settles , Estevam R. Hruschka Jr ., and Tom M. Mitchell . 2010. Toward an Architecture for Never-Ending Language Learning .. In Proceedings of the Twenty-Fourth AAAI Conference on Artificial Intelligence. AAAI Press , Atlanta, Georgia, 1306--1313. Andrew Carlson, Justin Betteridge, Bryan Kisiel, Burr Settles, Estevam R. Hruschka Jr., and Tom M. Mitchell. 2010. Toward an Architecture for Never-Ending Language Learning.. In Proceedings of the Twenty-Fourth AAAI Conference on Artificial Intelligence. AAAI Press, Atlanta, Georgia, 1306--1313."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2018.00066"},{"key":"e_1_3_2_1_4_1","volume-title":"Advances in Neural Information Processing Systems. Curran Associates","author":"Choi Joon Hee","unstructured":"Joon Hee Choi and S Vishwanathan . 2014. DFacTo: Distributed factorization of tensors . In Advances in Neural Information Processing Systems. Curran Associates , Inc., Montreal, Canada , 1296--1304. Joon Hee Choi and S Vishwanathan. 2014. DFacTo: Distributed factorization of tensors. In Advances in Neural Information Processing Systems. Curran Associates, Inc., Montreal, Canada, 1296--1304."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1327452.1327492"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.5555\/3000375.3000377"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2014.07.003"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2020408.2020426"},{"key":"e_1_3_2_1_9_1","volume-title":"Genetic algorithms and machine learning. Machine learning 3, 2","author":"Goldberg David E","year":"1988","unstructured":"David E Goldberg and John H Holland . 1988. Genetic algorithms and machine learning. Machine learning 3, 2 ( 1988 ), 95--99. David E Goldberg and John H Holland. 1988. Genetic algorithms and machine learning. Machine learning 3, 2 (1988), 95--99."},{"key":"e_1_3_2_1_10_1","unstructured":"Mark Harris. 2017. Nvidia dgx-1: The fastest deep learning system. (2017).  Mark Harris. 2017. Nvidia dgx-1: The fastest deep learning system. (2017)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2623330.2623658"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0186251"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2488608.2488693"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2015.7113355"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2339530.2339583"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2015.10.002"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2807591.2807624"},{"key":"e_1_3_2_1_18_1","first-page":"2057","article-title":"Matrix completion from noisy entries","author":"Keshavan Raghunandan H","year":"2010","unstructured":"Raghunandan H Keshavan , Andrea Montanari , and Sewoong Oh . 2010 . Matrix completion from noisy entries . Journal of Machine Learning Research 11 , Jul (2010), 2057 -- 2078 . Raghunandan H Keshavan, Andrea Montanari, and Sewoong Oh. 2010. Matrix completion from noisy entries. Journal of Machine Learning Research 11, Jul (2010), 2057--2078.","journal-title":"Journal of Machine Learning Research 11"},{"key":"e_1_3_2_1_19_1","series-title":"SIAM review 51, 3","volume-title":"Tensor decompositions and applications","author":"Kolda Tamara G","year":"2009","unstructured":"Tamara G Kolda and Brett W Bader . 2009. Tensor decompositions and applications . SIAM review 51, 3 ( 2009 ), 455--500. Tamara G Kolda and Brett W Bader. 2009. Tensor decompositions and applications. SIAM review 51, 3 (2009), 455--500."},{"key":"e_1_3_2_1_20_1","volume-title":"Practical leverage-based sampling for low-rank tensor decomposition. arXiv preprint arXiv:2006.16438","author":"Larsen Brett W","year":"2020","unstructured":"Brett W Larsen and Tamara G Kolda . 2020. Practical leverage-based sampling for low-rank tensor decomposition. arXiv preprint arXiv:2006.16438 ( 2020 ). Brett W Larsen and Tamara G Kolda. 2020. Practical leverage-based sampling for low-rank tensor decomposition. arXiv preprint arXiv:2006.16438 (2020)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00022"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2017.75"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1088\/1361-6560\/ab0db5"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2507157.2507163"},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the department of defense HPCMP users group conference","volume":"710","author":"Mucci Philip J","year":"1999","unstructured":"Philip J Mucci , Shirley Browne , Christine Deane , and George Ho . 1999 . PAPI: A portable interface to hardware performance counters . In Proceedings of the department of defense HPCMP users group conference , Vol. 710 . Citeseer. Philip J Mucci, Shirley Browne, Christine Deane, and George Ho. 1999. PAPI: A portable interface to hardware performance counters. In Proceedings of the department of defense HPCMP users group conference, Vol. 710. Citeseer."},{"key":"e_1_3_2_1_26_1","unstructured":"Maxim Naumov. 2011. Incomplete-LU and Cholesky preconditioned iterative methods using CUSPARSE and CUBLAS. (2011).  Maxim Naumov. 2011. Incomplete-LU and Cholesky preconditioned iterative methods using CUSPARSE and CUBLAS. (2011)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356216"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2019.00023"},{"key":"e_1_3_2_1_29_1","unstructured":"NVIDIA. 2019. The CUDA solver library (cuSOLVER). (2019). https:\/\/docs.nvidia.com\/cuda\/cusolver\/index.html  NVIDIA. 2019. The CUDA solver library (cuSOLVER). (2019). https:\/\/docs.nvidia.com\/cuda\/cusolver\/index.html"},{"key":"e_1_3_2_1_30_1","unstructured":"Nvidia. 2020. cuTENSOR: A High-Performance CUDA Library For Tensor Primitives. (2020). https:\/\/docs.nvidia.com\/cuda\/cutensor\/index.html  Nvidia. 2020. cuTENSOR: A High-Performance CUDA Library For Tensor Primitives. (2020). https:\/\/docs.nvidia.com\/cuda\/cutensor\/index.html"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1137\/18M1210691"},{"key":"e_1_3_2_1_32_1","volume-title":"Hogwild: A lock-free approach to parallelizing stochastic gradient descent. In Advances in neural information processing systems. Curran Associates","author":"Recht Benjamin","year":"2011","unstructured":"Benjamin Recht , Christopher Re , Stephen Wright , and Feng Niu . 2011 . Hogwild: A lock-free approach to parallelizing stochastic gradient descent. In Advances in neural information processing systems. Curran Associates , Inc., Granada, Spain , 693--701. Benjamin Recht, Christopher Re, Stephen Wright, and Feng Niu. 2011. Hogwild: A lock-free approach to parallelizing stochastic gradient descent. In Advances in neural information processing systems. Curran Associates, Inc., Granada, Spain, 693--701."},{"key":"e_1_3_2_1_33_1","unstructured":"GroupLens Research. 2019. MovieLens. (2019). http:\/\/grouplens.org\/datasets\/movielens\/  GroupLens Research. 2019. MovieLens. (2019). http:\/\/grouplens.org\/datasets\/movielens\/"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2348283.2348308"},{"key":"e_1_3_2_1_35_1","volume-title":"FROSTT: The Formidable Repository of Open Sparse Tensors and Tools.","author":"Smith Shaden","year":"2017","unstructured":"Shaden Smith , Jee W. Choi , Jiajia Li , Richard Vuduc , Jongsoo Park , Xing Liu , and George Karypis . 2017 . FROSTT: The Formidable Repository of Open Sparse Tensors and Tools. (2017). http:\/\/frostt.io\/ Shaden Smith, Jee W. Choi, Jiajia Li, Richard Vuduc, Jongsoo Park, Xing Liu, and George Karypis. 2017. FROSTT: The Formidable Repository of Open Sparse Tensors and Tools. (2017). http:\/\/frostt.io\/"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/2833179.2833183"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2017.11.002"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2015.27"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2014.06.002"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2017.52"},{"key":"e_1_3_2_1_41_1","volume-title":"Tensor completion algorithms in big data analytics. ACM Transactions on Knowledge Discovery from Data (TKDD) 13, 1","author":"Song Qingquan","year":"2019","unstructured":"Qingquan Song , Hancheng Ge , James Caverlee , and Xia Hu. 2019. Tensor completion algorithms in big data analytics. ACM Transactions on Knowledge Discovery from Data (TKDD) 13, 1 ( 2019 ), 1--48. Qingquan Song, Hancheng Ge, James Caverlee, and Xia Hu. 2019. Tensor completion algorithms in big data analytics. ACM Transactions on Knowledge Discovery from Data (TKDD) 13, 1 (2019), 1--48."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3091966.3091968"},{"key":"e_1_3_2_1_43_1","volume-title":"Tensaurus: A Versatile Accelerator for Mixed Sparse-Dense Tensor Computations. In 2020 IEEE International Symposium on High Performance Computer Architecture (HPCA). IEEE","author":"Srivastava Nitish","year":"2020","unstructured":"Nitish Srivastava , Hanchen Jin , Shaden Smith , Hongbo Rong , David Albonesi , and Zhiru Zhang . 2020 . Tensaurus: A Versatile Accelerator for Mixed Sparse-Dense Tensor Computations. In 2020 IEEE International Symposium on High Performance Computer Architecture (HPCA). IEEE , San Diego, USA, 689--702. Nitish Srivastava, Hanchen Jin, Shaden Smith, Hongbo Rong, David Albonesi, and Zhiru Zhang. 2020. Tensaurus: A Versatile Accelerator for Mixed Sparse-Dense Tensor Computations. In 2020 IEEE International Symposium on High Performance Computer Architecture (HPCA). IEEE, San Diego, USA, 689--702."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2013.11.020"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2011.5771445"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2783395"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498765.1498785"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/2442516.2442539"},{"key":"e_1_3_2_1_49_1","unstructured":"YELP. 2015. YELP. (2015). http:\/\/www.yelp.com\/dataset_challenge\/  YELP. 2015. YELP. (2015). http:\/\/www.yelp.com\/dataset_challenge\/"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2012.168"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2020.2975196"},{"key":"e_1_3_2_1_52_1","volume-title":"Enabling Distributed-Memory Tensor Completion in Python using New Sparse Tensor Kernels. arXiv preprint arXiv:1910.02371","author":"Zhang Zecheng","year":"2019","unstructured":"Zecheng Zhang , Xiaoxiao Wu , Naijing Zhang , Siyuan Zhang , and Edgar Solomonik . 2019. Enabling Distributed-Memory Tensor Completion in Python using New Sparse Tensor Kernels. arXiv preprint arXiv:1910.02371 ( 2019 ). Zecheng Zhang, Xiaoxiao Wu, Naijing Zhang, Siyuan Zhang, and Edgar Solomonik. 2019. Enabling Distributed-Memory Tensor Completion in Python using New Sparse Tensor Kernels. arXiv preprint arXiv:1910.02371 (2019)."}],"event":{"name":"ICS '21: 2021 International Conference on Supercomputing","location":"Virtual Event USA","acronym":"ICS '21","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the ACM International Conference on Supercomputing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3447818.3460692","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3447818.3460692","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T17:49:27Z","timestamp":1750268967000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3447818.3460692"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,3]]},"references-count":52,"alternative-id":["10.1145\/3447818.3460692","10.1145\/3447818"],"URL":"https:\/\/doi.org\/10.1145\/3447818.3460692","relation":{},"subject":[],"published":{"date-parts":[[2021,6,3]]},"assertion":[{"value":"2021-06-04","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}