{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T13:44:35Z","timestamp":1782999875747,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402525"],"award-info":[{"award-number":["62402525"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,6]]},"DOI":"10.1145\/3797905.3807865","type":"proceedings-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T11:50:37Z","timestamp":1782993037000},"page":"662-674","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Optimizing Streaming Tensor Decomposition on GPU"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-5519-2819","authenticated-orcid":false,"given":"Wenqing","family":"Lin","sequence":"first","affiliation":[{"name":"College of Artificial Intelligence, China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9135-7338","authenticated-orcid":false,"given":"Jianuo","family":"Sheng","sequence":"additional","affiliation":[{"name":"College of Artificial Intelligence, China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5160-7349","authenticated-orcid":false,"given":"Shuqin","family":"Feng","sequence":"additional","affiliation":[{"name":"College of Artificial Intelligence, China University of Petroleum-Beijing, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0664-9543","authenticated-orcid":false,"given":"Ming","family":"Dun","sequence":"additional","affiliation":[{"name":"State Key Lab of Processors, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1176-2521","authenticated-orcid":false,"given":"Huawei","family":"Cao","sequence":"additional","affiliation":[{"name":"State Key Lab of Processors, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2927-362X","authenticated-orcid":false,"given":"Qingxiao","family":"Sun","sequence":"additional","affiliation":[{"name":"China University of Petroleum-Beijing, Beijing, China and Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Karim Abed-Meraim Nguyen\u00a0Linh Trung Adel Hafiane et\u00a0al. 2023. Tracking online low-rank approximations of higher-order incomplete streaming tensors. Patterns (2023) 100763."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"crossref","unstructured":"J\u00a0Douglas Carroll and Jih-Jie Chang. 1970. Analysis of individual differences in multidimensional scaling via an N-way generalization of \u201cEckart-Young\u201d decomposition. Psychometrika (1970) 283\u2013319.","DOI":"10.1007\/BF02310791"},{"key":"e_1_3_3_1_4_2","first-page":"578","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI)","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, et\u00a0al. 2018. { TVM} : An automated { End-to-End} optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI). USENIX Association, 578\u2013594."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"Robert\u00a0L Cook Loren Carpenter and Edwin Catmull. 1987. The Reyes image rendering architecture. ACM SIGGRAPH Computer Graphics (ACM SIGGRAPH Comput. Graph.) (1987) 95\u2013102.","DOI":"10.1145\/37401.37414"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853551"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00071"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"H\u00a0Carter Edwards Daniel Sunderland Vicki Porter Chris Amsler and Sam Mish. 2012. Manycore Performance-Portability: Kokkos Multidimensional Array Library. Scientific Programming (Sci. Program.) (2012) 89\u2013114.","DOI":"10.1155\/2012\/917630"},{"key":"e_1_3_3_1_9_2","unstructured":"Richard\u00a0A Harshman et\u00a0al. 1970. Foundations of the PARAFAC procedure: Models and conditions for an \u201cexplanatory\u201d multi-modal factor analysis. UCLA working papers in phonetics (UCLA Work. Pap. Phonetics) (1970) 84."},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3447818.3461703"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359630"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/1864708.1864727"},{"key":"e_1_3_3_1_13_2","first-page":"1012","volume-title":"International conference on machine learning (ICML)","author":"Kasai Hiroyuki","year":"2016","unstructured":"Hiroyuki Kasai and Bamdev Mishra. 2016. Low-rank tensor completion: a Riemannian manifold preconditioning approach. In International conference on machine learning (ICML). PMLR, 1012\u20131021."},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Tamara\u00a0G Kolda and Brett\u00a0W Bader. 2009. Tensor decompositions and applications. SIAM review (SIAM Rev.) (2009) 455\u2013500.","DOI":"10.1137\/07070111X"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE51399.2021.00076"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00022"},{"key":"e_1_3_3_1_17_2","unstructured":"Ping Li Art Owen and Cun-Hui Zhang. 2012. One permutation hashing. Advances in Neural Information Processing Systems(NeurIPS) (2012) 3113\u20133121."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER59578.2024.00036"},{"key":"e_1_3_3_1_19_2","first-page":"77","volume-title":"IFIP International Conference on Network and Parallel Computing(NPC)","author":"Liu Guanxiong","year":"2024","unstructured":"Guanxiong Liu and Hao Wu. 2024. Hp-csf: An gpu optimization method for cp decomposition of incomplete tensors. In IFIP International Conference on Network and Parallel Computing(NPC). Springer, 77\u201390."},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Morten M\u00f8rup Lars\u00a0Kai Hansen and Sidse\u00a0M Arnfred. 2008. Algorithms for sparse nonnegative Tucker decompositions. Neural computation (Neural Comput.) (2008) 2112\u20132131.","DOI":"10.1162\/neco.2008.11-06-407"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356216"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.5555\/212559"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3592979.3593405"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD63220.2024.00083"},{"key":"e_1_3_3_1_25_2","unstructured":"Paul Resnick and Hal\u00a0R Varian. 1997. Recommender systems. Communications of the ACM (CACM) (1997) 1\u201349."},{"key":"e_1_3_3_1_26_2","volume-title":"Chemometrics","author":"Sharaf Muhammad\u00a0A","year":"1986","unstructured":"Muhammad\u00a0A Sharaf, Deborah\u00a0L Illman, and Bruce\u00a0R Kowalski. 1986. Chemometrics. John Wiley & Sons."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"crossref","unstructured":"Nicholas\u00a0D Sidiropoulos Lieven De\u00a0Lathauwer Xiao Fu Kejun Huang Evangelos\u00a0E Papalexakis and Christos Faloutsos. 2017. Tensor decomposition for signal processing and machine learning. IEEE Transactions on signal processing (TSP) (2017) 3551\u20133582.","DOI":"10.1109\/TSP.2017.2690524"},{"key":"e_1_3_3_1_28_2","volume-title":"FROSTT: The Formidable Repository of Open Sparse Tensors and Tools","author":"Smith Shaden","year":"2017","unstructured":"Shaden Smith, Jee\u00a0W. Choi, Jiajia Li, Richard Vuduc, Jongsoo Park, Xing Liu, and George Karypis. 2017. FROSTT: The Formidable Repository of Open Sparse Tensors and Tools. http:\/\/frostt.io\/"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611975321.10"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2015.27"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS49936.2021.00078"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS54959.2023.00048"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/3673038.3673128"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"crossref","unstructured":"Sangjun Son Yong-chan Park Minyong Cho and U Kang. 2022. DAO-CP: Data-Adaptive Online CP decomposition for tensor stream. Plos one (2022) e0267091.","DOI":"10.1371\/journal.pone.0267091"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00022"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/CGO57630.2024.10444812"},{"key":"e_1_3_3_1_37_2","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO.2017.8081290"},{"key":"e_1_3_3_1_38_2","doi-asserted-by":"crossref","unstructured":"Sasindu Wijeratne Rajgopal Kannan and Viktor Prasanna. 2025. Accelerating Sparse MTTKRP for Small Tensor Decomposition on GPU. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.18198 (2025) 1\u20136.","DOI":"10.2139\/ssrn.5526799"},{"key":"e_1_3_3_1_39_2","doi-asserted-by":"crossref","unstructured":"Jiaming Xu Shan Huang Jinhao Li Guyue Huang Yuan Xie Yu Wang and Guohao Dai. 2024. Enabling Efficient Sparse Multiplications on GPUs with Heuristic Adaptability. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems (TCAD) (2024) 2226 \u2013 2239.","DOI":"10.1109\/TCAD.2024.3518413"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/326"},{"key":"e_1_3_3_1_41_2","doi-asserted-by":"publisher","DOI":"10.1145\/3603165.3607434"},{"key":"e_1_3_3_1_42_2","doi-asserted-by":"publisher","DOI":"10.1145\/3572848.3577506"},{"key":"e_1_3_3_1_43_2","first-page":"10565","volume-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems(NeurIPS)","author":"Zhang Xi","year":"2025","unstructured":"Xi Zhang, Yanyi Li, Yisi Luo, Qi Xie, and Deyu Meng. 2025. Online Functional Tensor Decomposition via Continual Learning for Streaming Data Completion. In The Thirty-ninth Annual Conference on Neural Information Processing Systems(NeurIPS). NeurIPS Foundation, 10565\u201310593."},{"key":"e_1_3_3_1_44_2","first-page":"863","volume-title":"14th USENIX symposium on operating systems design and implementation (OSDI)","author":"Zheng Lianmin","year":"2020","unstructured":"Lianmin Zheng, Chengfan Jia, Minmin Sun, Zhao Wu, Cody\u00a0Hao Yu, Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, Koushik Sen, et\u00a0al. 2020. Ansor: Generating { High-Performance} tensor programs for deep learning. In 14th USENIX symposium on operating systems design and implementation (OSDI). USENIX Association, 863\u2013879."},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939763"}],"event":{"name":"ICS '26: 2026 International Conference on Supercomputing","location":"Belfast United Kingdom","acronym":"ICS '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 40th ACM International Conference on Supercomputing"],"original-title":[],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T12:59:54Z","timestamp":1782997194000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3797905.3807865"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":44,"alternative-id":["10.1145\/3797905.3807865","10.1145\/3797905"],"URL":"https:\/\/doi.org\/10.1145\/3797905.3807865","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}