{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T13:21:17Z","timestamp":1760016077292,"version":"3.37.3"},"reference-count":27,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,2,16]],"date-time":"2024-02-16T00:00:00Z","timestamp":1708041600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,16]],"date-time":"2024-02-16T00:00:00Z","timestamp":1708041600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100019062","name":"Tianjin Research Innovation Project for Postgraduate Students","doi-asserted-by":"publisher","award":["2022BKY023"],"award-info":[{"award-number":["2022BKY023"]}],"id":[{"id":"10.13039\/501100019062","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"crossref","award":["2021YFB0300104"],"award-info":[{"award-number":["2021YFB0300104"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["CCF Trans. HPC"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s42514-024-00181-3","type":"journal-article","created":{"date-parts":[[2024,2,16]],"date-time":"2024-02-16T20:02:38Z","timestamp":1708113758000},"page":"319-329","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["oclCUB: an OpenCL parallel computing library for deep learning operators"],"prefix":"10.1007","volume":"6","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-5732-2207","authenticated-orcid":false,"given":"Changqing","family":"Shi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yufei","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9763-7435","authenticated-orcid":false,"given":"Yicheng","family":"Sui","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuqiao","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haotian","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuzhi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,16]]},"reference":[{"key":"181_CR1","unstructured":"Abadi, M., et al.: Tensorflow: Large-scale machine learning on heterogeneous distributed systems. arXiv preprint arXiv:1603.04467\u00a0(2016)"},{"key":"181_CR2","unstructured":"Adinets, A., Merrill, D.: Onesweep: a faster least significant digit radix sort for GPUs. arXiv preprint arXiv:2206.01784 (2022)"},{"key":"181_CR3","unstructured":"AMD ROCm: A thin wrapper library on top of rocPRIM or CUB. https:\/\/github.com\/ROCmSoftwarePlatform\/hipCUB (2019a)"},{"key":"181_CR4","unstructured":"AMD ROCm: A C++ Runtime API and Kernel Language. https:\/\/github.com\/ROCm-Developer-Tools\/HIP (2019b)"},{"key":"181_CR5","unstructured":"AMD ROCm: AMD ROCm Platform Documentation. https:\/\/rocmdocs.amd.com\/ (2022a)"},{"key":"181_CR6","unstructured":"AMD ROCm. A header-only library providing HIP parallel primitives. https:\/\/github.com\/ROCmSoftwarePlatform\/rocPRIM (2022b)"},{"key":"181_CR7","first-page":"359","volume-title":"\"Thrust: A Productivity-Oriented Library for CUDA.\" GPU Computing Gems","author":"N Bell","year":"2012","unstructured":"Bell, N., Hoberock, J.: \u201cThrust: A Productivity-Oriented Library for CUDA.\u201d GPU Computing Gems, Jade, pp. 359\u2013371. Morgan Kaufmann (2012)","edition":"Jade"},{"key":"181_CR8","doi-asserted-by":"crossref","unstructured":"Cao, C., et al.: clMAGMA: high performance dense linear algebra with OpenCL. In: Proceedings of the International Workshop on OpenCL 2013 & 2014 (2014)","DOI":"10.1145\/2664666.2664667"},{"key":"181_CR9","unstructured":"Chen, T., et al.: Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv preprint arXiv:1512.01274 (2015)"},{"key":"181_CR10","unstructured":"Chetlur, S., et al. cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759 (2014)"},{"key":"181_CR11","volume-title":"Library","author":"NC Cublas","year":"2008","unstructured":"Cublas, N.C.: Library. NVIDIA Corporation, Santa Clara (2008)"},{"issue":"1","key":"181_CR12","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1109\/99.660313","volume":"5","author":"L Dagum","year":"1998","unstructured":"Dagum, L., Menon, R.: OpenMP: an industry standard API for shared-memory programming. IEEE Comput. Sci. Eng. 5(1), 46\u201355 (1998)","journal-title":"IEEE Comput. Sci. Eng."},{"key":"181_CR13","doi-asserted-by":"crossref","unstructured":"Fang, J., Varbanescu, A.L., Sips, H.: A comprehensive performance comparison of CUDA and OpenCL. In: 2011 International Conference on Parallel Processing. IEEE (2011)","DOI":"10.1109\/ICPP.2011.45"},{"key":"181_CR14","unstructured":"Intel: oneAPI Deep Neural Network Library. https:\/\/github.com\/oneapi-src\/oneDNN (2019)"},{"key":"181_CR15","doi-asserted-by":"publisher","first-page":"752","DOI":"10.1007\/s10766-014-0320-y","volume":"43","author":"P J\u00e4\u00e4skel\u00e4inen","year":"2015","unstructured":"J\u00e4\u00e4skel\u00e4inen, P., de La Lama, C.S., Schnetter, E., et al.: pocl: A performance-portable OpenCL implementation. Int. J. Parallel Prog. 43, 752\u2013785 (2015)","journal-title":"Int. J. Parallel Prog."},{"key":"181_CR16","unstructured":"Khan, J., et al.: Miopen: an open source library for deep learning primitives. arXiv preprint arXiv:1910.00078 (2019)"},{"key":"181_CR17","doi-asserted-by":"crossref","unstructured":"Kirk, D.: NVIDIA CUDA software and GPU parallel computing architecture. In: ISMM. Vol. 7 (2007)","DOI":"10.1145\/1296907.1296909"},{"key":"181_CR18","unstructured":"Komatsu, K., et al.: Evaluating performance and portability of OpenCL programs. In: The Fifth International Workshop on Automatic Performance Tuning. Vol. 66 (2010)"},{"issue":"2","key":"181_CR19","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1007\/s42514-022-00095-y","volume":"4","author":"K Lu","year":"2022","unstructured":"Lu, K., Wang, Y., Guo, Y., et al.: MT-3000: a heterogeneous multi-zone processor for HPC. CCF Trans. High Perform. Comput. 4(2), 150\u2013164 (2022)","journal-title":"CCF Trans. High Perform. Comput."},{"key":"181_CR20","doi-asserted-by":"crossref","unstructured":"Mart\u00edn, P.J., Ayuso, L.F., Torres, R., et al.: Algorithmic strategies for optimizing the parallel reduction primitive in CUDA. In: 2012 International Conference on High Performance Computing & Simulation (HPCS). IEEE, pp. 511\u2013519 (2012)","DOI":"10.1109\/HPCSim.2012.6266966"},{"key":"181_CR21","unstructured":"Merrill, D. CUB v1. 5.3: CUDA Unbound, a library of warp-wide, blockwide, and device-wide GPU parallel primitives. NVIDIA Res. (2015)"},{"key":"181_CR23","doi-asserted-by":"crossref","unstructured":"Nichols, D., et al.: MagmaDNN: accelerated deep learning using MAGMA. In: Proceedings of the Practice and Experience in Advanced Research Computing on Rise of the Machines (learning) (2019)","DOI":"10.1145\/3332186.3333047"},{"key":"181_CR24","unstructured":"Paszke, A., et al.: Pytorch: an imperative style, high-performance deep learning library. In: Advances in Neural Information Processing Systems 32 (2019)"},{"issue":"4","key":"181_CR25","first-page":"298","volume":"23","author":"C Pheatt","year":"2008","unstructured":"Pheatt, C.: Intel\u00ae threading building blocks. J. Comput. Sci. Coll. 23(4), 298\u2013298 (2008)","journal-title":"J. Comput. Sci. Coll."},{"issue":"5","key":"181_CR26","doi-asserted-by":"publisher","first-page":"S412","DOI":"10.1137\/15M1026419","volume":"38","author":"K Rupp","year":"2016","unstructured":"Rupp, K., et al.: ViennaCL\u2013-linear algebra library for multi-and many-core architectures. SIAM J. Sci. Comput. 38(5), S412\u2013S439 (2016)","journal-title":"SIAM J. Sci. Comput."},{"issue":"3","key":"181_CR28","doi-asserted-by":"publisher","first-page":"66","DOI":"10.1109\/MCSE.2010.69","volume":"12","author":"JE Stone","year":"2010","unstructured":"Stone, J.E., Gohara, D., Shi, G.: OpenCL: A parallel programming standard for heterogeneous computing systems. Comput. Sci. Eng. 12(3), 66 (2010)","journal-title":"Comput. Sci. Eng."},{"key":"181_CR29","doi-asserted-by":"crossref","unstructured":"Zhang, P., Fang, J., Yang, C., et al.: Mocl: an efficient OpenCL implementation for the matrix-2000 architecture. In: Proceedings of the 15th ACM International Conference on Computing Frontiers, pp. 26\u201335 (2018)","DOI":"10.1145\/3203217.3203244"}],"container-title":["CCF Transactions on High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-024-00181-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42514-024-00181-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-024-00181-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,24]],"date-time":"2024-06-24T07:05:14Z","timestamp":1719212714000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42514-024-00181-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,16]]},"references-count":27,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["181"],"URL":"https:\/\/doi.org\/10.1007\/s42514-024-00181-3","relation":{},"ISSN":["2524-4922","2524-4930"],"issn-type":[{"type":"print","value":"2524-4922"},{"type":"electronic","value":"2524-4930"}],"subject":[],"published":{"date-parts":[[2024,2,16]]},"assertion":[{"value":"6 October 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 January 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 February 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"On behalf of all authors, the corresponding author states that there is no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}