{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:19:51Z","timestamp":1784737191426,"version":"3.55.0"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2025,2,14]],"date-time":"2025-02-14T00:00:00Z","timestamp":1739491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,2,14]],"date-time":"2025-02-14T00:00:00Z","timestamp":1739491200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Key Field Research and Development Plan of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"name":"the Key Field Research and Development Plan of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"name":"the Key Field Research and Development Plan of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"name":"the Key Field Research and Development Plan of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"name":"the Key Field Research and Development Plan of Guangdong Province","award":["2022B0101070001"],"award-info":[{"award-number":["2022B0101070001"]}]},{"DOI":"10.13039\/501100003453","name":"Natural Science Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2024A1515010204"],"award-info":[{"award-number":["2024A1515010204"]}],"id":[{"id":"10.13039\/501100003453","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003453","name":"Natural Science Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2024A1515010204"],"award-info":[{"award-number":["2024A1515010204"]}],"id":[{"id":"10.13039\/501100003453","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003453","name":"Natural Science Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2024A1515010204"],"award-info":[{"award-number":["2024A1515010204"]}],"id":[{"id":"10.13039\/501100003453","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003453","name":"Natural Science Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2024A1515010204"],"award-info":[{"award-number":["2024A1515010204"]}],"id":[{"id":"10.13039\/501100003453","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003453","name":"Natural Science Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2024A1515010204"],"award-info":[{"award-number":["2024A1515010204"]}],"id":[{"id":"10.13039\/501100003453","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-06993-6","type":"journal-article","created":{"date-parts":[[2025,2,14]],"date-time":"2025-02-14T06:51:58Z","timestamp":1739515918000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An efficient quantized GEMV implementation for large language models inference with matrix core"],"prefix":"10.1007","volume":"81","author":[{"given":"Yu","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rong","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yijie","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhanyu","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,2,14]]},"reference":[{"key":"6993_CR1","unstructured":"Lyu C, Xu J, Wang L (2023) New trends in machine translation using large language models: case examples with chatgpt. arXiv preprint arXiv:2305.01181"},{"key":"6993_CR2","unstructured":"Kurisinkel LJ, Chen NF (2023) Llm based multi-document summarization exploiting main-event biased monotone submodular content extraction. arXiv preprint arXiv:2310.03414"},{"key":"6993_CR3","unstructured":"Minaee S, Mikolov T, Nikzad N, Chenaghlu M, Socher R, Amatriain X, Gao J (2024) Large language models: a survey. arXiv preprint arXiv:2402.06196"},{"key":"6993_CR4","doi-asserted-by":"crossref","unstructured":"Bansal G, Chamola V, Hussain A, Guizani M, Niyato D (2024) Transforming conversations with AI-a comprehensive study of chatgpt. Cogn Comput, 1\u201324","DOI":"10.1007\/s12559-023-10236-2"},{"key":"6993_CR5","doi-asserted-by":"crossref","unstructured":"Hadi MU, Qureshi R, Shah A, Irfan M, Zafar A, Shaikh MB, Akhtar N, Wu J, Mirjalili S et al (2023) A survey on large language models: applications, challenges, limitations, and practical usage. Authorea Preprints","DOI":"10.36227\/techrxiv.23589741.v1"},{"key":"6993_CR6","unstructured":"Valero-Lara P, Huante A, Lail MA, Godoy WF, Teranishi K, Balaprakash P, Vetter JS (2023) Comparing llama-2 and GPT-3 LLMS for HPC kernels generation. arXiv preprint arXiv:2309.07103"},{"key":"6993_CR7","unstructured":"Touvron H, Lavril T, Izacard G, Martinet X, Lachaux M-A, Lacroix T, Rozi\u00e8re B, Goyal N, Hambro E, Azhar F et al (2023) Llama: open and efficient foundation language models. arXiv preprint arXiv:2302.13971"},{"key":"6993_CR8","unstructured":"Frantar E, Ashkboos S, Hoefler T, Alistarh D (2023) Optq: Accurate quantization for generative pre-trained transformers. International Conference on Learning Representations"},{"key":"6993_CR9","unstructured":"Lin J, Tang J, Tang H, Yang S, Dang X, Han S (2023) AWQ: Activation-aware weight quantization for llm compression and acceleration. arXiv preprint arXiv:2306.00978"},{"key":"6993_CR10","unstructured":"Xu M, Xu YL, Mandic DP (2023) Tensorgpt: efficient compression of the embedding layer in LLMS based on the tensor-train decomposition. arXiv preprint arXiv:2307.00526"},{"key":"6993_CR11","doi-asserted-by":"crossref","unstructured":"Li L, Zhang Y, Chen L (2023) Prompt distillation for efficient LLM-based recommendation. In: Proceedings of the 32nd ACM International Conference on Information and Knowledge Management, 1348\u20131357","DOI":"10.1145\/3583780.3615017"},{"issue":"4","key":"6993_CR12","doi-asserted-by":"publisher","first-page":"485","DOI":"10.1109\/JPROC.2020.2976475","volume":"108","author":"L Deng","year":"2020","unstructured":"Deng L, Li G, Han S, Shi L, Xie Y (2020) Model compression and hardware acceleration for neural networks: a comprehensive survey. Proceed IEEE 108(4):485\u2013532","journal-title":"Proceed IEEE"},{"key":"6993_CR13","doi-asserted-by":"crossref","unstructured":"He Y, Lin J, Liu Z, Wang H, Li L-J, Han S (2018) AMC: automl for model compression and acceleration on mobile devices. In: Proceedings of the European Conference on Computer Vision (ECCV), 784\u2013800","DOI":"10.1007\/978-3-030-01234-2_48"},{"key":"6993_CR14","unstructured":"Dettmers T, Svirschevski R, Egiazarian V, Kuznedelev D, Frantar E, Ashkboos S, Borzunov A, Hoefler T, Alistarh D (2023) SPQR: A sparse-quantized representation for near-lossless LLM weight compression. arXiv preprint arXiv:2306.03078"},{"key":"6993_CR15","first-page":"30318","volume":"35","author":"T Dettmers","year":"2022","unstructured":"Dettmers T, Lewis M, Belkada Y, Zettlemoyer L (2022) Gpt3. int8 (): 8-bit matrix multiplication for transformers at scale. Adv Neural Inf Process Syst 35:30318\u201330332","journal-title":"Adv Neural Inf Process Syst"},{"key":"6993_CR16","doi-asserted-by":"crossref","unstructured":"Nagel M, Baalen Mv, Blankevoort T, Welling M (2019) Data-free quantization through weight equalization and bias correction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 1325\u20131334","DOI":"10.1109\/ICCV.2019.00141"},{"key":"6993_CR17","doi-asserted-by":"crossref","unstructured":"Zhong Y, Lin M, Nan G, Liu J, Zhang B, Tian Y, Ji R (2022) Intraq: Learning synthetic images with intra-class heterogeneity for zero-shot network quantization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 12339\u201312348","DOI":"10.1109\/CVPR52688.2022.01202"},{"key":"6993_CR18","doi-asserted-by":"crossref","unstructured":"Park S, Jang Y, Park E. (2022) Symmetry regularization and saturating nonlinearity for robust quantization. In: European Conference on Computer Vision, 206\u2013222. Springer","DOI":"10.1007\/978-3-031-20083-0_13"},{"key":"6993_CR19","doi-asserted-by":"crossref","unstructured":"Zhu X, Li J, Liu Y, Ma C, Wang W (2023) A survey on model compression for large language models. arXiv preprint arXiv:2308.07633","DOI":"10.1162\/tacl_a_00704"},{"key":"6993_CR20","doi-asserted-by":"crossref","unstructured":"Shen M, Liang F, Gong R, Li Y, Li C, Lin C, Yu F, Yan J, Ouyang W (2021) Once quantization-aware training: high performance extremely low-bit architecture search. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 5340\u20135349","DOI":"10.1109\/ICCV48922.2021.00529"},{"key":"6993_CR21","unstructured":"Pegolotti T, Frantar E, Alistarh D, P\u00fcschel M (2023) Generating efficient kernels for quantized inference on large language models. In: Workshop on Efficient Systems for Foundation Models@ ICML2023"},{"key":"6993_CR22","doi-asserted-by":"crossref","unstructured":"Wang F, Shen M (2023) Automatic kernel generation for large language models on deep learning accelerators. In: 2023 IEEE\/ACM International Conference on Computer Aided Design (ICCAD), 1\u20139. IEEE","DOI":"10.1109\/ICCAD57390.2023.10323944"},{"key":"6993_CR23","unstructured":"AMD: rocBLAS: ROCm Basic Linear Algebra Subprograms (BLAS) library. https:\/\/github.com\/ROCm\/rocBLAS (2023)"},{"key":"6993_CR24","unstructured":"NVIDIA: cuBLAS: Basic Linear Algebra on NVIDIA GPUs. https:\/\/developer.nvidia.com\/cublas (2023)"},{"key":"6993_CR25","unstructured":"Park G, Park B, Kim M, Lee S, Kim J, Kwon B, Kwon S.J, Kim B, Lee Y, Lee D (2022) Lut-gemm: Quantized matrix multiplication based on luts for efficient inference in large-scale generative language models. arXiv preprint arXiv:2206.09557"},{"key":"6993_CR26","unstructured":"Exllama: A more memory-efficient rewrite of the HF transformers implementation of llama for use with quantized weights. https:\/\/github.com\/turboderp\/exllama (2023)"},{"key":"6993_CR27","unstructured":"AMD-lab-notes: AMD matrix cores. https:\/\/gpuopen.com\/learn\/amd-lab-notes\/amd-lab-notes-matrix-cores-readme\/ (2023)"},{"key":"6993_CR28","unstructured":"AMD: AMD Instinst MI200 Series Accelerator. https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/instinct-tech-docs\/instinct-mi200-datasheet.pdf (2023)"},{"key":"6993_CR29","unstructured":"AMD: AMD Instinct MI200 Instruction SetArchitecture. https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/instinct-tech-docs\/instruction-set-architectures\/instinct-mi200-cdna2-instruction-set-architecture.pdf (2023)"},{"key":"6993_CR30","doi-asserted-by":"crossref","unstructured":"Jeon Y, Park B, Kwon SJ, Kim B, Yun J, Lee D (2020) Biqgemm: matrix multiplication with lookup table for binary-coding-based quantized dnns. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, 1\u201314. IEEE","DOI":"10.1109\/SC41405.2020.00099"},{"key":"6993_CR31","doi-asserted-by":"crossref","unstructured":"Gholami A, Kim S, Dong Z, Yao Z, Mahoney MW, Keutzer K (2022) A survey of quantization methods for efficient neural network inference. Low-Power Computer Vision. Chapman and Hall\/CRC, USA, 291\u2013326","DOI":"10.1201\/9781003162810-13"},{"key":"6993_CR32","unstructured":"Zhao Y, Lin C-Y, Zhu K, Ye Z, Chen L, Zheng S, Ceze L, Krishnamurthy A, Chen T, Kasikci B (2023) Atom: Low-bit quantization for efficient and accurate LLM serving. arXiv preprint arXiv:2310.19102"},{"key":"6993_CR33","unstructured":"Hong K, Dai G, Xu J, Mao Q, Li X, Liu J, Chen K, Dong H, Wang Y (2023) Flashdecoding++: Faster large language model inference on GPUS. arXiv preprint arXiv:2311.01282"},{"key":"6993_CR34","unstructured":"Kim T, Lee J, Ahn D, Kim S, Choi J, Kim M, Kim H (2024) Quick: Quantization-aware interleaving and conflict-free kernel for efficient LLM inference. arXiv preprint arXiv:2402.10076"},{"key":"6993_CR35","doi-asserted-by":"crossref","unstructured":"Mukunoki D, Imamura T, Takahashi D (2015) Fast implementation of general matrix-vector multiplication (gemv) on kepler GPUS. In: 2015 23rd Euromicro International Conference on Parallel, Distributed, and Network-Based Processing, 642\u2013650. IEEE","DOI":"10.1109\/PDP.2015.66"},{"issue":"7","key":"6993_CR36","doi-asserted-by":"publisher","first-page":"1566","DOI":"10.1016\/j.jvcir.2014.06.002","volume":"25","author":"J Yin","year":"2014","unstructured":"Yin J, Yu H, Xu W, Wang Y, Tian Z, Zhang Y, Chen B (2014) Highly parallel Gemv with register blocking method on GPU architecture. J Vis Commun Image Represent 25(7):1566\u20131573","journal-title":"J Vis Commun Image Represent"},{"key":"6993_CR37","doi-asserted-by":"crossref","unstructured":"Cheng J, Liu X, Cao Y, Zhang W, Han Z, Peng B, Liu Y, Zhang D, Han Y, Xu X et al (2022) Cache-major: A hardware architecture and scheduling policy for improving dram access efficiency in Gemv. In: 2022 IEEE 16th International Conference on Solid-State and Integrated Circuit Technology (ICSICT), 1\u20133. IEEE","DOI":"10.1109\/ICSICT55466.2022.9963310"},{"key":"6993_CR38","unstructured":"AMD: rocWMMA: ROCm Wavefront-level Matrix Multiply and Accumulate. https:\/\/github.com\/ROCmSoftwarePlatform\/rocWMMA (2023)"},{"key":"6993_CR39","doi-asserted-by":"crossref","unstructured":"Markidis S, Der Chien SW, Laure E, Peng IB, Vetter JS (2018) Nvidia tensor core programmability, performance and precision. In: 2018 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW), 522\u2013531. IEEE","DOI":"10.1109\/IPDPSW.2018.00091"},{"key":"6993_CR40","doi-asserted-by":"crossref","unstructured":"Ashkboos S, Markov I, Frantar E, Zhong T, Wang X, Ren J, Hoefler T, Alistarh D (2023) Towards end-to-end 4-bit inference on generative large language models. arXiv preprint arXiv:2310.09259","DOI":"10.18653\/v1\/2024.emnlp-main.197"},{"issue":"11","key":"6993_CR41","doi-asserted-by":"publisher","first-page":"13393","DOI":"10.1007\/s11227-022-04336-3","volume":"78","author":"Z Yang","year":"2022","unstructured":"Yang Z, Lu L, Wang R (2022) A batched GEMM optimization framework for deep learning. J Supercomput 78(11):13393\u201313408","journal-title":"J Supercomput"},{"issue":"17","key":"6993_CR42","doi-asserted-by":"publisher","first-page":"19547","DOI":"10.1007\/s11227-023-05399-6","volume":"79","author":"Y Guo","year":"2023","unstructured":"Guo Y, Lu L, Zhu S (2023) Novel accelerated methods for convolution neural network with matrix core. J Supercomput 79(17):19547\u201319573","journal-title":"J Supercomput"},{"key":"6993_CR43","doi-asserted-by":"crossref","unstructured":"Lu G, Xu D, Wang N, Zhang X, Zhen D, Lei H, Bai Y, Kong D, Ruan H, Chi Z et al (2020) A design of 16tops efficient gemm module in deep learning accelerator. In: 2020 IEEE International Conference on Integrated Circuits, Technologies and Applications (ICTA), 59\u201360. IEEE","DOI":"10.1109\/ICTA50426.2020.9332090"},{"key":"6993_CR44","doi-asserted-by":"crossref","unstructured":"Smith A, James N (2022) Amd $$\\text{instinct}^{{\\rm TM}}$$ mi200 series accelerator and node architectures. In: HCS, 1\u201323","DOI":"10.1109\/HCS55958.2022.9895477"},{"issue":"6","key":"6993_CR45","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3653304","volume":"18","author":"J Yang","year":"2024","unstructured":"Yang J, Jin H, Tang R, Han X, Feng Q, Jiang H, Zhong S, Yin B, Hu X (2024) Harnessing the power of LLMS in practice: a survey on chatgpt and beyond. ACM Trans Knowl Discov Data 18(6):1\u201332","journal-title":"ACM Trans Knowl Discov Data"},{"issue":"6","key":"6993_CR46","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11704-024-40231-1","volume":"18","author":"L Wang","year":"2024","unstructured":"Wang L, Ma C, Feng X, Zhang Z, Yang H, Zhang J, Chen Z, Tang J, Chen X, Lin Y et al (2024) A survey on large language model based autonomous agents. Front Comput Sci 18(6):1\u201326","journal-title":"Front Comput Sci"},{"key":"6993_CR47","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown T, Mann B, Ryder N, Subbiah M, Kaplan JD, Dhariwal P, Neelakantan A, Shyam P, Sastry G, Askell A et al (2020) Language models are few-shot learners. Adv Neural Inf Process Syst 33:1877\u20131901","journal-title":"Adv Neural Inf Process Syst"},{"issue":"10","key":"6993_CR48","doi-asserted-by":"publisher","first-page":"1370","DOI":"10.1016\/j.jpdc.2008.05.014","volume":"68","author":"S Che","year":"2008","unstructured":"Che S, Boyer M, Meng J, Tarjan D, Sheaffer JW, Skadron K (2008) A performance study of general-purpose applications on graphics processors using Cuda. J Parallel Distrib Comput 68(10):1370\u20131380","journal-title":"J Parallel Distrib Comput"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-06993-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-06993-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-06993-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,14]],"date-time":"2025-02-14T06:52:30Z","timestamp":1739515950000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-06993-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,14]]},"references-count":48,"journal-issue":{"issue":"3","published-online":{"date-parts":[[2025,2]]}},"alternative-id":["6993"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-06993-6","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2,14]]},"assertion":[{"value":"27 January 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 February 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"496"}}