{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T18:36:04Z","timestamp":1758652564827,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100006374","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["2022R1C1C1011021"],"award-info":[{"award-number":["2022R1C1C1011021"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,12]]},"DOI":"10.1145\/3673038.3673045","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T18:29:01Z","timestamp":1723141741000},"page":"1012-1021","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["VitBit: Enhancing Embedded GPU Performance for AI Workloads through Register Operand Packing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-2947-043X","authenticated-orcid":false,"given":"Jaebeom","family":"Jeon","sequence":"first","affiliation":[{"name":"Korea University, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8664-9840","authenticated-orcid":false,"given":"Minseong","family":"Gil","sequence":"additional","affiliation":[{"name":"Korea University, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1582-5212","authenticated-orcid":false,"given":"Junsu","family":"Kim","sequence":"additional","affiliation":[{"name":"Korea University, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4373-3963","authenticated-orcid":false,"given":"Jaeyong","family":"Park","sequence":"additional","affiliation":[{"name":"Korea University, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1706-6850","authenticated-orcid":false,"given":"Gunjae","family":"Koo","sequence":"additional","affiliation":[{"name":"Korea University, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9332-0251","authenticated-orcid":false,"given":"Myung Kuk","family":"Yoon","sequence":"additional","affiliation":[{"name":"Ewha Womans University, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6442-3705","authenticated-orcid":false,"given":"Yunho","family":"Oh","sequence":"additional","affiliation":[{"name":"Korea University, South Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,8,12]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304026"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2020.3047003"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/IOLTS52814.2021.9486694"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TDMR.2022.3159089"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISLPED.2019.8824934"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3316279.3316282"},{"key":"e_1_3_2_2_7_1","volume-title":"Quip: 2-bit quantization of large language models with guarantees. Advances in Neural Information Processing Systems 36","author":"Chee Jerry","year":"2024","unstructured":"Jerry Chee, Yaohui Cai, Volodymyr Kuleshov, and Christopher\u00a0M De\u00a0Sa. 2024. Quip: 2-bit quantization of large language models with guarantees. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_2_8_1","volume-title":"International conference on machine learning. PMLR, 3025\u20133039","author":"Chen Tianlong","year":"2022","unstructured":"Tianlong Chen, Xuxi Chen, Xiaolong Ma, Yanzhi Wang, and Zhangyang Wang. 2022. Coarsening the granularity: Towards structurally sparse lottery tickets. In International conference on machine learning. PMLR, 3025\u20133039."},{"key":"e_1_3_2_2_9_1","volume-title":"2023 USENIX Annual Technical Conference (USENIX ATC 23)","author":"Choi Sangjin","year":"2023","unstructured":"Sangjin Choi, Inhoe Koo, Jeongseob Ahn, Myeongjae Jeon, and Youngjin Kwon. 2023. { EnvPipe} : Performance-preserving { DNN} Training Framework for Saving Energy. In 2023 USENIX Annual Technical Conference (USENIX ATC 23). 851\u2013864."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2021.3061394"},{"key":"e_1_3_2_2_11_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_2_12_1","volume-title":"Training dnns with hybrid block floating point. Advances in Neural Information Processing Systems 31","author":"Drumond Mario","year":"2018","unstructured":"Mario Drumond, Tao Lin, Martin Jaggi, and Babak Falsafi. 2018. Training dnns with hybrid block floating point. Advances in Neural Information Processing Systems 31 (2018)."},{"key":"e_1_3_2_2_13_1","volume-title":"The lottery ticket hypothesis: Finding sparse, trainable neural networks. arXiv preprint arXiv:1803.03635","author":"Frankle Jonathan","year":"2018","unstructured":"Jonathan Frankle and Michael Carbin. 2018. The lottery ticket hypothesis: Finding sparse, trainable neural networks. arXiv preprint arXiv:1803.03635 (2018)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589039"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISVLSI54635.2022.00051"},{"key":"e_1_3_2_2_16_1","unstructured":"Huggingface. 2024. Transformers. https:\/\/github.com\/huggingface\/transformers\/tree\/main\/examples\/pytorch"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3508391"},{"key":"e_1_3_2_2_18_1","volume-title":"Rethinking floating point for deep learning. arXiv preprint arXiv:1811.01721","author":"Johnson Jeff","year":"2018","unstructured":"Jeff Johnson. 2018. Rethinking floating point for deep learning. arXiv preprint arXiv:1811.01721 (2018)."},{"volume-title":"NVIDIA Jetson AGX Orin Series: A Giant Leap Forward for Robotics and Edge AI Applications","author":"Karumbunathan S.","key":"e_1_3_2_2_19_1","unstructured":"Leela\u00a0S. Karumbunathan. 2022. NVIDIA Jetson AGX Orin Series: A Giant Leap Forward for Robotics and Edge AI Applications. NVIDIA Corporation."},{"key":"e_1_3_2_2_20_1","volume-title":"Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems 25","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey\u00a0E Hinton. 2012. Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems 25 (2012)."},{"key":"e_1_3_2_2_21_1","volume-title":"Q-vit: Accurate and fully quantized low-bit vision transformer. Advances in neural information processing systems 35","author":"Li Yanjing","year":"2022","unstructured":"Yanjing Li, Sheng Xu, Baochang Zhang, Xianbin Cao, Peng Gao, and Guodong Guo. 2022. Q-vit: Accurate and fully quantized low-bit vision transformer. Advances in neural information processing systems 35 (2022), 34451\u201334463."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01565"},{"key":"e_1_3_2_2_23_1","volume-title":"RepQuant: Towards Accurate Post-Training Quantization of Large Transformer Models via Scale Reparameterization. arXiv preprint arXiv:2402.05628","author":"Li Zhikai","year":"2024","unstructured":"Zhikai Li, Xuewen Liu, Jing Zhang, and Qingyi Gu. 2024. RepQuant: Towards Accurate Post-Training Quantization of Large Transformer Models via Scale Reparameterization. arXiv preprint arXiv:2402.05628 (2024)."},{"key":"e_1_3_2_2_24_1","volume-title":"Fq-vit: Post-training quantization for fully quantized vision transformer. arXiv preprint arXiv:2111.13824","author":"Lin Yang","year":"2021","unstructured":"Yang Lin, Tianyu Zhang, Peiqin Sun, Zheng Li, and Shuchang Zhou. 2021. Fq-vit: Post-training quantization for fully quantized vision transformer. arXiv preprint arXiv:2111.13824 (2021)."},{"key":"e_1_3_2_2_25_1","volume-title":"Llm-fp4: 4-bit floating-point quantized transformers. arXiv preprint arXiv:2310.16836","author":"Liu Zechun","year":"2023","unstructured":"Shih-yang Liu, Zechun Liu, Xijie Huang, Pingcheng Dong, and Kwang-Ting Cheng. 2023. Llm-fp4: 4-bit floating-point quantized transformers. arXiv preprint arXiv:2310.16836 (2023)."},{"key":"e_1_3_2_2_26_1","unstructured":"NVIDIA. 2024. CUBLAS Library. https:\/\/docs.nvidia.com\/cuda\/pdf\/CUBLAS_Library.pdf"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18072.2020.9218547"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC47752.2019.9042000"},{"key":"e_1_3_2_2_29_1","volume-title":"LRP-QViT: Mixed-Precision Vision Transformer Quantization via Layer-wise Relevance Propagation. arXiv preprint arXiv:2401.11243","author":"Ranjan Navin","year":"2024","unstructured":"Navin Ranjan and Andreas Savakis. 2024. LRP-QViT: Mixed-Precision Vision Transformer Quantization via Layer-wise Relevance Propagation. arXiv preprint arXiv:2401.11243 (2024)."},{"key":"e_1_3_2_2_30_1","volume-title":"Microscaling data formats for deep learning. arXiv preprint arXiv:2310.10537","author":"Rouhani Bita\u00a0Darvish","year":"2023","unstructured":"Bita\u00a0Darvish Rouhani, Ritchie Zhao, Ankit More, Mathew Hall, Alireza Khodamoradi, Summer Deng, Dhruv Choudhary, Marius Cornea, Eric Dellinger, Kristof Denolf, 2023. Microscaling data formats for deep learning. arXiv preprint arXiv:2310.10537 (2023)."},{"key":"e_1_3_2_2_31_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Sekikawa Yusuke","year":"2022","unstructured":"Yusuke Sekikawa and Shingo Yashima. 2022. Bit-pruning: A sparse multiplication-less dot-product. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_32_1","volume-title":"Hybrid 8-bit floating point (HFP8) training and inference for deep neural networks. Advances in neural information processing systems 32","author":"Sun Xiao","year":"2019","unstructured":"Xiao Sun, Jungwook Choi, Chia-Yu Chen, Naigang Wang, Swagath Venkataramani, Vijayalakshmi\u00a0Viji Srinivasan, Xiaodong Cui, Wei Zhang, and Kailash Gopalakrishnan. 2019. Hybrid 8-bit floating point (HFP8) training and inference for deep neural networks. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/GreenCom-CPSCom.2010.102"},{"key":"e_1_3_2_2_34_1","volume-title":"Training deep neural networks with 8-bit floating point numbers. Advances in neural information processing systems 31","author":"Wang Naigang","year":"2018","unstructured":"Naigang Wang, Jungwook Choi, Daniel Brand, Chia-Yu Chen, and Kailash Gopalakrishnan. 2018. Training deep neural networks with 8-bit floating point numbers. Advances in neural information processing systems 31 (2018)."},{"volume-title":"Gpu register packing: Dynamically exploiting narrow-width operands to improve performance. In 2017 IEEE Trustcom\/BigDataSE\/ICESS","author":"Wang Xin","key":"e_1_3_2_2_35_1","unstructured":"Xin Wang and Wei Zhang. 2017. Gpu register packing: Dynamically exploiting narrow-width operands to improve performance. In 2017 IEEE Trustcom\/BigDataSE\/ICESS. IEEE, 745\u2013752."},{"volume-title":"Energy-efficient dnn computing on gpus through register file management. In 2018 IEEE High Performance extreme Computing Conference (HPEC)","author":"Wang Xin","key":"e_1_3_2_2_36_1","unstructured":"Xin Wang and Wei Zhang. 2018. Energy-efficient dnn computing on gpus through register file management. In 2018 IEEE High Performance extreme Computing Conference (HPEC). IEEE, 1\u20137."},{"key":"e_1_3_2_2_37_1","volume-title":"FP6-LLM: Efficiently Serving Large Language Models Through FP6-Centric Algorithm-System Co-Design. arXiv:2401.14112","author":"Xia Haojun","year":"2024","unstructured":"Haojun Xia, Zhen Zheng, Xiaoxia Wu, Shiyang Chen, Zhewei Yao, Stephen Youn, Arash Bakhtiari, Michael Wyatt, Donglin Zhuang, Zhongzhu Zhou, 2024. FP6-LLM: Efficiently Serving Large Language Models Through FP6-Centric Algorithm-System Co-Design. arXiv:2401.14112 (2024)."},{"key":"e_1_3_2_2_38_1","volume-title":"Deephoyer: Learning sparser neural network with differentiable scale-invariant sparsity measures. arXiv preprint arXiv:1908.09979","author":"Yang Huanrui","year":"2019","unstructured":"Huanrui Yang, Wei Wen, and Hai Li. 2019. Deephoyer: Learning sparser neural network with differentiable scale-invariant sparsity measures. arXiv preprint arXiv:1908.09979 (2019)."},{"key":"e_1_3_2_2_39_1","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"You Jie","year":"2023","unstructured":"Jie You, Jae-Won Chung, and Mosharaf Chowdhury. 2023. Zeus: Understanding and optimizing { GPU} energy consumption of { DNN} training. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). 119\u2013139."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19775-8_12"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614309"},{"key":"e_1_3_2_2_42_1","volume-title":"Systems for Machine Learning Workshop at NeurIPS, Vol.\u00a02018","author":"Zhang Jian","year":"2018","unstructured":"Jian Zhang, Jiyan Yang, and Hector Yuen. 2018. Training with low-precision embedding tables. In Systems for Machine Learning Workshop at NeurIPS, Vol.\u00a02018."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00064"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00204"}],"event":{"name":"ICPP '24: the 53rd International Conference on Parallel Processing","acronym":"ICPP '24","location":"Gotland Sweden"},"container-title":["Proceedings of the 53rd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673045","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3673038.3673045","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T17:31:27Z","timestamp":1758648687000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673045"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,12]]},"references-count":44,"alternative-id":["10.1145\/3673038.3673045","10.1145\/3673038"],"URL":"https:\/\/doi.org\/10.1145\/3673038.3673045","relation":{},"subject":[],"published":{"date-parts":[[2024,8,12]]},"assertion":[{"value":"2024-08-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}