{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T15:28:42Z","timestamp":1759332522693,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,5]],"date-time":"2024-08-05T00:00:00Z","timestamp":1722816000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,5]]},"DOI":"10.1145\/3665314.3670806","type":"proceedings-article","created":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T19:31:18Z","timestamp":1725910278000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Hardware-friendly Hessian-driven Row-wise Quantization and FPGA Acceleration for Transformer-based Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5087-4582","authenticated-orcid":false,"given":"Woohong","family":"Byun","sequence":"first","affiliation":[{"name":"School of Electrical and Computer Engineering, Georgia Institute of Technology, Atlanta, GA, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8030-2356","authenticated-orcid":false,"given":"Jongseok","family":"Woo","sequence":"additional","affiliation":[{"name":"School of Electrical and Computer Engineering, Georgia Institute of Technology, Atlanta, GA, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8894-3390","authenticated-orcid":false,"given":"Saibal","family":"Mukhopadhyay","sequence":"additional","affiliation":[{"name":"School of Electrical and Computer Engineering, Georgia Institute of Technology, Atlanta, GA, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,9,9]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding,\" arXiv:1810.04805","author":"Devlin J.","year":"2018","unstructured":"J. Devlin et al., \"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding,\" arXiv:1810.04805, 2018."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"W. Byun S. Mukhopadhyay \"Hessian-based parameter quantization method for BERT \" 2023 IEEE International Midwest Symposium on Circuits and Systems (MWSCAS) 2023.","DOI":"10.1109\/MWSCAS57524.2023.10406106"},{"key":"e_1_3_2_1_3_1","volume-title":"Q-BERT: Hessian Based Ultra Low Precision Quantization of BERT,\" In AAAI Conference on Artificial Intelligence","author":"Shen S.","year":"2020","unstructured":"S. Shen et al., \"Q-BERT: Hessian Based Ultra Low Precision Quantization of BERT,\" In AAAI Conference on Artificial Intelligence, 2020."},{"key":"e_1_3_2_1_4_1","volume-title":"Q8BERT: Quantized 8bit BERT,\" arXiv:1910.06188","author":"Zafrir O.","year":"2019","unstructured":"O. Zafrir et al., \"Q8BERT: Quantized 8bit BERT,\" arXiv:1910.06188, 2019."},{"key":"e_1_3_2_1_5_1","volume-title":"Gobo: Quantizing attention-based nlp models for low latency and energy efficient inference,\" arXiv:2005.03842","author":"Zadeh A. H.","year":"2020","unstructured":"A. H. Zadeh and A. Moshovos, \"Gobo: Quantizing attention-based nlp models for low latency and energy efficient inference,\" arXiv:2005.03842, 2020."},{"key":"e_1_3_2_1_6_1","article-title":"Robust processing-in-memory with multi-bit reram using hessian-driven mixed-precision computation","author":"Dash S.","year":"2021","unstructured":"S. Dash et al., \"Robust processing-in-memory with multi-bit reram using hessian-driven mixed-precision computation,\" IEEE Trans. Comput.-Aided Des. Integr. Circuits Syst., 2021.","journal-title":"IEEE Trans. Comput.-Aided Des. Integr. Circuits Syst."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00069"},{"key":"e_1_3_2_1_8_1","volume-title":"Swifttron: An efficient hardware accelerator for quantized transformers,\" arXiv preprint arXiv:2304.03986","author":"Marchisio A.","year":"2023","unstructured":"A. Marchisio et al., \"Swifttron: An efficient hardware accelerator for quantized transformers,\" arXiv preprint arXiv:2304.03986, 2023."},{"key":"e_1_3_2_1_9_1","volume-title":"Int. Conf. Mach. Learn","author":"Kim S.","year":"2021","unstructured":"S. Kim et al., \"I-bert: Integer-only bert quantization,\" in Proc. Int. Conf. Mach. Learn, 2021."},{"key":"e_1_3_2_1_10_1","first-page":"509","volume-title":"Conf. Empirical Methods Natural Lang. Process. (EMNLP)","author":"Zhang W.","year":"2020","unstructured":"W. Zhang et al., \"TernaryBERT: Distillation-aware ultra-low bit bert,\" in Proc. Conf. Empirical Methods Natural Lang. Process. (EMNLP), 2020, pp. 509--521."},{"key":"e_1_3_2_1_11_1","volume-title":"GLUE: A multi-task benchmark and analysis platform for natural language understanding","author":"Wang A.","year":"2018","unstructured":"A. Wang et al., \"GLUE: A multi-task benchmark and analysis platform for natural language understanding,\" 2018, arXiv:1804.07461."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3370748.3406567"},{"key":"e_1_3_2_1_13_1","first-page":"227","volume-title":"ACM\/SIGDA Int. Symp. Field-Programmable Gate Arrays","author":"Khan H.","year":"2021","unstructured":"H. Khan et al., \"NPE: An FPGA-based Overlay Processor for Natural Language Processing,\" in Proc. ACM\/SIGDA Int. Symp. Field-Programmable Gate Arrays, 2021, pp. 227--227."},{"key":"e_1_3_2_1_14_1","volume-title":"Hardware acceleration of fully quantized bert for efficient natural language processing,\" arXiv preprint arXiv:2103.02800","author":"Liu Z.","year":"2021","unstructured":"Z. Liu et al., \"Hardware acceleration of fully quantized bert for efficient natural language processing,\" arXiv preprint arXiv:2103.02800, 2021."},{"key":"e_1_3_2_1_15_1","volume-title":"Conf. on Field-Programmable Logic and Applications (FPL)","author":"Han Y.","year":"2023","unstructured":"Y. Han et al., \"HPTA: A High Performance Transformer Accelerator Based on FPGA,\" Int. Conf. on Field-Programmable Logic and Applications (FPL), 2023."},{"key":"e_1_3_2_1_16_1","volume-title":"Symp. High- Perform. Comput. Architecture","author":"Hojabr R.","year":"2021","unstructured":"R. Hojabr et al., \"SPAGHETTI: Streaming accelerators for highly sparse GEMM on FPGAs,\" IEEE Int. Symp. High- Perform. Comput. Architecture, 2021."},{"key":"e_1_3_2_1_17_1","first-page":"11875","volume-title":"38th Int. Conf. Mach. Learn. (ICML)","author":"Yao Z.","year":"2021","unstructured":"Z. Yao et al., \"HAWQV3: dyadic neural network quantization,\" in Proc. 38th Int. Conf. Mach. Learn. (ICML), pp. 11875--11886, 2021."},{"key":"e_1_3_2_1_18_1","volume-title":"ACM\/SIGDA Int. Symp. Field-Programmable Gate","author":"Sun M.","year":"2022","unstructured":"M. Sun et al., \"FILM-QNN: Efficient FPGA Acceleration of Deep Neural Networks with Intra-Layer, Mixed-Precision Quantization,\" in Proc. ACM\/SIGDA Int. Symp. Field-Programmable Gate, 2022."},{"key":"e_1_3_2_1_19_1","volume-title":"Conf. on Machine Learning Technologies (ICMLT)","author":"Okado K.","year":"2022","unstructured":"K. Okado et al., \"Channel-wise quantization without accuracy degradation using loss analysis,\" 2022 7th Int. Conf. on Machine Learning Technologies (ICMLT), 2022"},{"key":"e_1_3_2_1_20_1","volume-title":"Smoothquant: Accurate and efficient post-training quantization for large language models,\" arXiv preprint arXiv:2211.10438","author":"Xiao G.","year":"2022","unstructured":"G. Xiao et al., \"Smoothquant: Accurate and efficient post-training quantization for large language models,\" arXiv preprint arXiv:2211.10438, 2022."},{"key":"e_1_3_2_1_21_1","first-page":"2675","volume-title":"Appl. Comput. Vis.","author":"Hong C.","year":"2022","unstructured":"C. Hong et al., \"DAQ: Channel-wise distribution-aware quantization for deep image super-resolution networks,\" IEEE\/CVF Winter Conf. Appl. Comput. Vis., Jan. 2022, pp. 2675--2684."},{"volume-title":"Integer quantization for deep learning inference: Principles and empirical evaluation,\" arXiv preprint arXiv:2004.09602","year":"2020","key":"e_1_3_2_1_22_1","unstructured":"Hao Wu et al., \"Integer quantization for deep learning inference: Principles and empirical evaluation,\" arXiv preprint arXiv:2004.09602, 2020."},{"key":"e_1_3_2_1_23_1","volume-title":"Roberta: A robustly optimized bert pretraining approach,\" arXiv:1907.11692","author":"Liu Y.","year":"2019","unstructured":"Y. Liu et al., \"Roberta: A robustly optimized bert pretraining approach,\" arXiv:1907.11692, 2019."},{"issue":"140","key":"e_1_3_2_1_24_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel C.","year":"2020","unstructured":"C. Raffel et al., \"Exploring the limits of transfer learning with a unified text-to-text transformer,\" J. of Mach. Learn. Res., vol. 21, no. 140, pp. 1--67, 2020.","journal-title":"J. of Mach. Learn. Res."},{"key":"e_1_3_2_1_25_1","volume-title":"ALBERT: A lite bert for self-supervised learning of language representations","author":"Lan Z.","year":"2019","unstructured":"Z. Lan et al., \"ALBERT: A lite bert for self-supervised learning of language representations,\" 2019, arXiv:1909.11942"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"}],"event":{"name":"ISLPED '24: 29th ACM\/IEEE International Symposium on Low Power Electronics and Design","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE CAS","IEEE EDA"],"location":"Newport Beach CA USA","acronym":"ISLPED '24"},"container-title":["Proceedings of the 29th ACM\/IEEE International Symposium on Low Power Electronics and Design"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3665314.3670806","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3665314.3670806","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:58:34Z","timestamp":1750294714000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3665314.3670806"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,5]]},"references-count":26,"alternative-id":["10.1145\/3665314.3670806","10.1145\/3665314"],"URL":"https:\/\/doi.org\/10.1145\/3665314.3670806","relation":{},"subject":[],"published":{"date-parts":[[2024,8,5]]},"assertion":[{"value":"2024-09-09","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}