{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T16:09:48Z","timestamp":1781885388280,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,6,23]],"date-time":"2024-06-23T00:00:00Z","timestamp":1719100800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,6,23]]},"DOI":"10.1145\/3649329.3657314","type":"proceedings-article","created":{"date-parts":[[2024,11,7]],"date-time":"2024-11-07T19:27:22Z","timestamp":1731007642000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["Genetic Quantization-Aware Approximation for Non-Linear Operations in Transformers"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3975-286X","authenticated-orcid":false,"given":"Pingcheng","family":"Dong","sequence":"first","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5372-5863","authenticated-orcid":false,"given":"Yonghao","family":"Tan","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4543-2179","authenticated-orcid":false,"given":"Dong","family":"Zhang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1045-6698","authenticated-orcid":false,"given":"Tianwei","family":"Ni","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6788-1593","authenticated-orcid":false,"given":"Xuejiao","family":"Liu","sequence":"additional","affiliation":[{"name":"AI Chip Center for Emerging Smart System (ACCESS), Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1170-3858","authenticated-orcid":false,"given":"Yu","family":"Liu","sequence":"additional","affiliation":[{"name":"AI Chip Center for Emerging Smart System (ACCESS), Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0256-9545","authenticated-orcid":false,"given":"Peng","family":"Luo","sequence":"additional","affiliation":[{"name":"AI Chip Center for Emerging Smart System (ACCESS), Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4190-006X","authenticated-orcid":false,"given":"Luhong","family":"Liang","sequence":"additional","affiliation":[{"name":"AI Chip Center for Emerging Smart System (ACCESS), Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1997-0843","authenticated-orcid":false,"given":"Shih-Yang","family":"Liu","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5508-5433","authenticated-orcid":false,"given":"Xijie","family":"Huang","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6918-4088","authenticated-orcid":false,"given":"Huaiyu","family":"Zhu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9335-4291","authenticated-orcid":false,"given":"Yun","family":"Pan","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7554-7938","authenticated-orcid":false,"given":"Fengwei","family":"An","sequence":"additional","affiliation":[{"name":"Southern University of Science and Technology, Shenzhen, Guangdong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3885-4912","authenticated-orcid":false,"given":"Kwang-Ting","family":"Cheng","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,11,7]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"2","volume-title":"Proceedings of naacL-HLT","volume":"1","author":"Jacob","year":"2019","unstructured":"Jacob Devlin et al. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of naacL-HLT, volume 1, page 2, 2019."},{"key":"e_1_3_2_1_2_1","first-page":"10012","volume-title":"Proceedings of the IEEE\/CVF international conference on computer vision","author":"Ze","year":"2021","unstructured":"Ze Liu et al. Swin transformer: Hierarchical vision transformer using shifted windows. In Proceedings of the IEEE\/CVF international conference on computer vision, pages 10012--10022, 2021."},{"key":"e_1_3_2_1_3_1","first-page":"12077","article-title":"Segformer: Simple and efficient design for semantic segmentation with transformers","volume":"34","author":"Enze Xie","year":"2021","unstructured":"Enze Xie et al. Segformer: Simple and efficient design for semantic segmentation with transformers. Advances in Neural Information Processing Systems, 34:12077--12090, 2021.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_4_1","first-page":"17302","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","author":"Han","year":"2023","unstructured":"Han Cai et al. Efficientvit: Lightweight multi-scale attention for high-resolution dense prediction. In Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pages 17302--17313, October 2023."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-021-3590-1"},{"key":"e_1_3_2_1_6_1","first-page":"5506","volume-title":"International conference on machine learning","author":"Sehoon","year":"2021","unstructured":"Sehoon Kim et al. I-bert: Integer-only bert quantization. In International conference on machine learning, pages 5506--5518. PMLR, 2021."},{"key":"e_1_3_2_1_7_1","first-page":"21813","volume-title":"Proceedings of the 40th International Conference on Machine Learning","volume":"202","author":"Shih-Yang","year":"2023","unstructured":"Shih-Yang Liu et al. Oscillation-free quantization for low-bit vision transformers. In Proceedings of the 40th International Conference on Machine Learning, volume 202, pages 21813--21824. PMLR, 23--29 Jul 2023."},{"key":"e_1_3_2_1_8_1","first-page":"592","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing","author":"Shih","year":"2023","unstructured":"Shih-yang Liu et al. Llm-fp4: 4-bit floating-point quantized transformers. In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pages 592--605, 2023."},{"key":"e_1_3_2_1_9_1","article-title":"Multcim: Digital computing-in-memory-based multimodal transformer accelerator with attention-token-bit hybrid sparsity","author":"Fengbin Tu","year":"2023","unstructured":"Fengbin Tu et al. Multcim: Digital computing-in-memory-based multimodal transformer accelerator with attention-token-bit hybrid sparsity. IEEE Journal of Solid-State Circuits, 2023.","journal-title":"IEEE Journal of Solid-State Circuits"},{"key":"e_1_3_2_1_10_1","first-page":"469","volume-title":"2021 58th ACM\/IEEE Design Automation Conference (DAC)","author":"Jacob R","year":"2021","unstructured":"Jacob R Stevens et al. Softermax: Hardware\/software co-design of an efficient softmax for transformers. In 2021 58th ACM\/IEEE Design Automation Conference (DAC), pages 469--474. IEEE, 2021."},{"key":"e_1_3_2_1_11_1","first-page":"577","volume-title":"2023 59th ACM\/IEEE Design Automation Conference (DAC)","author":"Joonsang","year":"2022","unstructured":"Joonsang Yu et al. Nn-lut: neural approximation of non-linear operations for efficient transformer inference. In 2023 59th ACM\/IEEE Design Automation Conference (DAC), pages 577--582, 2022."},{"key":"e_1_3_2_1_12_1","first-page":"1","volume-title":"2023 60th ACM\/IEEE Design Automation Conference (DAC)","author":"Janghyeon","year":"2023","unstructured":"Janghyeon Kim et al. Range-invariant approximation of non-linear operations for efficient bert fine-tuning. In 2023 60th ACM\/IEEE Design Automation Conference (DAC), pages 1--6. IEEE, 2023."},{"key":"e_1_3_2_1_13_1","first-page":"9295","volume-title":"International Conference on Machine Learning","author":"Xijie","year":"2022","unstructured":"Xijie Huang et al. Sdq: Stochastic differentiable quantization with mixed precision. In International Conference on Machine Learning, pages 9295--9309. PMLR, 2022."},{"issue":"8","key":"e_1_3_2_1_14_1","first-page":"3079","article-title":"A tiny accelerator for mixed-bit sparse cnn based on efficient fetch method of simo spad","volume":"70","author":"Xianghong Hu","year":"2023","unstructured":"Xianghong Hu et al. A tiny accelerator for mixed-bit sparse cnn based on efficient fetch method of simo spad. IEEE Transactions on Circuits and Systems II: Express Briefs, 70(8):3079--3083, 2023.","journal-title":"IEEE Transactions on Circuits and Systems II: Express Briefs"},{"key":"e_1_3_2_1_15_1","first-page":"2704","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","author":"Benoit","year":"2018","unstructured":"Benoit Jacob et al. Quantization and training of neural networks for efficient integer-arithmetic-only inference. In Proceedings of the IEEE conference on computer vision and pattern recognition, pages 2704--2713, 2018."},{"key":"e_1_3_2_1_16_1","volume-title":"Attention is all you need. Advances in neural information processing systems, 30","author":"Ashish Vaswani","year":"2017","unstructured":"Ashish Vaswani et al. Attention is all you need. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_17_1","first-page":"5961","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Dongchen","year":"2023","unstructured":"Dongchen Han et al. Flatten transformer: Vision transformer using focused linear attention. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pages 5961--5971, 2023."},{"key":"e_1_3_2_1_18_1","volume-title":"Estimating or propagating gradients through stochastic neurons for conditional computation. arXiv preprint arXiv:1308.3432","author":"Yoshua Bengio","year":"2013","unstructured":"Yoshua Bengio et al. Estimating or propagating gradients through stochastic neurons for conditional computation. arXiv preprint arXiv:1308.3432, 2013."},{"key":"e_1_3_2_1_19_1","volume-title":"8th International Conference on Learning Representations, ICLR 2020","author":"Steven K.","year":"2020","unstructured":"Steven K. Esser et al. Learned step size quantization. In 8th International Conference on Learning Representations, ICLR 2020, Addis Ababa, Ethiopia, April 26-30, 2020, 2020."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/1090.001.0001"},{"key":"e_1_3_2_1_21_1","first-page":"3213","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","author":"Marius","year":"2016","unstructured":"Marius Cordts et al. The cityscapes dataset for semantic urban scene understanding. In Proceedings of the IEEE conference on computer vision and pattern recognition, pages 3213--3223, 2016."},{"key":"e_1_3_2_1_22_1","first-page":"2380","volume-title":"Proceedings of the 30th ACM International Conference on Multimedia","author":"Dong","year":"2022","unstructured":"Dong Zhang et al. Graph reasoning transformer for image parsing. In Proceedings of the 30th ACM International Conference on Multimedia, pages 2380--2389, 2022."}],"event":{"name":"DAC '24: 61st ACM\/IEEE Design Automation Conference","location":"San Francisco CA USA","acronym":"DAC '24","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE-CEDA","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 61st ACM\/IEEE Design Automation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649329.3657314","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3649329.3657314","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:56Z","timestamp":1750295876000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3649329.3657314"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,23]]},"references-count":22,"alternative-id":["10.1145\/3649329.3657314","10.1145\/3649329"],"URL":"https:\/\/doi.org\/10.1145\/3649329.3657314","relation":{},"subject":[],"published":{"date-parts":[[2024,6,23]]},"assertion":[{"value":"2024-11-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}