{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T16:12:05Z","timestamp":1783613525465,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,12]],"date-time":"2026-04-12T00:00:00Z","timestamp":1775952000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100003816","name":"Huawei Technologies","doi-asserted-by":"publisher","award":["Edinburgh Joint Lab"],"award-info":[{"award-number":["Edinburgh Joint Lab"]}],"id":[{"id":"10.13039\/501100003816","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,12]]},"DOI":"10.1145\/3786582.3786800","type":"proceedings-article","created":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T11:52:47Z","timestamp":1783511567000},"page":"6-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["A Selective Quantization Tuner for ONNX Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1878-7679","authenticated-orcid":false,"given":"Nikolaos","family":"Louloudakis","sequence":"first","affiliation":[{"name":"University of Edinburgh, Edinburgh, Midlothian, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3765-3075","authenticated-orcid":false,"given":"Ajitha","family":"Rajan","sequence":"additional","affiliation":[{"name":"University of Edinburgh, Edinburgh, Midlothian, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,8]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"2019. Language Models are Unsupervised Multitask Learners author=Alec Radford and Jeff Wu and Rewon Child and David Luan and Dario Amodei and Ilya Sutskever. https:\/\/api.semanticscholar.org\/CorpusID:160025533"},{"key":"e_1_3_3_2_3_2","unstructured":"2019. ONNX Quantizer. https:\/\/onnxruntime.ai\/docs\/performance\/model-optimizations\/quantization.html. [Accessed 23-Apr-2025]."},{"key":"e_1_3_3_2_4_2","unstructured":"2019. ONNX Runtime. https:\/\/onnxruntime.ai. [Accessed 23-Apr-2025]."},{"key":"e_1_3_3_2_5_2","unstructured":"2020. Datasets. https:\/\/huggingface.co\/docs\/datasets. [Accessed 23-Apr-2025]."},{"key":"e_1_3_3_2_6_2","unstructured":"2023. ONNX Model Hub. https:\/\/github.com\/onnx\/onnx\/blob\/main\/docs\/Hub.md. [Accessed 23-Apr-2025]."},{"key":"e_1_3_3_2_7_2","unstructured":"2023. Open Neural Network Exchange. https:\/\/onnx.ai. [Accessed 23-Apr-2025]."},{"key":"e_1_3_3_2_8_2","unstructured":"2025. pymoo: Multi-objective Optimization in Python. https:\/\/pymoo.org\/. [Accessed 8-Jul-2025]."},{"key":"e_1_3_3_2_9_2","series-title":"(OSDI\u201916)","first-page":"265","volume-title":"Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, Manjunath Kudlur, Josh Levenberg, Rajat Monga, Sherry Moore, et\u00a0al. 2016. TensorFlow: a system for large-scale machine learning. In Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation (Savannah, GA, USA) (OSDI\u201916). USENIX Association, USA, 265\u2013283."},{"key":"e_1_3_3_2_10_2","volume-title":"Post training 4-bit quantization of convolutional networks for rapid-deployment","author":"Banner Ron","year":"2019","unstructured":"Ron Banner, Yury Nahshan, and Daniel Soudry. 2019. Post training 4-bit quantization of convolutional networks for rapid-deployment. Curran Associates Inc., Red Hook, NY, USA."},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00356"},{"key":"e_1_3_3_2_12_2","first-page":"578","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, Carlos Guestrin, and Arvind Krishnamurthy. 2018. TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). 578\u2013594."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","unstructured":"Peter Christen David\u00a0J. Hand and Nishadi Kirielle. 2023. A Review of the F-Measure: Its History Properties Criticism and Alternatives. ACM Comput. Surv. 56 3 Article 73 (Oct. 2023) 24\u00a0pages. 10.1145\/3606367","DOI":"10.1145\/3606367"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Indraneel Das and J.\u00a0E. Dennis. 1998. Normal-Boundary Intersection: A New Method for Generating the Pareto Surface in Nonlinear Multicriteria Optimization Problems. SIAM Journal on Optimization 8 3 (1998) 631\u2013657. arXiv:https:\/\/doi.org\/10.1137\/S105262349630751010.1137\/S1052623496307510","DOI":"10.1137\/S1052623496307510"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-45356-3_83"},{"key":"e_1_3_3_2_16_2","unstructured":"Huabin Diao Gongyan Li Shaoyun Xu and Yuexing Hao. 2022. Attention Round for Post-Training Quantization. arxiv:https:\/\/arXiv.org\/abs\/2207.03088\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2207.03088"},{"key":"e_1_3_3_2_17_2","unstructured":"Zhen Dong Zhewei Yao Yaohui Cai Daiyaan Arfeen Amir Gholami Michael\u00a0W. Mahoney and Kurt Keutzer. 2019. HAWQ-V2: Hessian Aware trace-Weighted Quantization of Neural Networks. arxiv:https:\/\/arXiv.org\/abs\/1911.03852\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1911.03852"},{"key":"e_1_3_3_2_18_2","unstructured":"Zhen Dong Zhewei Yao Amir Gholami Michael Mahoney and Kurt Keutzer. 2019. HAWQ: Hessian AWare Quantization of Neural Networks with Mixed-Precision. arxiv:https:\/\/arXiv.org\/abs\/1905.03696\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1905.03696"},{"key":"e_1_3_3_2_19_2","unstructured":"Steven\u00a0K. Esser Jeffrey\u00a0L. McKinstry Deepika Bablani Rathinakumar Appuswamy and Dharmendra\u00a0S. Modha. 2020. Learned Step Size Quantization. arxiv:https:\/\/arXiv.org\/abs\/1902.08153\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/1902.08153"},{"key":"e_1_3_3_2_20_2","unstructured":"Amir Gholami Sehoon Kim Zhen Dong Zhewei Yao Michael\u00a0W. Mahoney and Kurt Keutzer. 2021. A Survey of Quantization Methods for Efficient Neural Network Inference. ArXiv abs\/2103.13630 (2021). https:\/\/api.semanticscholar.org\/CorpusID:232352683"},{"key":"e_1_3_3_2_21_2","volume-title":"4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings","author":"Han Song","year":"2016","unstructured":"Song Han, Huizi Mao, and William\u00a0J. Dally. 2016. Deep Compression: Compressing Deep Neural Network with Pruning, Trained Quantization and Huffman Coding. In 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, 2016, Conference Track Proceedings, Yoshua Bengio and Yann LeCun (Eds.). http:\/\/arxiv.org\/abs\/1510.00149"},{"key":"e_1_3_3_2_22_2","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2015. Deep Residual Learning for Image Recognition. CoRR abs\/1512.03385 (2015). arXiv:https:\/\/arXiv.org\/abs\/1512.03385http:\/\/arxiv.org\/abs\/1512.03385"},{"key":"e_1_3_3_2_23_2","unstructured":"Xijie Huang Zhiqiang Shen Shichao Li Zechun Liu Xianghong Hu Jeffry Wicaksana Eric Xing and Kwang-Ting Cheng. 2022. SDQ: Stochastic Differentiable Quantization with Mixed Precision. arxiv:https:\/\/arXiv.org\/abs\/2206.04459\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2206.04459"},{"key":"e_1_3_3_2_24_2","unstructured":"Itay Hubara Yury Nahshan Yair Hanani Ron Banner and Daniel Soudry. 2020. Improving Post Training Neural Quantization: Layer-wise Calibration and Integer Programming. arxiv:https:\/\/arXiv.org\/abs\/2006.10518\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2006.10518"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00286"},{"key":"e_1_3_3_2_26_2","unstructured":"Raghuraman Krishnamoorthi. 2018. Quantizing deep convolutional networks for efficient inference: A whitepaper. arxiv:https:\/\/arXiv.org\/abs\/1806.08342\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/1806.08342"},{"key":"e_1_3_3_2_27_2","unstructured":"Kai Liu Qian Zheng Kaiwen Tao Zhiteng Li Haotong Qin Wenbo Li Yong Guo Xianglong Liu Linghe Kong Guihai Chen Yulun Zhang and Xiaokang Yang. 2025. Low-bit Model Quantization for Deep Neural Networks: A Survey. arxiv:https:\/\/arXiv.org\/abs\/2505.05530\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2505.05530"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","unstructured":"Nikolaos Louloudakis. 2025. SeQTO. 10.5281\/zenodo.18155147","DOI":"10.5281\/zenodo.18155147"},{"key":"e_1_3_3_2_30_2","unstructured":"Yuexiao Ma Taisong Jin Xiawu Zheng Yan Wang Huixia Li Yongjian Wu Guannan Jiang Wei Zhang and Rongrong Ji. 2022. OMPQ: Orthogonal Mixed Precision Quantization. arxiv:https:\/\/arXiv.org\/abs\/2109.07865\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2109.07865"},{"key":"e_1_3_3_2_31_2","unstructured":"Markus Nagel Rana\u00a0Ali Amjad Mart van Baalen Christos Louizos and Tijmen Blankevoort. 2020. Up or Down? Adaptive Rounding for Post-Training Quantization. arxiv:https:\/\/arXiv.org\/abs\/2004.10568\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2004.10568"},{"key":"e_1_3_3_2_32_2","unstructured":"Markus Nagel Marios Fournarakis Yelysei Bondarenko and Tijmen Blankevoort. 2022. Overcoming Oscillations in Quantization-Aware Training. arxiv:https:\/\/arXiv.org\/abs\/2203.11086\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2203.11086"},{"key":"e_1_3_3_2_33_2","unstructured":"Nilesh\u00a0Prasad Pandey Markus Nagel Mart van Baalen Yin Huang Chirag Patel and Tijmen Blankevoort. 2023. A Practical Mixed Precision Algorithm for Post-Training Quantization. arxiv:https:\/\/arXiv.org\/abs\/2302.05397\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2302.05397"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_3_2_35_2","unstructured":"Dominika Przewlocka-Rus Syed\u00a0Shakib Sarwar H.\u00a0Ekin Sumbul Yuecheng Li and Barbara\u00a0De Salvo. 2022. Power-of-Two Quantization for Low Bitwidth and Hardware Compliant Neural Networks. arxiv:https:\/\/arXiv.org\/abs\/2203.05025\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2203.05025"},{"key":"e_1_3_3_2_36_2","unstructured":"Colin Raffel Noam Shazeer Adam Roberts Katherine Lee Sharan Narang Michael Matena Yanqi Zhou Wei Li and Peter\u00a0J. Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21 1 Article 140 (Jan. 2020) 67\u00a0pages."},{"key":"e_1_3_3_2_37_2","unstructured":"Joseph Redmon and Ali Farhadi. 2018. YOLOv3: An Incremental Improvement. ArXiv abs\/1804.02767 (2018). https:\/\/api.semanticscholar.org\/CorpusID:4714433"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","unstructured":"Babak Rokh Ali Azarpeyvand and Alireza Khanteymoori. 2023. A Comprehensive Survey on Model Quantization for Deep Neural Networks in Image Classification. ACM Trans. Intell. Syst. Technol. 14 6 Article 97 (Nov. 2023) 50\u00a0pages. 10.1145\/3623402","DOI":"10.1145\/3623402"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","unstructured":"Olga Russakovsky Jia Deng Hao Su Jonathan Krause Sanjeev Satheesh Sean Ma Zhiheng Huang Andrej Karpathy Aditya Khosla Michael Bernstein Alexander\u00a0C. Berg and Li Fei-Fei. 2015. ImageNet Large Scale Visual Recognition Challenge. International Journal of Computer Vision (IJCV) 115 3 (2015) 211\u2013252. 10.1007\/s11263-015-0816-y","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_3_2_40_2","unstructured":"Mark Sandler Andrew\u00a0G. Howard Menglong Zhu Andrey Zhmoginov and Liang-Chieh Chen. 2018. Inverted Residuals and Linear Bottlenecks: Mobile Networks for Classification Detection and Segmentation. CoRR abs\/1801.04381 (2018). arXiv:https:\/\/arXiv.org\/abs\/1801.04381http:\/\/arxiv.org\/abs\/1801.04381"},{"key":"e_1_3_3_2_41_2","unstructured":"Clemens\u00a0JS Schaefer Elfie Guo Caitlin Stanton Xiaofan Zhang Tom Jablin Navid Lambert-Shirzad Jian Li Chiachen Chou Siddharth Joshi and Yu\u00a0Emma Wang. 2023. Mixed Precision Post Training Quantization of Neural Networks with Sensitivity Guided Search. arxiv:https:\/\/arXiv.org\/abs\/2302.01382\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2302.01382"},{"key":"e_1_3_3_2_42_2","unstructured":"Sangeetha Siddegowda Marios Fournarakis Markus Nagel Tijmen Blankevoort Chirag Patel and Abhijit Khobare. 2022. Neural Network Quantization with AI Model Efficiency Toolkit (AIMET). arxiv:https:\/\/arXiv.org\/abs\/2201.08442\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2201.08442"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-47115-5_18"},{"key":"e_1_3_3_2_44_2","unstructured":"Mingxing Tan et\u00a0al. 2020. EfficientNet: Rethinking Model Scaling for Convolutional Neural Networks. arxiv:https:\/\/arXiv.org\/abs\/1905.11946\u00a0[cs.LG]"},{"key":"e_1_3_3_2_45_2","unstructured":"Olivia Weng. 2023. Neural Network Quantization for Efficient Inference: A Survey. arxiv:https:\/\/arXiv.org\/abs\/2112.06126\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2112.06126"},{"key":"e_1_3_3_2_46_2","unstructured":"Bichen Wu Yanghan Wang Peizhao Zhang Yuandong Tian Peter Vajda and Kurt Keutzer. 2018. Mixed Precision Quantization of ConvNets via Differentiable Neural Architecture Search. arxiv:https:\/\/arXiv.org\/abs\/1812.00090\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1812.00090"},{"key":"e_1_3_3_2_47_2","unstructured":"Linjie Yang and Qing Jin. 2020. FracBits: Mixed Precision Quantization via Fractional Bit-Widths. arxiv:https:\/\/arXiv.org\/abs\/2007.02017\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2007.02017"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00716"}],"event":{"name":"ICSE-NIER '26: 2026 IEEE\/ACM 48th International Conference on Software Engineering","location":"Rio de Janeiro Brazil","acronym":"ICSE-NIER '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the IEEE\/ACM 48th International Conference on Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3786582.3786800","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T11:55:16Z","timestamp":1783511716000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3786582.3786800"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":47,"alternative-id":["10.1145\/3786582.3786800","10.1145\/3786582"],"URL":"https:\/\/doi.org\/10.1145\/3786582.3786800","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-07-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}