{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:11:26Z","timestamp":1783437086104,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","funder":[{"name":"European Union?s Horizon 2020 research and innovation programme","award":["957407"],"award-info":[{"award-number":["957407"]}]},{"name":"German Research Foundation","award":["ref. 414984028 and ref. 556566056"],"award-info":[{"award-number":["ref. 414984028 and ref. 556566056"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,22]]},"DOI":"10.1145\/3735654.3735940","type":"proceedings-article","created":{"date-parts":[[2025,6,22]],"date-time":"2025-06-22T21:36:31Z","timestamp":1750628191000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["PQ Bench: Benchmarking Pruning and Quantization Techniques"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-0228-9549","authenticated-orcid":false,"given":"Jonas","family":"Schulze","sequence":"first","affiliation":[{"name":"Hasso Plattner Institute, University of Potsdam, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2569-2549","authenticated-orcid":false,"given":"Nils","family":"Strassenburg","sequence":"additional","affiliation":[{"name":"Hasso Plattner Institute, University of Potsdam, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3335-8045","authenticated-orcid":false,"given":"Tilmann","family":"Rabl","sequence":"additional","affiliation":[{"name":"Hasso Plattner Institute, University of Potsdam, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,22]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"John.","author":"Blalock","year":"2020","unstructured":"Blalock, Davis and Gonzalez Ortiz, Jose Javier and Frankle, Jonathan and Guttag, John. 2020. Shrinkbench. https:\/\/github.com\/JJGO\/shrinkbench Accessed: -2024-11-27."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3318464.3389711"},{"key":"e_1_3_2_1_3_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. CoRR abs\/2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2020. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. CoRR abs\/2010.11929 (2020). arXiv:2010.11929 https:\/\/arxiv.org\/abs\/2010.11929"},{"key":"e_1_3_2_1_4_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=YicbFdNTTy","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=YicbFdNTTy"},{"key":"e_1_3_2_1_5_1","unstructured":"fast.ai. 2019. ImageNette. https:\/\/github.com\/fastai\/imagenette. Accessed: 2025-02-05."},{"key":"e_1_3_2_1_6_1","unstructured":"fast.ai. 2019. ImageWoof. https:\/\/github.com\/fastai\/imagenette?tab=readme-ov-file#imagewoof. Accessed: 2025-02-05."},{"key":"e_1_3_2_1_7_1","volume-title":"Muhammad Hamza Sajjad, and Laura Brinkholm Justesen","author":"Fladmark Eirik","year":"2023","unstructured":"Eirik Fladmark, Muhammad Hamza Sajjad, and Laura Brinkholm Justesen. 2023. Exploring the Performance of Pruning Methods in Neural Networks: An Empirical Study of the Lottery Ticket Hypothesis. arXiv:2303.15479 [cs.LG] https:\/\/arxiv.org\/abs\/2303.15479"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSI.2009.2019392"},{"key":"e_1_3_2_1_9_1","unstructured":"Seyyed Hossein Hasanpour Mohammad Rouhani Mohsen Fayyaz and Mohammad Sabokrou. 2023. Lets keep it simple Using simple architectures to outperform deeper and more complex architectures. arXiv:1608.06037 [cs.CV] https:\/\/arxiv.org\/abs\/1608.06037"},{"key":"e_1_3_2_1_10_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2015. Deep Residual Learning for Image Recognition. arXiv:1512.03385 [cs.CV] https:\/\/arxiv.org\/abs\/1512.03385"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.14778\/3551793.3551799"},{"key":"e_1_3_2_1_12_1","volume-title":"Weinberger","author":"Huang Gao","year":"2018","unstructured":"Gao Huang, Zhuang Liu, Laurens van der Maaten, and Kilian Q. Weinberger. 2018. Densely Connected Convolutional Networks. arXiv:1608.06993 [cs.CV] https:\/\/arxiv.org\/abs\/1608.06993"},{"key":"e_1_3_2_1_13_1","first-page":"16305","article-title":"Rethinking the pruning criteria for convolutional neural network","volume":"34","author":"Huang Zhongzhan","year":"2021","unstructured":"Zhongzhan Huang, Wenqi Shao, Xinjiang Wang, Liang Lin, and Ping Luo. 2021. Rethinking the pruning criteria for convolutional neural network. Advances in Neural Information Processing Systems 34 (2021), 16305--16318.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_14_1","unstructured":"Forrest N. Iandola Song Han Matthew W. Moskewicz Khalid Ashraf William J. Dally and Kurt Keutzer. 2016. SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and &lt;0.5MB model size. arXiv:1602.07360 [cs.CV] https:\/\/arxiv.org\/abs\/1602.07360"},{"key":"e_1_3_2_1_15_1","unstructured":"Java Point. 2023. Pruning in Machine Learning. https:\/\/www.javatpoint.com\/pruning-in-machine-learning. Accessed: 2024-10-18."},{"key":"e_1_3_2_1_16_1","unstructured":"Jeff Pool Abhishek Sawarkar Jay Rodge. 2021. Accelerating Inference with Sparsity Using the NVIDIA Ampere Architecture and NVIDIA TensorRT. https:\/\/developer.nvidia.com\/blog\/accelerating-inference-with-sparsity-using-ampere-and-tensorrt\/. Accessed: 2024-11-25."},{"key":"e_1_3_2_1_17_1","unstructured":"Joel Nicholls. 2018. Quantization in Deep Learning. https:\/\/medium.com\/@joel_34050\/quantization-in-deep-learning-478417eab72b. Accessed: 2024-10-18."},{"key":"e_1_3_2_1_18_1","unstructured":"Jonas Schulze. 2024. Master Thesis - Benchmarking of Pruning and Quantization Techniques. https:\/\/github.com\/sjoze\/master-thesis\/blob\/main\/master_thesis_final_Jonas_Schulze.pdf. Accessed: 2025-01-31."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3514221.3517897"},{"key":"e_1_3_2_1_20_1","unstructured":"Katsuya Hyodo. 2024. TensorRT Excecution Provide. https:\/\/zenn.dev\/pinto0309\/scraps\/42587e1074fc53. Accessed: 2024-11-27."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3183713.3196909"},{"key":"e_1_3_2_1_22_1","unstructured":"Alex Krizhevsky. 2014. One weird trick for parallelizing convolutional neural networks. arXiv:1404.5997 [cs.NE] https:\/\/arxiv.org\/abs\/1404.5997"},{"key":"e_1_3_2_1_23_1","unstructured":"Hao Li Asim Kadav Igor Durdanovic Hanan Samet and Hans Peter Graf. 2017. Pruning Filters for Efficient ConvNets. arXiv:1608.08710 [cs.CV] https:\/\/arxiv.org\/abs\/1608.08710"},{"key":"e_1_3_2_1_24_1","unstructured":"Ningning Ma Xiangyu Zhang Hai-Tao Zheng and Jian Sun. 2018. ShuffieNet V2: Practical Guidelines for Efficient CNN Architecture Design. arXiv:1807.11164 [cs.CV] https:\/\/arxiv.org\/abs\/1807.11164"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Marco Ancona. 2020. TorchPruner. https:\/\/github.com\/marcoancona\/TorchPruner\/tree\/master. Accessed: 2024-11-12.","DOI":"10.3917\/aep.2020.0011"},{"key":"e_1_3_2_1_26_1","volume-title":"Neo: A Learned Query Optimizer. Proceedings of the VLDB Endowment 12","author":"Marcus Ryan","unstructured":"Ryan Marcus, Parimarjan Negi, Hongzi Mao, Chi Zhang, Mohammad Alizadeh, Tim Kraska, Olga Papaemmanouil, and Nesime Tatbul23. [n. d.]. Neo: A Learned Query Optimizer. Proceedings of the VLDB Endowment 12, 11 ([n. d.])."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3211954.3211957"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 9th Biennial Conference on Innovative Data Systems Research (CIDR '19)","author":"Marcus Ryan","year":"2019","unstructured":"Ryan Marcus and Olga Papaemmanouil. 2019. Towards a Hands-Free Query Optimizer through Deep Learning. In Proceedings of the 9th Biennial Conference on Innovative Data Systems Research (CIDR '19)."},{"key":"e_1_3_2_1_29_1","unstructured":"Mark Kurtz. 2020. What is Pruning in Machine Learning? https:\/\/opendatascience.com\/what-is-pruning-in-machine-learning\/. Accessed: 2024-10-18."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Deepak Mittal Shweta Bhardwaj Mitesh M. Khapra and Balaraman Ravindran. 2018. Studying the Plasticity in Deep Convolutional Neural Networks using Random Pruning. arXiv:1812.10240 [cs.LG] https:\/\/arxiv.org\/abs\/1812.10240","DOI":"10.1109\/WACV.2018.00098"},{"key":"e_1_3_2_1_31_1","volume-title":"Pruning Convolutional Neural Networks for Resource Efficient Inference. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SJGCiw5gl","author":"Molchanov Pavlo","year":"2017","unstructured":"Pavlo Molchanov, Stephen Tyree, Tero Karras, Timo Aila, and Jan Kautz. 2017. Pruning Convolutional Neural Networks for Resource Efficient Inference. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SJGCiw5gl"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/EMC2-NIPS53020.2019.00020"},{"key":"e_1_3_2_1_33_1","unstructured":"NVIDIA. 2016. TensorRT. https:\/\/developer.nvidia.com\/tensorrt. Accessed: 2024-11-17."},{"key":"e_1_3_2_1_34_1","unstructured":"NVIDIA. 2017. NVIDIA TESLA V100 GPU ARCHITECTURE. https:\/\/images.nvidia.com\/content\/volta-architecture\/pdf\/volta-architecture-whitepaper.pdf. Accessed: 2025-01-28."},{"key":"e_1_3_2_1_35_1","unstructured":"NVIDIA. 2020. NVIDIA TensorRT Documentation. https:\/\/docs.nvidia.com\/deeplearning\/tensorrt\/developer-guide\/index.html. Accessed: 2024-11-25."},{"key":"e_1_3_2_1_36_1","unstructured":"NVIDIA. 2020. NVIDIA TensorRT's capabilities. https:\/\/docs.nvidia.com\/deeplearning\/tensorrt\/10.8.0\/architecture\/capabilities.html. Accessed: 2025-02-02."},{"key":"e_1_3_2_1_37_1","unstructured":"ONNX. 2021. ONNX Execution Provider. https:\/\/onnxruntime.ai\/docs\/execution-providers\/TensorRT-ExecutionProvider.html. Accessed: 2024-11-24."},{"key":"e_1_3_2_1_38_1","unstructured":"PyTorch. 2020. L1 Unstructured PyTorch. https:\/\/pytorch.org\/docs\/stable\/generated\/torch.nn.utils.prune.l1_unstructured.html. Accessed: 2025-02-02."},{"key":"e_1_3_2_1_39_1","unstructured":"PyTorch. 2020. Random Unstructured PyTorch. https:\/\/pytorch.org\/docs\/stable\/generated\/torch.nn.utils.prune.random_unstructured.html. Accessed: 2025-02-02."},{"key":"e_1_3_2_1_40_1","unstructured":"PyTorch. 2022. Post Training Quantization (PTQ). https:\/\/pytorch.org\/TensorRT\/tutorials\/ptq.html. Accessed: 2025-02-02."},{"key":"e_1_3_2_1_41_1","unstructured":"PyTorch. 2023. Torch Prune. https:\/\/pytorch.org\/docs\/stable\/nn.html. Accessed: 2024-11-19."},{"key":"e_1_3_2_1_42_1","volume-title":"A comprehensive survey on model quantization for deep neural networks. arXiv preprint arXiv:2205.07877","author":"Rokh Babak","year":"2022","unstructured":"Babak Rokh, Ali Azarpeyvand, and Alireza Khanteymoori. 2022. A comprehensive survey on model quantization for deep neural networks. arXiv preprint arXiv:2205.07877 (2022)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Mark Sandler Andrew Howard Menglong Zhu Andrey Zhmoginov and Liang-Chieh Chen. 2019. MobileNetV2: Inverted Residuals and Linear Bottlenecks. arXiv:1801.04381 [cs.CV] https:\/\/arxiv.org\/abs\/1801.04381","DOI":"10.1109\/CVPR.2018.00474"},{"key":"e_1_3_2_1_44_1","unstructured":"Song Han. 2024. TinyML and Efficient Deep Learning Computing Lecture (MIT). https:\/\/hanlab.mit.edu\/courses\/2024-fall-65940. Accessed: 2025-01-31."},{"key":"e_1_3_2_1_45_1","volume-title":"Stanford University","author":"Lab Stanford Vision","year":"2007","unstructured":"Stanford Vision Lab, Stanford University. 2007. ImageNet. https:\/\/www.imagenet.org\/. Accessed: 2025-02-05."},{"key":"e_1_3_2_1_46_1","volume-title":"ARC 2018, Santorini, Greece, May 2-4, 2018, Proceedings 14","author":"Su Jiang","year":"2018","unstructured":"Jiang Su, Julian Faraone, Junyi Liu, Yiren Zhao, David B Thomas, Philip HW Leong, and Peter YK Cheung. 2018. Redundancy-reduced mobilenet acceleration on reconfigurable logic for imagenet classification. In Applied Reconfigurable Computing. Architectures, Tools, and Applications: 14th International Symposium, ARC 2018, Santorini, Greece, May 2-4, 2018, Proceedings 14. Springer, 16--28."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Wei Liu Yangqing Jia Pierre Sermanet Scott Reed Dragomir Anguelov Dumitru Erhan Vincent Vanhoucke and Andrew Rabinovich. 2014. Going Deeper with Convolutions. arXiv:1409.4842 [cs.CV] https:\/\/arxiv.org\/abs\/1409.4842","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_48_1","unstructured":"Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jonathon Shlens and Zbigniew Wojna. 2015. Rethinking the Inception Architecture for Computer Vision. arXiv:1512.00567 [cs.CV] https:\/\/arxiv.org\/abs\/1512.00567"},{"key":"e_1_3_2_1_49_1","unstructured":"The Linux Foundation. 2017. ONNX. https:\/\/onnx.ai\/. Accessed: 2025-01-31."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3318464.3380589"},{"key":"e_1_3_2_1_51_1","unstructured":"Hao Wu Patrick Judd Xiaojie Zhang Mikhail Isaev and Paulius Micikevicius. 2020. Integer Quantization for Deep Learning Inference: Principles and Empirical Evaluation. arXiv:2004.09602 [cs.LG] https:\/\/arxiv.org\/abs\/2004.09602"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.14778\/3547305.3547310"},{"key":"e_1_3_2_1_53_1","volume-title":"Understanding Straight-Through Estimator in Training Activation Quantized Neural Nets. CoRR abs\/1903.05662","author":"Yin Penghang","year":"2019","unstructured":"Penghang Yin, Jiancheng Lyu, Shuai Zhang, Stanley J. Osher, Yingyong Qi, and Jack Xin. 2019. Understanding Straight-Through Estimator in Training Activation Quantized Neural Nets. CoRR abs\/1903.05662 (2019). arXiv:1903.05662 http:\/\/arxiv.org\/abs\/1903.05662"},{"key":"e_1_3_2_1_54_1","volume-title":"TensorRT Implementations of Model Quantization on Edge SoC. In 2023 IEEE 16th International Symposium on Embedded Multicore\/Many-core Systems-on-Chip (MCSoC). IEEE, 486--493","author":"Zhou Yuxiao","year":"2023","unstructured":"Yuxiao Zhou, Zhishan Guo, Zheng Dong, and Kecheng Yang. 2023. TensorRT Implementations of Model Quantization on Edge SoC. In 2023 IEEE 16th International Symposium on Embedded Multicore\/Many-core Systems-on-Chip (MCSoC). IEEE, 486--493."}],"event":{"name":"SIGMOD\/PODS '25: International Conference on Management of Data","location":"Berlin Germany","acronym":"SIGMOD\/PODS '25","sponsor":["SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the Workshop on Data Management for End-to-End Machine Learning"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3735654.3735940","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T18:28:31Z","timestamp":1755973711000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3735654.3735940"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,22]]},"references-count":54,"alternative-id":["10.1145\/3735654.3735940","10.1145\/3735654"],"URL":"https:\/\/doi.org\/10.1145\/3735654.3735940","relation":{},"subject":[],"published":{"date-parts":[[2025,6,22]]},"assertion":[{"value":"2025-06-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}