{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:46:57Z","timestamp":1783036017537,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":115,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,2,17]],"date-time":"2024-02-17T00:00:00Z","timestamp":1708128000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,2,17]]},"DOI":"10.1145\/3640537.3641566","type":"proceedings-article","created":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T21:43:05Z","timestamp":1708465385000},"page":"212-226","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["YFlows: Systematic Dataflow Exploration and Code Generation for Efficient Neural Network Inference using SIMD Architectures on CPUs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8768-0659","authenticated-orcid":false,"given":"Cyrus","family":"Zhou","sequence":"first","affiliation":[{"name":"University of Chicago, Chicago, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3392-9296","authenticated-orcid":false,"given":"Zack","family":"Hassman","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8436-8796","authenticated-orcid":false,"given":"Dhirpal","family":"Shah","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1048-9532","authenticated-orcid":false,"given":"Vaughn","family":"Richard","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0124-0463","authenticated-orcid":false,"given":"Yanjing","family":"Li","sequence":"additional","affiliation":[{"name":"University of Chicago, Chicago, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,2,20]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n. d.]. GNU Compiler Collection. https:\/\/gcc.gnu.org\/ Accessed: 2023-08-28"},{"key":"e_1_3_2_1_2_1","unstructured":"Mart\u00edn Abadi Ashish Agarwal Paul Barham Eugene Brevdo Zhifeng Chen Craig Citro Greg S. Corrado Andy Davis Jeffrey Dean Matthieu Devin Sanjay Ghemawat Ian Goodfellow Andrew Harp Geoffrey Irving Michael Isard Yangqing Jia Rafal Jozefowicz Lukasz Kaiser Manjunath Kudlur Josh Levenberg Dandelion Man\u00e9 Rajat Monga Sherry Moore Derek Murray Chris Olah Mike Schuster Jonathon Shlens Benoit Steiner Ilya Sutskever Kunal Talwar Paul Tucker Vincent Vanhoucke Vijay Vasudevan Fernanda Vi\u00e9gas Oriol Vinyals Pete Warden Martin Wattenberg Martin Wicke Yuan Yu and Xiaoqiang Zheng. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/ Software available from tensorflow.org"},{"key":"e_1_3_2_1_3_1","first-page":"44","article-title":"Ordering chaos: Memory-aware scheduling of irregularly wired neural networks for edge devices","volume":"2","author":"Ahn Byung Hoon","year":"2020","unstructured":"Byung Hoon Ahn, Jinwon Lee, Jamie Menjay Lin, Hsin-Pai Cheng, Jilei Hou, and Hadi Esmaeilzadeh. 2020. Ordering chaos: Memory-aware scheduling of irregularly wired neural networks for edge devices. Proceedings of Machine Learning and Systems, 2 (2020), 44\u201357.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3524069","article-title":"Winograd convolution for deep neural networks: Efficient point selection","volume":"21","author":"Alam Syed Asad","year":"2022","unstructured":"Syed Asad Alam, Andrew Anderson, Barbara Barabasz, and David Gregg. 2022. Winograd convolution for deep neural networks: Efficient point selection. ACM Transactions on Embedded Computing Systems, 21, 6 (2022), 1\u201328.","journal-title":"ACM Transactions on Embedded Computing Systems"},{"key":"e_1_3_2_1_5_1","volume-title":"2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO). 1\u201312","author":"Alwani M.","year":"2016","unstructured":"M. Alwani, H. Chen, M. Ferdman, and P. Milder. 2016. Fused-layer CNN accelerators. In 2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO). 1\u201312. https:\/\/doi.org\/10.1109\/MICRO.2016.7783725 10.1109\/MICRO.2016.7783725"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","first-page":"83","DOI":"10.1016\/j.jpdc.2019.09.012","article-title":"SIMD programming using Intel vector extensions","volume":"135","author":"Amiri Hossein","year":"2020","unstructured":"Hossein Amiri and Asadollah Shahbahrami. 2020. SIMD programming using Intel vector extensions. J. Parallel and Distrib. Comput., 135 (2020), 83\u2013100.","journal-title":"J. Parallel and Distrib. Comput."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Yelysei Bondarenko Markus Nagel and Tijmen Blankevoort. 2021. Understanding and overcoming the challenges of efficient transformer quantization. arXiv preprint arXiv:2109.12948.","DOI":"10.18653\/v1\/2021.emnlp-main.627"},{"key":"e_1_3_2_1_8_1","first-page":"1","article-title":"Optimus: An operator fusion framework for deep neural networks","volume":"22","author":"Cai Xuyi","year":"2022","unstructured":"Xuyi Cai, Ying Wang, and Lei Zhang. 2022. Optimus: An operator fusion framework for deep neural networks. ACM Transactions on Embedded Computing Systems, 22, 1 (2022), 1\u201326.","journal-title":"ACM Transactions on Embedded Computing Systems"},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of HICSS-29: 29th Hawaii International Conference on System Sciences. 1, 183\u2013192","author":"Carr Steve","year":"1996","unstructured":"Steve Carr, Chen Ding, and Philip Sweany. 1996. Improving software pipelining with unroll-and-jam. In Proceedings of HICSS-29: 29th Hawaii International Conference on System Sciences. 1, 183\u2013192."},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of 30th Annual International Symposium on Microarchitecture. 349\u2013357","author":"Carr Steve","year":"1997","unstructured":"Steve Carr and Yiping Guan. 1997. Unroll-and-jam using uniformly generated sets. In Proceedings of 30th Annual International Symposium on Microarchitecture. 349\u2013357."},{"key":"e_1_3_2_1_11_1","volume-title":"High Performance Convolutional Neural Networks for Document Processing. In Tenth International Workshop on Frontiers in Handwriting Recognition.","author":"Chellapilla K.","unstructured":"K. Chellapilla, S. Puri, and P. Simard. 2006. High Performance Convolutional Neural Networks for Document Processing. In Tenth International Workshop on Frontiers in Handwriting Recognition."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 19th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS \u201914)","author":"Chen Tianshi","year":"2014","unstructured":"Tianshi Chen, Zidong Du, Ninghui Sun, Jia Wang, Chengyong Wu, Yunji Chen, and Olivier Temam. 2014. DianNao: A Small-Footprint High-Throughput Accelerator for Ubiquitous Machine-Learning. In Proceedings of the 19th International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS \u201914). Association for Computing Machinery, New York, NY, USA. 269\u2013284. isbn:9781450323055 https:\/\/doi.org\/10.1145\/2541940.2541967 10.1145\/2541940.2541967"},{"key":"e_1_3_2_1_13_1","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen T.","unstructured":"T. Chen, T. Moreau, Z. Jiang, L. Zheng, E. Yan, H. Shen, ..., and Y. Chen. 2018. TVM: An automated end-to-end optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). 578\u2013594."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 12052\u201312062","author":"Chen Xianing","year":"2022","unstructured":"Xianing Chen, Qiong Cao, Yujie Zhong, Jing Zhang, Shenghua Gao, and Dacheng Tao. 2022. Dearkd: data-efficient early knowledge distillation for vision transformers. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 12052\u201312062."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2017.54"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 12507\u201312516","author":"Chikin Vladimir","year":"2022","unstructured":"Vladimir Chikin and Vladimir Kryzhanovskiy. 2022. Channel balancing for accurate quantization of winograd convolutions. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 12507\u201312516."},{"key":"e_1_3_2_1_17_1","volume-title":"International Conference on Learning Representations.","author":"Choi Jiahao","year":"2018","unstructured":"Jiahao Choi, Mostafa El-Khamy, and Jungwon Lee. 2018. Towards the limit of network quantization. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition. 1251\u20131258","author":"Chollet Francois","year":"2017","unstructured":"Francois Chollet. 2017. Xception: Deep Learning with Depthwise Separable Convolutions. In Proceedings of the IEEE conference on computer vision and pattern recognition. 1251\u20131258."},{"key":"e_1_3_2_1_19_1","volume-title":"Yongkweon Jeon, Baeseong Park, Sangha Kim, and Dongsoo Lee.","author":"Chung Insoo","year":"2020","unstructured":"Insoo Chung, Byeongwook Kim, Yoonjung Choi, Se Jung Kwon, Yongkweon Jeon, Baeseong Park, Sangha Kim, and Dongsoo Lee. 2020. Extremely low bit transformer quantization for on-device neural machine translation. arXiv preprint arXiv:2009.07453."},{"key":"e_1_3_2_1_20_1","volume-title":"Larq: An Open-Source Library for Training Binarized Neural Networks. https:\/\/github.com\/larq\/larq GitHub repository","author":"Contributors Larq","year":"2023","unstructured":"Larq Contributors. 2023. Larq: An Open-Source Library for Training Binarized Neural Networks. https:\/\/github.com\/larq\/larq GitHub repository"},{"key":"e_1_3_2_1_21_1","unstructured":"Intel Corporation. 2023. Intel Intrinsics Guide. https:\/\/software.intel.com\/sites\/landingpage\/IntrinsicsGuide\/ Accessed: 2023-05-19"},{"key":"e_1_3_2_1_22_1","volume-title":"Binaryconnect: Training deep neural networks with binary weights during propagations. In Advances in Neural Information Processing Systems. 3123\u20133131.","author":"Courbariaux Matthieu","year":"2015","unstructured":"Matthieu Courbariaux, Yoshua Bengio, and Jean-Pierre David. 2015. Binaryconnect: Training deep neural networks with binary weights during propagations. In Advances in Neural Information Processing Systems. 3123\u20133131."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the 18th ACM\/IEEE International Symposium on Code Generation and Optimization (CGO","author":"Cowan Meghan","year":"2020","unstructured":"Meghan Cowan, Thierry Moreau, Tianqi Chen, James Bornholt, and Luis Ceze. 2020. Automatic Generation of High-Performance Quantized Machine Learning Kernels. In Proceedings of the 18th ACM\/IEEE International Symposium on Code Generation and Optimization (CGO 2020). Association for Computing Machinery, New York, NY, USA. 305\u2013316. isbn:9781450370479 https:\/\/doi.org\/10.1145\/3368826.3377912 10.1145\/3368826.3377912"},{"key":"e_1_3_2_1_24_1","unstructured":"Dave Dice and Alex Kogan. 2021. Optimizing Inference Performance of Transformers on CPUs. arxiv:2102.06621."},{"key":"e_1_3_2_1_25_1","unstructured":"Javier Fernandez-Marques. 2020. Even Faster Convolutions: Winograd Convolutions meet Integer Quantization and Architecture Search. https:\/\/community.arm.com\/arm-research\/b\/articles\/posts\/even-faster-convolutions-winograd-convolutions-meet-integer-quantization-and-architecture-search Accessed: [Your Access Date Here]"},{"key":"e_1_3_2_1_26_1","first-page":"14","article-title":"Searching for winograd-aware quantized networks","volume":"2","author":"Fernandez-Marques Javier","year":"2020","unstructured":"Javier Fernandez-Marques, Paul Whatmough, Andrew Mundy, and Matthew Mattina. 2020. Searching for winograd-aware quantized networks. Proceedings of Machine Learning and Systems, 2 (2020), 14\u201329.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_27_1","unstructured":"Apache Software Foundation. 2023. tvm.autotvm \u2014 tvm 0.14.dev0 documentation. https:\/\/tvm.apache.org\/docs\/reference\/api\/python\/autotvm.html Accessed: 2023-08-31"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 22nd international conference on Parallel architectures and compilation techniques. 341\u2013351","author":"Govindaraju Venkatraman","year":"2013","unstructured":"Venkatraman Govindaraju, Tony Nowatzki, and Karthikeyan Sankaralingam. 2013. Breaking SIMD shackles with an exposed flexible microarchitecture and the access execute PDG. In Proceedings of the 22nd international conference on Parallel architectures and compilation techniques. 341\u2013351."},{"key":"e_1_3_2_1_29_1","volume-title":"2019 IEEE International Symposium on Workload Characterization (IISWC). 35\u201348","author":"Hadidi Ramyad","year":"2019","unstructured":"Ramyad Hadidi, Jiashen Cao, Yilun Xie, Bahar Asgari, Tushar Krishna, and Hyesoon Kim. 2019. Characterizing the deployment of deep neural networks on commercial edge devices. In 2019 IEEE International Symposium on Workload Characterization (IISWC). 35\u201348."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 18th ACM\/IEEE International Symposium on Code Generation and Optimization. 242\u2013255","author":"Haj-Ali Ameer","year":"2020","unstructured":"Ameer Haj-Ali, Nesreen K Ahmed, Ted Willke, Yakun Sophia Shao, Krste Asanovic, and Ion Stoica. 2020. Neurovectorizer: End-to-end vectorization with deep reinforcement learning. In Proceedings of the 18th ACM\/IEEE International Symposium on Code Generation and Optimization. 242\u2013255."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 43rd International Symposium on Computer Architecture. 243\u2013254","author":"Han Song","year":"2016","unstructured":"Song Han, Xingyu Liu, Huizi Mao, Jing Pu, Ardavan Pedram, Mark A Horowitz, and William J Dally. 2016. EIE: efficient inference engine on compressed deep neural network. In Proceedings of the 43rd International Symposium on Computer Architecture. 243\u2013254."},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 43rd International Symposium on Computer Architecture (ISCA \u201916)","author":"Han Song","year":"2016","unstructured":"Song Han, Xingyu Liu, Huizi Mao, Jing Pu, Ardavan Pedram, Mark A. Horowitz, and William J. Dally. 2016. EIE: efficient inference engine on compressed deep neural network. In Proceedings of the 43rd International Symposium on Computer Architecture (ISCA \u201916). 243\u2013254. https:\/\/doi.org\/10.1109\/ISCA.2016.30 10.1109\/ISCA.2016.30"},{"key":"e_1_3_2_1_33_1","volume-title":"International Conference on Learning Representations.","author":"Han Song","year":"2016","unstructured":"Song Han, Huizi Mao, and William J Dally. 2016. Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition. 770\u2013778","author":"He Kaiming","year":"2016","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition. 770\u2013778."},{"key":"e_1_3_2_1_35_1","volume-title":"Patterson","author":"Hennessy John L.","year":"2011","unstructured":"John L. Hennessy and David A. Patterson. 2011. Computer Architecture: A Quantitative Approach. Elsevier."},{"key":"e_1_3_2_1_36_1","unstructured":"Elad Hoffer Itay Hubara and Daniel Soudry. 2017. Train longer generalize better: closing the generalization gap in large batch training of neural networks. In Advances in Neural Information Processing Systems. 1731\u20131741."},{"key":"e_1_3_2_1_37_1","volume-title":"DFX: A Low-latency Multi-FPGA Appliance for Accelerating Transformer-based Text Generation. In 2022 55th IEEE\/ACM International Symposium on Microarchitecture (MICRO). 616\u2013630","author":"Hong Seongmin","year":"2022","unstructured":"Seongmin Hong, Seungjae Moon, Junsoo Kim, Sungjae Lee, Minsub Kim, Dongsoo Lee, and Joo-Young Kim. 2022. DFX: A Low-latency Multi-FPGA Appliance for Accelerating Transformer-based Text Generation. In 2022 55th IEEE\/ACM International Symposium on Microarchitecture (MICRO). 616\u2013630."},{"key":"e_1_3_2_1_38_1","volume-title":"Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861.","author":"Howard Andrew G","year":"2017","unstructured":"Andrew G Howard, Menglong Zhu, Bo Chen, Dmitry Kalenichenko, Weijun Wang, Tobias Weyand, Marco Andreetto, and Hartwig Adam. 2017. Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861."},{"key":"e_1_3_2_1_39_1","volume-title":"BitFlow: Exploiting Vector Parallelism for Binary Neural Networks on CPU. In 2018 IEEE International Parallel and Distributed Processing Symposium (IPDPS). 244\u2013253","author":"Hu Y.","year":"2018","unstructured":"Y. Hu and et al.. 2018. BitFlow: Exploiting Vector Parallelism for Binary Neural Networks on CPU. In 2018 IEEE International Parallel and Distributed Processing Symposium (IPDPS). 244\u2013253. https:\/\/doi.org\/10.1109\/IPDPS.2018.00034 10.1109\/IPDPS.2018.00034"},{"key":"e_1_3_2_1_40_1","volume-title":"Squeezenet: Alexnet-level accuracy with 50x fewer parameters and< 0.5 mb model size. arXiv preprint arXiv:1602.07360.","author":"Iandola Forrest N","year":"2016","unstructured":"Forrest N Iandola, Song Han, Matthew W Moskewicz, Khalid Ashraf, William J Dally, and Kurt Keutzer. 2016. Squeezenet: Alexnet-level accuracy with 50x fewer parameters and< 0.5 mb model size. arXiv preprint arXiv:1602.07360."},{"key":"e_1_3_2_1_41_1","volume-title":"4th Gen Xeon Scalable Processors. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/docs\/processors\/xeon-accelerated\/4th-gen-xeon-scalable-processors.html Accessed","year":"2023","unstructured":"Intel. 2023. 4th Gen Xeon Scalable Processors. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/docs\/processors\/xeon-accelerated\/4th-gen-xeon-scalable-processors.html Accessed: Aug 23, 2023"},{"key":"e_1_3_2_1_42_1","unstructured":"Intel. 2023. Intel oneDNN Developer Guide and Reference. https:\/\/github.com\/oneapi-src\/oneDNN\/blob\/master\/src\/cpu\/ Accessed: 18-05-2023"},{"key":"e_1_3_2_1_43_1","volume-title":"Intel\u00ae Advanced Matrix Extensions Overview. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/docs\/accelerator-engines\/advanced-matrix-extensions\/overview.html Accessed","year":"2023","unstructured":"Intel. 2023. Intel\u00ae Advanced Matrix Extensions Overview. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/docs\/accelerator-engines\/advanced-matrix-extensions\/overview.html Accessed: Aug 23, 2023"},{"key":"e_1_3_2_1_44_1","first-page":"711","article-title":"Data movement is all you need: A case study on optimizing transformers","volume":"3","author":"Ivanov Andrei","year":"2021","unstructured":"Andrei Ivanov, Nikoli Dryden, Tal Ben-Nun, Shigang Li, and Torsten Hoefler. 2021. Data movement is all you need: A case study on optimizing transformers. Proceedings of Machine Learning and Systems, 3 (2021), 711\u2013732.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the 27th International Conference on Field Programmable Logic and Applications (FPL).","author":"Jia Y.","unstructured":"Y. Jia, S. Yin, C. He, and T. Zhang. 2018. MLFusion: Multi-Layer Fusion for FPGA-Based CNN Accelerators. In Proceedings of the 27th International Conference on Field Programmable Logic and Applications (FPL)."},{"key":"e_1_3_2_1_46_1","volume-title":"Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. 109\u2013123","author":"Jia Zhen","year":"2018","unstructured":"Zhen Jia, Aleksandar Zlateski, Fredo Durand, and Kai Li. 2018. Optimizing N-dimensional, winograd-based convolution for manycore CPUs. In Proceedings of the 23rd ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming. 109\u2013123."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the 51st International Conference on Parallel Processing. 1\u201311","author":"Jiang Jiazhi","year":"2022","unstructured":"Jiazhi Jiang, Jiangsu Du, Dan Huang, Dongsheng Li, Jiang Zheng, and Yutong Lu. 2022. Characterizing and optimizing transformer inference on arm many-core processor. In Proceedings of the 51st International Conference on Parallel Processing. 1\u201311."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Yidi Jiang Bidisha Sharma Maulik Madhavi and Haizhou Li. 2021. Knowledge distillation from bert transformer to speech transformer for intent classification. arXiv preprint arXiv:2108.02598.","DOI":"10.21437\/Interspeech.2021-402"},{"key":"e_1_3_2_1_49_1","volume-title":"Fahad Shahbaz Khan, and Mubarak Shah.","author":"Khan Salman","year":"2022","unstructured":"Salman Khan, Muzammal Naseer, Munawar Hayat, Syed Waqas Zamir, Fahad Shahbaz Khan, and Mubarak Shah. 2022. Transformers in vision: A survey. ACM computing surveys (CSUR), 54, 10s (2022), 1\u201341."},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of the 17th ACM SIGPLAN symposium on Principles and Practice of Parallel Programming. 55\u201364","author":"Kim Seonggun","year":"2012","unstructured":"Seonggun Kim and Hwansoo Han. 2012. Efficient SIMD code generation for irregular kernels. In Proceedings of the 17th ACM SIGPLAN symposium on Principles and Practice of Parallel Programming. 55\u201364."},{"key":"e_1_3_2_1_51_1","volume-title":"Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, 25","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, 25 (2012)."},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the 52nd Annual IEEE\/ACM International Symposium on Microarchitecture. 754\u2013768","author":"Kwon Hyoukjun","year":"2019","unstructured":"Hyoukjun Kwon, Prasanth Chatarasi, Michael Pellauer, Angshuman Parashar, Vivek Sarkar, and Tushar Krishna. 2019. Understanding reuse, performance, and hardware cost of dnn dataflow: A data-centric approach. In Proceedings of the 52nd Annual IEEE\/ACM International Symposium on Microarchitecture. 754\u2013768."},{"key":"e_1_3_2_1_53_1","first-page":"24101","article-title":"A fast post-training pruning framework for transformers","volume":"35","author":"Kwon Woosuk","year":"2022","unstructured":"Woosuk Kwon, Sehoon Kim, Michael W Mahoney, Joseph Hassoun, Kurt Keutzer, and Amir Gholami. 2022. A fast post-training pruning framework for transformers. Advances in Neural Information Processing Systems, 35 (2022), 24101\u201324116.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Fran\u00e7ois Lagunas Ella Charlaix Victor Sanh and Alexander M Rush. 2021. Block pruning for faster transformers. arXiv preprint arXiv:2109.04838.","DOI":"10.18653\/v1\/2021.emnlp-main.829"},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the 15th International Conference on Information Processing in Sensor Networks, 1\u201312","author":"Lane Nicholas D","year":"2016","unstructured":"Nicholas D Lane, Sourav Bhattacharya, Petko Georgiev, Claudio Forlivesi, and Fahim Kawsar. 2016. Deepx: A software accelerator for low-power deep learning inference on mobile devices. Proceedings of the 15th International Conference on Information Processing in Sensor Networks, 1\u201312."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"crossref","first-page":"145","DOI":"10.1145\/358438.349320","article-title":"Exploiting superword level parallelism with multimedia instruction sets","volume":"35","author":"Larsen Samuel","year":"2000","unstructured":"Samuel Larsen and Saman Amarasinghe. 2000. Exploiting superword level parallelism with multimedia instruction sets. Acm Sigplan Notices, 35, 5 (2000), 145\u2013156.","journal-title":"Acm Sigplan Notices"},{"key":"e_1_3_2_1_57_1","volume-title":"Proceedings of the International Symposium on Code Generation and Optimization: Feedback-Directed and Runtime Optimization (CGO \u201904)","author":"Lattner Chris","year":"2004","unstructured":"Chris Lattner and Vikram Adve. 2004. LLVM: A Compilation Framework for Lifelong Program Analysis & Transformation. In Proceedings of the International Symposium on Code Generation and Optimization: Feedback-Directed and Runtime Optimization (CGO \u201904). IEEE Computer Society, USA. 75. isbn:0769521029"},{"key":"e_1_3_2_1_58_1","volume-title":"International conference on artificial neural networks. 60","author":"LeCun Yann","year":"1995","unstructured":"Yann LeCun, Larry Jackel, Leon Bottou, A Brunot, Corinna Cortes, John Denker, Harris Drucker, Isabelle Guyon, UA Muller, and Eduard Sackinger. 1995. Comparison of learning algorithms for handwritten digit recognition. In International conference on artificial neural networks. 60, 53\u201360."},{"key":"e_1_3_2_1_59_1","volume-title":"Proceedings of the 4th International Conference on Communication and Information Processing. 174\u2013179","author":"Lee Sung-Jin","year":"2018","unstructured":"Sung-Jin Lee, Sang-Soo Park, and Ki-Seok Chung. 2018. Efficient SIMD implementation for accelerating convolutional neural network. In Proceedings of the 4th International Conference on Communication and Information Processing. 174\u2013179."},{"key":"e_1_3_2_1_60_1","volume-title":"Proceedings of the 4th International Conference on Communication and Information Processing (ICCIP \u201918)","author":"Lee Sung-Jin","year":"2018","unstructured":"Sung-Jin Lee, Sang-Soo Park, and Ki-Seok Chung. 2018. Efficient SIMD Implementation for Accelerating Convolutional Neural Network. In Proceedings of the 4th International Conference on Communication and Information Processing (ICCIP \u201918). Association for Computing Machinery, New York, NY, USA. 174\u2013179. isbn:9781450365345 https:\/\/doi.org\/10.1145\/3290420.3290444 10.1145\/3290420.3290444"},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the 50th International Conference on Parallel Processing. 1\u201312","author":"Li Dongsheng","year":"2021","unstructured":"Dongsheng Li, Dan Huang, Zhiguang Chen, and Yutong Lu. 2021. Optimizing massively parallel winograd convolution on arm processor. In Proceedings of the 50th International Conference on Parallel Processing. 1\u201312."},{"key":"e_1_3_2_1_62_1","volume-title":"Proceedings of the 50th International Conference on Parallel Processing. 1\u201311","author":"Li Guangli","year":"2021","unstructured":"Guangli Li, Zhen Jia, Xiaobing Feng, and Yida Wang. 2021. Lowino: Towards efficient low-precision winograd convolutions on modern cpus. In Proceedings of the 50th International Conference on Parallel Processing. 1\u201311."},{"key":"e_1_3_2_1_63_1","unstructured":"ARM Limited. 2023. Architectures | Instruction sets | Intrinsics. https:\/\/developer.arm.com\/architectures\/instruction-sets\/intrinsics\/ Accessed: 2023-08-27"},{"key":"e_1_3_2_1_64_1","unstructured":"Arm Limited. 2023. NEON Programmer\u2019s Guide for Armv8-A. https:\/\/developer.arm.com\/documentation\/100069\/0101\/ Accessed: 2023-05-19"},{"key":"e_1_3_2_1_65_1","volume-title":"European Conference on Computer Vision. 33\u201349","author":"Liu Jihao","year":"2022","unstructured":"Jihao Liu, Xin Huang, Guanglu Song, Hongsheng Li, and Yu Liu. 2022. Uninet: Unified architecture search with convolution, transformer, and mlp. In European Conference on Computer Vision. 33\u201349."},{"key":"e_1_3_2_1_66_1","unstructured":"Ruiping Liu Kailun Yang Alina Roitberg Jiaming Zhang Kunyu Peng Huayao Liu and Rainer Stiefelhagen. 2022. TransKD: Transformer knowledge distillation for efficient semantic segmentation. arXiv preprint arXiv:2202.13393."},{"key":"e_1_3_2_1_67_1","unstructured":"Xingyu Liu Jeff Pool Song Han and William J Dally. 2018. Efficient sparse-winograd convolutional neural networks. arXiv preprint arXiv:1802.06367."},{"key":"e_1_3_2_1_68_1","volume-title":"Proc. USENIX Annu. Tech. Conf.. 1025\u20131040","author":"Liu Y.","unstructured":"Y. Liu, Y. Wang, R. Yu, M. Li, V. Sharma, and Y. Wang. 2019. Optimizing CNN model inference on CPUs. In Proc. USENIX Annu. Tech. Conf.. 1025\u20131040."},{"key":"e_1_3_2_1_69_1","volume-title":"Minotaur: A SIMD-Oriented Synthesizing Superoptimizer. arxiv:2306.00229.","author":"Liu Zhengyang","year":"2023","unstructured":"Zhengyang Liu, Stefan Mada, and John Regehr. 2023. Minotaur: A SIMD-Oriented Synthesizing Superoptimizer. arxiv:2306.00229."},{"key":"e_1_3_2_1_70_1","first-page":"28092","article-title":"Post-training quantization for vision transformer","volume":"34","author":"Liu Zhenhua","year":"2021","unstructured":"Zhenhua Liu, Yunhe Wang, Kai Han, Wei Zhang, Siwei Ma, and Wen Gao. 2021. Post-training quantization for vision transformer. Advances in Neural Information Processing Systems, 34 (2021), 28092\u201328103.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_71_1","volume-title":"MICRO-54: 54th Annual IEEE\/ACM International Symposium on Microarchitecture. 977\u2013991","author":"Lu Liqiang","year":"2021","unstructured":"Liqiang Lu, Yicheng Jin, Hangrui Bi, Zizhang Luo, Peng Li, Tao Wang, and Yun Liang. 2021. Sanger: A co-design framework for enabling sparse attention using reconfigurable architecture. In MICRO-54: 54th Annual IEEE\/ACM International Symposium on Microarchitecture. 977\u2013991."},{"key":"e_1_3_2_1_72_1","volume-title":"Proceedings of the 55th Annual Design Automation Conference (DAC \u201918)","author":"Lu Liqiang","year":"2018","unstructured":"Liqiang Lu and Yun Liang. 2018. SpWA: An Efficient Sparse Winograd Convolutional Neural Networks Accelerator on FPGAs. In Proceedings of the 55th Annual Design Automation Conference (DAC \u201918). Association for Computing Machinery, New York, NY, USA. Article 135, 6 pages. isbn:9781450357005 https:\/\/doi.org\/10.1145\/3195970.3196120 10.1145\/3195970.3196120"},{"key":"e_1_3_2_1_73_1","volume-title":"2019 2nd Workshop on Energy Efficient Machine Learning and Cognitive Computing for Embedded Applications (EMC2). 1\u20135.","author":"Maji Partha","year":"2019","unstructured":"Partha Maji, Andrew Mundy, Ganesh Dasika, Jesse Beu, Matthew Mattina, and Robert Mullins. 2019. Efficient winograd or cook-toom convolution kernel implementation on widely used mobile cpus. In 2019 2nd Workshop on Energy Efficient Machine Learning and Cognitive Computing for Embedded Applications (EMC2). 1\u20135."},{"key":"e_1_3_2_1_74_1","volume-title":"Optimizing Convolutions in State-of-the-Art Convolutional Neural Networks on Intel Xeon Phi. Ph. D. Dissertation","author":"Mandal Ankush","unstructured":"Ankush Mandal. 2017. Optimizing Convolutions in State-of-the-Art Convolutional Neural Networks on Intel Xeon Phi. Ph. D. Dissertation. Rice University."},{"key":"e_1_3_2_1_75_1","first-page":"1","article-title":"Tprune: Efficient transformer pruning for mobile devices","volume":"5","author":"Mao Jiachen","year":"2021","unstructured":"Jiachen Mao, Huanrui Yang, Ang Li, Hai Li, and Yiran Chen. 2021. Tprune: Efficient transformer pruning for mobile devices. ACM Transactions on Cyber-Physical Systems, 5, 3 (2021), 1\u201322.","journal-title":"ACM Transactions on Cyber-Physical Systems"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"crossref","first-page":"225","DOI":"10.1177\/1094342004038951","article-title":"Optimizing sparse matrix\u2013vector product computations using unroll and jam","volume":"18","author":"Mellor-Crummey John","year":"2004","unstructured":"John Mellor-Crummey and John Garvin. 2004. Optimizing sparse matrix\u2013vector product computations using unroll and jam. The International Journal of High Performance Computing Applications, 18, 2 (2004), 225\u2013236.","journal-title":"The International Journal of High Performance Computing Applications"},{"key":"e_1_3_2_1_77_1","volume-title":"Proc. ACM Program. Lang., 2, OOPSLA","author":"Mendis Charith","year":"2018","unstructured":"Charith Mendis and Saman Amarasinghe. 2018. GoSLP: Globally Optimized Superword Level Parallelism Framework. Proc. ACM Program. Lang., 2, OOPSLA (2018), Article 110, oct, 28 pages. https:\/\/doi.org\/10.1145\/3276480 10.1145\/3276480"},{"key":"e_1_3_2_1_78_1","unstructured":"Lingchuan Meng and John Brothers. 2019. Efficient winograd convolution via integer arithmetic. arXiv preprint arXiv:1901.01965."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"crossref","first-page":"5095","DOI":"10.1109\/TNNLS.2021.3071762","article-title":"A survey of deep learning on CPUs: opportunities and co-optimizations","volume":"33","author":"Mittal Sparsh","year":"2021","unstructured":"Sparsh Mittal, Poonam Rajput, and Sreenivas Subramoney. 2021. A survey of deep learning on CPUs: opportunities and co-optimizations. IEEE Transactions on Neural Networks and Learning Systems, 33, 10 (2021), 5095\u20135115.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"e_1_3_2_1_80_1","volume-title":"PyTorch: An Imperative Style","author":"Paszke Adam","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas Kopf, Edward Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: An Imperative Style, High-Performance Deep Learning Library."},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"crossref","unstructured":"Gabriele Prato Ella Charlaix and Mehdi Rezagholizadeh. 2019. Fully quantized transformer for improved translation.","DOI":"10.18653\/v1\/2020.findings-emnlp.1"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"crossref","first-page":"472","DOI":"10.1109\/TCSVT.2011.2125590","article-title":"From Xetal-II to Xetal-Pro: On the road toward an ultralow-energy and high-throughput SIMD processor","volume":"21","author":"Pu Yu","year":"2011","unstructured":"Yu Pu, Yifan He, Zhenyu Ye, Sebastian Moreno Londono, Anteneh Alemu Abbo, Richard Kleihorst, and Henk Corporaal. 2011. From Xetal-II to Xetal-Pro: On the road toward an ultralow-energy and high-throughput SIMD processor. IEEE Transactions on Circuits and Systems for Video Technology, 21, 4 (2011), 472\u2013484.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"e_1_3_2_1_83_1","volume-title":"Layer Fusion for Memory-Efficient Inference of Convolutional Neural Networks on GPUs. In International Symposium on Benchmarking, Measuring and Optimizing (Bench).","author":"Qiao Y.","unstructured":"Y. Qiao, Y. Zhang, J. Wang, T. Tang, and Y. Wang. 2019. Layer Fusion for Memory-Efficient Inference of Convolutional Neural Networks on GPUs. In International Symposium on Benchmarking, Measuring and Optimizing (Bench)."},{"key":"e_1_3_2_1_84_1","volume-title":"Proceedings of the 34th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI).","author":"Ragan-Kelley J.","unstructured":"J. Ragan-Kelley, C. Barnes, A. Adams, S. Paris, F. Durand, and S. Amarasinghe. 2013. Halide: A Language and Compiler for Optimizing Parallelism, Locality, and Recomputation in Image Processing Pipelines. In Proceedings of the 34th ACM SIGPLAN Conference on Programming Language Design and Implementation (PLDI)."},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"crossref","unstructured":"M. Rhu N. Gimelshein J. Clemons A. Zulfiqar and S. W. Keckler. 2016. vDNN: Virtualized Deep Neural Networks for Scalable Memory-Efficient Neural Network Design. In Advances in Neural Information Processing Systems (NIPS).","DOI":"10.1109\/MICRO.2016.7783721"},{"key":"e_1_3_2_1_86_1","volume-title":"Scale-sim: Systolic cnn accelerator simulator. arXiv preprint arXiv:1811.02883.","author":"Samajdar Ananda","year":"2018","unstructured":"Ananda Samajdar, Yuhao Zhu, Paul Whatmough, Matthew Mattina, and Tushar Krishna. 2018. Scale-sim: Systolic cnn accelerator simulator. arXiv preprint arXiv:1811.02883."},{"key":"e_1_3_2_1_87_1","volume-title":"Proceedings of the 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming (PPoPP \u201923)","author":"de Limas Santana Alexandre","year":"2023","unstructured":"Alexandre de Limas Santana, Adri\u00e0 Armejach, and Marc Casas. 2023. Efficient Direct Convolution Using Long SIMD Instructions. In Proceedings of the 28th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming (PPoPP \u201923). Association for Computing Machinery, New York, NY, USA. 342\u2013353. isbn:9798400700156 https:\/\/doi.org\/10.1145\/3572848.3577435 10.1145\/3572848.3577435"},{"key":"e_1_3_2_1_88_1","volume-title":"Muhammad Haris Khan, Munawar Hayat, Fahad Shahbaz Khan, and Huazhu Fu.","author":"Shamshad Fahad","year":"2023","unstructured":"Fahad Shamshad, Salman Khan, Syed Waqas Zamir, Muhammad Haris Khan, Munawar Hayat, Fahad Shahbaz Khan, and Huazhu Fu. 2023. Transformers in medical imaging: A survey. Medical Image Analysis, 102802."},{"key":"e_1_3_2_1_89_1","volume-title":"Proceedings of the 59th ACM\/IEEE Design Automation Conference. 571\u2013576","author":"Shen Guan","year":"2022","unstructured":"Guan Shen, Jieru Zhao, Quan Chen, Jingwen Leng, Chao Li, and Minyi Guo. 2022. SALO: an efficient spatial accelerator enabling hybrid sparse attention mechanisms for long sequences. In Proceedings of the 59th ACM\/IEEE Design Automation Conference. 571\u2013576."},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2016.2579198"},{"key":"e_1_3_2_1_91_1","volume-title":"Edge computing: Vision and challenges","author":"Shi Weisong","year":"2016","unstructured":"Weisong Shi, Jie Cao, Quan Zhang, Youhuizi Li, and Lanyu Xu. 2016. Edge computing: Vision and challenges. IEEE internet of things journal, 3, 5 (2016), 637\u2013646."},{"key":"e_1_3_2_1_92_1","unstructured":"Laurent Sifre and St\u00e9phane Mallat. 2014. Rigid-motion scattering for texture classification. arXiv preprint arXiv:1403.1687."},{"key":"e_1_3_2_1_93_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556."},{"key":"e_1_3_2_1_94_1","volume-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 21\u201325","author":"Subakan Cem","year":"2021","unstructured":"Cem Subakan, Mirco Ravanelli, Samuele Cornell, Mirko Bronzi, and Jianyuan Zhong. 2021. Attention is all you need in speech separation. In ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 21\u201325."},{"key":"e_1_3_2_1_95_1","doi-asserted-by":"crossref","first-page":"2295","DOI":"10.1109\/JPROC.2017.2761740","article-title":"Efficient processing of deep neural networks: A tutorial and survey","volume":"105","author":"Sze Vivienne","year":"2017","unstructured":"Vivienne Sze, Yu-Hsin Chen, Tien-Ju Yang, and Joel S Emer. 2017. Efficient processing of deep neural networks: A tutorial and survey. Proc. IEEE, 105, 12 (2017), 2295\u20132329.","journal-title":"Proc. IEEE"},{"key":"e_1_3_2_1_96_1","volume-title":"Tensor Comprehensions: Framework-Agnostic High-Performance Machine Learning Abstractions. In International Conference on Learning Representations (ICLR).","author":"Vasilache N.","unstructured":"N. Vasilache, J. Johnson, M. Mathieu, S. Chintala, S. Piantino, and Y. LeCun. 2018. Tensor Comprehensions: Framework-Agnostic High-Performance Machine Learning Abstractions. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_97_1","volume-title":"\u0141 ukasz Kaiser, and Illia Polosukhin","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141 ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, 30 (2017)."},{"key":"e_1_3_2_1_98_1","doi-asserted-by":"crossref","first-page":"1275","DOI":"10.1177\/1094342019866247","article-title":"SWIRL: High-performance many-core CPU code generation for deep neural networks","volume":"33","author":"Venkat Anand","year":"2019","unstructured":"Anand Venkat, Tharindu Rusira, Raj Barik, Mary Hall, and Leonard Truong. 2019. SWIRL: High-performance many-core CPU code generation for deep neural networks. The International Journal of High Performance Computing Applications, 33, 6 (2019), 1275\u20131289.","journal-title":"The International Journal of High Performance Computing Applications"},{"key":"e_1_3_2_1_99_1","volume-title":"Hat: Hardware-aware transformers for efficient natural language processing. arXiv preprint arXiv:2005.14187.","author":"Wang Hanrui","year":"2020","unstructured":"Hanrui Wang, Zhanghao Wu, Zhijian Liu, Han Cai, Ligeng Zhu, Chuang Gan, and Song Han. 2020. Hat: Hardware-aware transformers for efficient natural language processing. arXiv preprint arXiv:2005.14187."},{"key":"e_1_3_2_1_100_1","first-page":"5776","article-title":"Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers","volume":"33","author":"Wang Wenhui","year":"2020","unstructured":"Wenhui Wang, Furu Wei, Li Dong, Hangbo Bao, Nan Yang, and Ming Zhou. 2020. Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. Advances in Neural Information Processing Systems, 33 (2020), 5776\u20135788.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_101_1","doi-asserted-by":"crossref","unstructured":"Shmuel Winograd. 1980. Arithmetic complexity of computations. 33 Siam.","DOI":"10.1137\/1.9781611970364"},{"key":"e_1_3_2_1_102_1","volume-title":"Proceedings of the ACM Web Conference 2022 (WWW \u201922)","author":"Wu Ruofan","year":"2022","unstructured":"Ruofan Wu, Feng Zhang, Jiawei Guan, Zhen Zheng, Xiaoyong Du, and Xipeng Shen. 2022. DREW: Efficient Winograd CNN Inference with Deep Reuse. In Proceedings of the ACM Web Conference 2022 (WWW \u201922). Association for Computing Machinery, New York, NY, USA. 1807\u20131816. isbn:9781450390965 https:\/\/doi.org\/10.1145\/3485447.3511985 10.1145\/3485447.3511985"},{"key":"e_1_3_2_1_103_1","unstructured":"Yao Xiao Nesreen Ahmed Mihai Capot\u0103 Guixiang Ma Theodore L Willke Shahin Nazarian and Paul Bogdan. 2022. Structural Code Representation Learning for Auto-Vectorization."},{"key":"e_1_3_2_1_104_1","volume-title":"Proceedings of the 25th ACM SIGPLAN symposium on principles and practice of parallel programming. 32\u201344","author":"Yan Da","year":"2020","unstructured":"Da Yan, Wei Wang, and Xiaowen Chu. 2020. Optimizing batched winograd convolution on GPUs. In Proceedings of the 25th ACM SIGPLAN symposium on principles and practice of parallel programming. 32\u201344."},{"key":"e_1_3_2_1_105_1","unstructured":"Chenghao Yang Hongyuan Mei and Jason Eisner. 2021. Transformer embeddings of irregularly spaced events and their participants. arXiv preprint arXiv:2201.00044."},{"key":"e_1_3_2_1_106_1","volume-title":"2020 53rd Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO). 711\u2013724","author":"Yang Dingqing","year":"2020","unstructured":"Dingqing Yang, Amin Ghasemazar, Xiaowei Ren, Maximilian Golub, Guy Lemieux, and Mieszko Lis. 2020. Procrustes: a dataflow and accelerator for sparse deep neural network training. In 2020 53rd Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO). 711\u2013724."},{"key":"e_1_3_2_1_107_1","doi-asserted-by":"crossref","first-page":"1935","DOI":"10.1093\/jamia\/ocaa189","article-title":"Clinical concept extraction using transformers","volume":"27","author":"Yang Xi","year":"2020","unstructured":"Xi Yang, Jiang Bian, William R Hogan, and Yonghui Wu. 2020. Clinical concept extraction using transformers. Journal of the American Medical Informatics Association, 27, 12 (2020), 1935\u20131942.","journal-title":"Journal of the American Medical Informatics Association"},{"key":"e_1_3_2_1_108_1","volume-title":"Proceedings of the Twenty-Fifth International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS \u201920)","author":"Yang Xuan","year":"2020","unstructured":"Xuan Yang, Mingyu Gao, Qiaoyi Liu, Jeff Setter, Jing Pu, Ankita Nayak, Steven Bell, Kaidi Cao, Heonjae Ha, Priyanka Raina, Christos Kozyrakis, and Mark Horowitz. 2020. Interstellar: Using Halide\u2019s Scheduling Language to Analyze DNN Accelerators. In Proceedings of the Twenty-Fifth International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS \u201920). 369\u2013383. https:\/\/doi.org\/10.1145\/3373376.3378514 10.1145\/3373376.3378514"},{"key":"e_1_3_2_1_109_1","volume-title":"Autotinybert: Automatic hyper-parameter optimization for efficient pre-trained language models. arXiv preprint arXiv:2107.13686.","author":"Yin Yichun","year":"2021","unstructured":"Yichun Yin, Cheng Chen, Lifeng Shang, Xin Jiang, Xiao Chen, and Qun Liu. 2021. Autotinybert: Automatic hyper-parameter optimization for efficient pre-trained language models. arXiv preprint arXiv:2107.13686."},{"key":"e_1_3_2_1_110_1","volume-title":"2023 IEEE International Symposium on High-Performance Computer Architecture (HPCA). 273\u2013286","author":"You Haoran","year":"2023","unstructured":"Haoran You, Zhanyi Sun, Huihong Shi, Zhongzhi Yu, Yang Zhao, Yongan Zhang, Chaojian Li, Baopu Li, and Yingyan Lin. 2023. Vitcod: Vision transformer acceleration via dedicated algorithm and accelerator co-design. In 2023 IEEE International Symposium on High-Performance Computer Architecture (HPCA). 273\u2013286."},{"key":"e_1_3_2_1_111_1","doi-asserted-by":"crossref","unstructured":"Xiangyu Zhang Xinyu Zhou Mengxiao Lin and Jian Sun. 2017. ShuffleNet: An Extremely Efficient Convolutional Neural Network for Mobile Devices. arxiv:1707.01083.","DOI":"10.1109\/CVPR.2018.00716"},{"key":"e_1_3_2_1_112_1","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition, 6848\u20136856","author":"Zhang Xiangyu","year":"2018","unstructured":"Xiangyu Zhang, Xinyu Zhou, Mengxiao Lin, and Jian Sun. 2018. ShuffleNet: An Extremely Efficient Convolutional Neural Network for Mobile Devices. Proceedings of the IEEE conference on computer vision and pattern recognition, 6848\u20136856."},{"key":"e_1_3_2_1_113_1","doi-asserted-by":"crossref","first-page":"1982","DOI":"10.1109\/TPDS.2023.3269530","article-title":"NIOT: A Novel Inference Optimization of Transformers on Modern CPUs","volume":"34","author":"Zhang Zining","year":"2023","unstructured":"Zining Zhang, Yao Chen, Bingsheng He, and Zhenjie Zhang. 2023. NIOT: A Novel Inference Optimization of Transformers on Modern CPUs. In IEEE Transactions on Parallel and Distributed Systems. 34, 1982\u20131995.","journal-title":"IEEE Transactions on Parallel and Distributed Systems."},{"key":"e_1_3_2_1_114_1","first-page":"281","article-title":"An fpga-based transformer accelerator using output block stationary dataflow for object recognition applications","volume":"70","author":"Zhao Zhongyu","year":"2022","unstructured":"Zhongyu Zhao, Rujian Cao, Ka-Fai Un, Wei-Han Yu, Pui-In Mak, and Rui P Martins. 2022. An fpga-based transformer accelerator using output block stationary dataflow for object recognition applications. IEEE Transactions on Circuits and Systems II: Express Briefs, 70, 1 (2022), 281\u2013285.","journal-title":"IEEE Transactions on Circuits and Systems II: Express Briefs"},{"key":"e_1_3_2_1_115_1","unstructured":"Mingjian Zhu Yehui Tang and Kai Han. 2021. Vision transformer pruning. arXiv preprint arXiv:2104.08500."}],"event":{"name":"CC '24: 33rd ACM SIGPLAN International Conference on Compiler Construction","location":"Edinburgh United Kingdom","acronym":"CC '24","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 33rd ACM SIGPLAN International Conference on Compiler Construction"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640537.3641566","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3640537.3641566","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T22:50:24Z","timestamp":1750287024000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3640537.3641566"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,17]]},"references-count":115,"alternative-id":["10.1145\/3640537.3641566","10.1145\/3640537"],"URL":"https:\/\/doi.org\/10.1145\/3640537.3641566","relation":{},"subject":[],"published":{"date-parts":[[2024,2,17]]},"assertion":[{"value":"2024-02-20","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}