{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T17:58:40Z","timestamp":1785952720835,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,2,22]],"date-time":"2022-02-22T00:00:00Z","timestamp":1645488000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,2,28]]},"DOI":"10.1145\/3503222.3507723","type":"proceedings-article","created":{"date-parts":[[2022,2,22]],"date-time":"2022-02-22T20:49:01Z","timestamp":1645562941000},"page":"359-373","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":76,"title":["AStitch: enabling a new multi-dimensional optimization space for memory-intensive ML training and inference on modern SIMT architectures"],"prefix":"10.1145","author":[{"given":"Zhen","family":"Zheng","sequence":"first","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuanda","family":"Yang","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pengzhan","family":"Zhao","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guoping","family":"Long","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kai","family":"Zhu","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feiwen","family":"Zhu","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenyi","family":"Zhao","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoyong","family":"Liu","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Yang","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jidong","family":"Zhai","sequence":"additional","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuaiwen Leon","family":"Song","sequence":"additional","affiliation":[{"name":"University of Sydney, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Lin","sequence":"additional","affiliation":[{"name":"Alibaba Group, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,2,22]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Cited July 2021. AMD GPU-Powered Machine Learning Solutions. https:\/\/www.amd.com\/en\/graphics\/servers-radeon-instinct-deep-learning"},{"key":"e_1_3_2_1_2_1","unstructured":"Cited July 2021. Automatic Mixed Precision for Deep Learning. https:\/\/developer.nvidia.com\/automatic-mixed-precision"},{"key":"e_1_3_2_1_3_1","unstructured":"Cited July 2021. CUDA Achieved Occupancy. https:\/\/docs.nvidia.com\/gameworks\/content\/developertools\/desktop\/analysis\/report\/cudaexperiments\/kernellevel\/achievedoccupancy.htm"},{"key":"e_1_3_2_1_4_1","unstructured":"Cited July 2021. CUDA Occupancy Calculator. https:\/\/docs.nvidia.com\/cuda\/cuda-occupancy-calculator\/index.html"},{"key":"e_1_3_2_1_5_1","unstructured":"Cited July 2021. Getting Started with CUDA Graphs. https:\/\/developer.nvidia.com\/blog\/cuda-graphs\/"},{"key":"e_1_3_2_1_6_1","unstructured":"Cited July 2021. GPU Dominates AI Accelerator Market. https:\/\/www.informationweek.com\/ai-or-machine-learning\/gpus-continue-to-dominate-the-ai-accelerator-market-for-now"},{"key":"e_1_3_2_1_7_1","unstructured":"Cited July 2021. NVIDIA cuBLAS. https:\/\/developer.nvidia.com\/cublas"},{"key":"e_1_3_2_1_8_1","unstructured":"Cited July 2021. NVIDIA cuDNN. https:\/\/developer.nvidia.com\/cudnn"},{"key":"e_1_3_2_1_9_1","unstructured":"Cited July 2021. NVIDIA FasterTransformer. https:\/\/github.com\/NVIDIA\/FasterTransformer"},{"key":"e_1_3_2_1_10_1","unstructured":"Cited July 2021. Nvprof Profiling Tool. https:\/\/docs.nvidia.com\/cuda\/profiler-users-guide\/index.html####nvprof-overview"},{"key":"e_1_3_2_1_11_1","unstructured":"Cited July 2021. TensorFlow XLA. https:\/\/www.tensorflow.org\/xla"},{"key":"e_1_3_2_1_12_1","volume-title":"Tensorflow: A system for large-scale machine learning. In 12th $USENIX$ symposium on operating systems design and implementation ($OSDI$ 16). 265\u2013283.","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, and Michael Isard. 2016. Tensorflow: A system for large-scale machine learning. In 12th $USENIX$ symposium on operating systems design and implementation ($OSDI$ 16). 265\u2013283."},{"key":"e_1_3_2_1_13_1","volume-title":"NeurIPS ML for Systems Workshop.","author":"Abdolrashidi Amirali","year":"2019","unstructured":"Amirali Abdolrashidi, Qiumin Xu, Shibo Wang, Sudip Roy, and Yanqi Zhou. 2019. Learning to fuse. In NeurIPS ML for Systems Workshop."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3306346.3322967"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2858788.2688521"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.5555\/3314872.3314896"},{"key":"e_1_3_2_1_17_1","volume-title":"Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv preprint arXiv:1512.01274.","author":"Chen Tianqi","year":"2015","unstructured":"Tianqi Chen, Mu Li, Yutian Li, Min Lin, Naiyan Wang, Minjie Wang, Tianjun Xiao, Bing Xu, Chiyuan Zhang, and Zheng Zhang. 2015. Mxnet: A flexible and efficient machine learning library for heterogeneous distributed systems. arXiv preprint arXiv:1512.01274."},{"key":"e_1_3_2_1_18_1","volume-title":"13th $USENIX$ Symposium on Operating Systems Design and Implementation ($OSDI$ 18). 578\u2013594.","author":"Chen Tianqi","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, and Luis Ceze. 2018. $TVM$: An automated end-to-end optimizing compiler for deep learning. In 13th $USENIX$ Symposium on Operating Systems Design and Implementation ($OSDI$ 18). 578\u2013594."},{"key":"e_1_3_2_1_19_1","first-page":"1","article-title":"A simple, fast dominance algorithm","volume":"4","author":"Cooper Keith D","year":"2001","unstructured":"Keith D Cooper, Timothy J Harvey, and Ken Kennedy. 2001. A simple, fast dominance algorithm. Software Practice & Experience, 4, 1-10 (2001), 1\u20138.","journal-title":"Software Practice & Experience"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3368826.3377912"},{"key":"e_1_3_2_1_21_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805.","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805."},{"key":"e_1_3_2_1_22_1","volume-title":"International Conference on Machine Learning. 2024\u20132033","author":"Diamos Greg","year":"2016","unstructured":"Greg Diamos, Shubho Sengupta, Bryan Catanzaro, Mike Chrzanowski, Adam Coates, Erich Elsen, Jesse Engel, Awni Hannun, and Sanjeev Satheesh. 2016. Persistent rnns: Stashing recurrent weights on-chip. In International Conference on Machine Learning. 2024\u20132033."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441578"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3123970"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359630"},{"key":"e_1_3_2_1_26_1","unstructured":"Wonkyung Jung Daejin Jung Sunjung Lee Wonjong Rhee and Jung Ho Ahn. 2018. Restructuring batch normalization to accelerate CNN training. arXiv preprint arXiv:1807.01702."},{"key":"e_1_3_2_1_27_1","volume-title":"Joanna P Simpson, Andrew D Kane, David K Menon, Daniel Rueckert, and Ben Glocker.","author":"Kamnitsas Konstantinos","year":"2017","unstructured":"Konstantinos Kamnitsas, Christian Ledig, Virginia FJ Newcombe, Joanna P Simpson, Andrew D Kane, David K Menon, Daniel Rueckert, and Ben Glocker. 2017. Efficient multi-scale 3D CNN with fully connected CRF for accurate brain lesion segmentation. Medical image analysis, 36 (2017), 61\u201378."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.5555\/3314872.3314885"},{"key":"e_1_3_2_1_29_1","volume-title":"Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, 25","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, 25 (2012), 1097\u20131105."},{"key":"e_1_3_2_1_30_1","volume-title":"Nimble: Lightweight and Parallel GPU Task Scheduling for Deep Learning. arXiv preprint arXiv:2012.02732.","author":"Kwon Woosuk","year":"2020","unstructured":"Woosuk Kwon, Gyeong-In Yu, Eunji Jeong, and Byung-Gon Chun. 2020. Nimble: Lightweight and Parallel GPU Task Scheduling for Deep Learning. arXiv preprint arXiv:2012.02732."},{"key":"e_1_3_2_1_31_1","unstructured":"Ao Li Bojian Zheng Gennady Pekhimenko and Fan Long. 2020. Automatic Horizontal Fusion for GPU Kernels. arXiv preprint arXiv:2007.01277."},{"key":"e_1_3_2_1_32_1","volume-title":"Rammer: Enabling Holistic Deep Learning Compiler Optimizations with rTasks. In 14th $USENIX$ Symposium on Operating Systems Design and Implementation ($OSDI$ 20). 881\u2013897.","author":"Ma Lingxiao","year":"2020","unstructured":"Lingxiao Ma, Zhiqiang Xie, Zhi Yang, Jilong Xue, Youshan Miao, Wei Cui, Wenxiang Hu, Fan Yang, Lintao Zhang, and Lidong Zhou. 2020. Rammer: Enabling Holistic Deep Learning Compiler Optimizations with rTasks. In 14th $USENIX$ Symposium on Operating Systems Design and Implementation ($OSDI$ 20). 881\u2013897."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3075564.3077382"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453483.3454083"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Alberto Parravicini Arnaud Delamare Marco Arnaboldi and Marco D Santambrogio. 2020. DAG-based Scheduling with Resource Sharing for Multi-task Applications in a Polyglot GPU Runtime. arXiv preprint arXiv:2012.09646.","DOI":"10.1109\/IPDPS49936.2021.00020"},{"key":"e_1_3_2_1_36_1","volume-title":"Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703.","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, and Luca Antiga. 2019. Pytorch: An imperative style, high-performance deep learning library. arXiv preprint arXiv:1912.01703."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3207719.3207723"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2019.8661176"},{"key":"e_1_3_2_1_39_1","volume-title":"Nimble: Efficiently compiling dynamic neural networks for model inference. arXiv preprint arXiv:2006.03031.","author":"Shen Haichen","year":"2020","unstructured":"Haichen Shen, Jared Roesch, Zhi Chen, Wei Chen, Yong Wu, Mu Li, Vin Sharma, Zachary Tatlock, and Yida Wang. 2020. Nimble: Efficiently compiling dynamic neural networks for model inference. arXiv preprint arXiv:2006.03031."},{"key":"e_1_3_2_1_40_1","volume-title":"An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition","author":"Shi Baoguang","year":"2016","unstructured":"Baoguang Shi, Xiang Bai, and Cong Yao. 2016. An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE transactions on pattern analysis and machine intelligence, 39, 11 (2016), 2298\u20132304."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304072"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/2908080.2908105"},{"key":"e_1_3_2_1_43_1","unstructured":"Nicolas Vasilache Oleksandr Zinenko Theodoros Theodoridis Priya Goyal Zachary DeVito William S Moses Sven Verdoolaege Andrew Adams and Albert Cohen. 2018. Tensor comprehensions: Framework-agnostic high-performance machine learning abstractions. arXiv preprint arXiv:1802.04730."},{"key":"e_1_3_2_1_44_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. arXiv preprint arXiv:1706.03762."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2014.21"},{"key":"e_1_3_2_1_46_1","volume-title":"Accelerating Deep Learning Inference with Cross-Layer Data Reuse on GPUs. In European Conference on Parallel Processing. 219\u2013233","author":"Wang Xueying","year":"2020","unstructured":"Xueying Wang, Guangli Li, Xiao Dong, Jiansong Li, Lei Liu, and Xiaobing Feng. 2020. Accelerating Deep Learning Inference with Cross-Layer Data Reuse on GPUs. In European Conference on Parallel Processing. 219\u2013233."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Thomas Wolf Lysandre Debut Victor Sanh Julien Chaumond Clement Delangue Anthony Moi Pierric Cistac Tim Rault R\u00e9mi Louf and Morgan Funtowicz. 2019. Huggingface\u2019s transformers: State-of-the-art natural language processing. arXiv preprint arXiv:1910.03771.","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2012.19"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2010.5470477"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE51399.2021.00148"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2016.2586074"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3093234"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00778-020-00636-3"},{"key":"e_1_3_2_1_55_1","volume-title":"Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, and Koushik Sen.","author":"Zheng Lianmin","year":"2020","unstructured":"Lianmin Zheng, Chengfan Jia, Minmin Sun, Zhao Wu, Cody Hao Yu, Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, and Koushik Sen. 2020. Ansor: Generating high-performance tensor programs for deep learning. In 14th $USENIX$ Symposium on Operating Systems Design and Implementation ($OSDI$ 20). 863\u2013879."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3123978"},{"key":"e_1_3_2_1_57_1","unstructured":"Zhen Zheng Pengzhan Zhao Guoping Long Feiwen Zhu Kai Zhu Wenyi Zhao Lansong Diao Jun Yang and Wei Lin. 2020. Fusionstitching: boosting memory intensive computations for deep learning workloads. arXiv preprint arXiv:2009.10924."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015941"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437984.3458838"}],"event":{"name":"ASPLOS '22: 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Lausanne Switzerland","acronym":"ASPLOS '22","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGOPS ACM Special Interest Group on Operating Systems","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503222.3507723","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503222.3507723","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:11:39Z","timestamp":1750191099000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503222.3507723"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,2,22]]},"references-count":59,"alternative-id":["10.1145\/3503222.3507723","10.1145\/3503222"],"URL":"https:\/\/doi.org\/10.1145\/3503222.3507723","relation":{},"subject":[],"published":{"date-parts":[[2022,2,22]]},"assertion":[{"value":"2022-02-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}