{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T08:20:26Z","timestamp":1781857226180,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,2,17]],"date-time":"2023-02-17T00:00:00Z","timestamp":1676592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,2,17]]},"DOI":"10.1145\/3579990.3580015","type":"proceedings-article","created":{"date-parts":[[2023,2,22]],"date-time":"2023-02-22T10:27:10Z","timestamp":1677061630000},"page":"236-248","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["Accelerating Deep Neural Networks on Mobile Multicore NPUs"],"prefix":"10.1145","author":[{"given":"Hanwoong","family":"Jung","sequence":"first","affiliation":[{"name":"Samsung Advanced Institute of Technology, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hexiang","family":"Ji","sequence":"additional","affiliation":[{"name":"Samsung Research, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alexey","family":"Pushchin","sequence":"additional","affiliation":[{"name":"Samsung Research, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maxim","family":"Ostapenko","sequence":"additional","affiliation":[{"name":"Samsung Advanced Institute of Technology, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenlong","family":"Niu","sequence":"additional","affiliation":[{"name":"Samsung Research, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ilya","family":"Palachev","sequence":"additional","affiliation":[{"name":"Samsung Research, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yutian","family":"Qu","sequence":"additional","affiliation":[{"name":"Samsung Research, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pavel","family":"Fedin","sequence":"additional","affiliation":[{"name":"Samsung Research, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuri","family":"Gribov","sequence":"additional","affiliation":[{"name":"Samsung Research, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Heewoo","family":"Nam","sequence":"additional","affiliation":[{"name":"Samsung Advanced Institute of Technology, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongguen","family":"Lim","sequence":"additional","affiliation":[{"name":"Samsung Advanced Institute of Technology, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hyunjun","family":"Kim","sequence":"additional","affiliation":[{"name":"Samsung Advanced Institute of Technology, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Joonho","family":"Song","sequence":"additional","affiliation":[{"name":"Samsung Advanced Institute of Technology, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Seungwon","family":"Lee","sequence":"additional","affiliation":[{"name":"Samsung Advanced Institute of Technology, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hwansoo","family":"Han","sequence":"additional","affiliation":[{"name":"Sungkyunkwan University, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,2,22]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3306346.3322967"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","unstructured":"Byung Hoon Ahn Prannoy Pilligundla and Hadi Esmaeilzadeh. 2019. Reinforcement Learning and Adaptive Sampling for Optimized DNN Compilation. https:\/\/doi.org\/10.48550\/ARXIV.1905.12799","DOI":"10.48550\/ARXIV.1905.12799"},{"key":"e_1_3_2_1_3_1","volume-title":"2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO). 1\u201312","author":"Alwani M.","unstructured":"M. Alwani, H. Chen, M. Ferdman, and P. Milder. 2016. Fused-layer CNN accelerators. In 2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO). 1\u201312."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2207.00032"},{"key":"e_1_3_2_1_5_1","unstructured":"Apache. 2020. TVM. https:\/\/tvm.apache.org\/"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Liang-Chieh Chen Yukun Zhu George Papandreou Florian Schroff and Hartwig Adam. 2018. Encoder-Decoder with Atrous Separable Convolution for Semantic Image Segmentation. In ECCV.","DOI":"10.1007\/978-3-030-01234-2_49"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2541940.2541967"},{"key":"e_1_3_2_1_8_1","volume-title":"TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, Carlos Guestrin, and Arvind Krishnamurthy. 2018. TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). USENIX Association, Carlsbad, CA. 578\u2013594. isbn:978-1-939133-08-3 https:\/\/www.usenix.org\/conference\/osdi18\/presentation\/chen"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001177"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.23919\/DATE51398.2021.9473965"},{"key":"e_1_3_2_1_11_1","first-page":"3","volume-title":"Proceedings of Machine Learning and Systems","author":"Ding Yaoyao","year":"2021","unstructured":"Yaoyao Ding, Ligeng Zhu, Zhihao Jia, Gennady Pekhimenko, and Song Han. 2021. Ios: Inter-operator scheduler for cnn acceleration. Proceedings of Machine Learning and Systems, 3 (2021)."},{"key":"e_1_3_2_1_12_1","unstructured":"ETH. 2020. AI-Benchmark. http:\/\/ai-benchmark.com\/"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441593"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304014"},{"key":"e_1_3_2_1_15_1","unstructured":"Halide. 2012. A language for fast portable computation on images and tensors. https:\/\/halide-lang.org\/"},{"key":"e_1_3_2_1_16_1","unstructured":"Huawei. 2019. Huawei launches Ascend 910 the world\u2019s most powerful AI processor and MindSpore an all-scenario AI computing framework. https:\/\/www.huawei.com\/en\/news\/2019\/8\/huawei-ascend-910-most-powerful-ai-processor"},{"key":"e_1_3_2_1_17_1","volume-title":"Sparsity-Aware and Re-configurable NPU Architecture for Samsung Flagship Mobile SoC. 2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture (ISCA), 15\u201328","author":"Jang Jun-Woo","year":"2021","unstructured":"Jun-Woo Jang, Sehwan Lee, Dongyoung Kim, Hyunsun Park, Ali Shafiee Ardestani, Yeongjae Choi, Channoh Kim, Yoojin Kim, Hyeongseok Yu, Hamzah Abdel-Aziz, Jun-Seok Park, Heonsoo Lee, Dongwoo Lee, Myeong Woo Kim, Hanwoong Jung, Heewoo Nam, Dongguen Lim, Seungwon Lee, Joon-Ho Song, Suknam Kwon, Joseph Hassoun, SukHwan Lim, and Changkyu Choi. 2021. Sparsity-Aware and Re-configurable NPU Architecture for Samsung Flagship Mobile SoC. 2021 ACM\/IEEE 48th Annual International Symposium on Computer Architecture (ISCA), 15\u201328."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2977496"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2019.2946140"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.23919\/DATE.2018.8342033"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446759"},{"key":"e_1_3_2_1_22_1","volume-title":"Berg","author":"Liu Wei","year":"2016","unstructured":"Wei Liu, Dragomir Anguelov, Dumitru Erhan, Christian Szegedy, Scott Reed, Cheng-Yang Fu, and Alexander C. Berg. 2016. SSD: Single Shot MultiBox Detector. In ECCV."},{"key":"e_1_3_2_1_23_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Ma Lingxiao","year":"2020","unstructured":"Lingxiao Ma, Zhiqiang Xie, Zhi Yang, Jilong Xue, Youshan Miao, Wei Cui, Wenxiang Hu, Fan Yang, Lintao Zhang, and Lidong Zhou. 2020. Rammer: Enabling Holistic Deep Learning Compiler Optimizations with rTasks. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). USENIX Association, 881\u2013897. isbn:978-1-939133-19-9 https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/ma"},{"key":"e_1_3_2_1_24_1","volume-title":"Automation Test in Europe Conference Exhibition (DATE)","author":"Mao J.","year":"2017","unstructured":"J. Mao, X. Chen, K. W. Nixon, C. Krieger, and Y. Chen. 2017. MoDNN: Local distributed mobile computing system for Deep Neural Network. In Design, Automation Test in Europe Conference Exhibition (DATE), 2017. 1396\u20131401."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2897824.2925952"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC42613.2021.9365928"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","unstructured":"Samyam Rajbhandari Olatunji Ruwase Jeff Rasley Shaden Smith and Yuxiong He. 2021. ZeRO-Infinity: Breaking the GPU Memory Wall for Extreme Scale Deep Learning. https:\/\/doi.org\/10.48550\/ARXIV.2104.07857","DOI":"10.48550\/ARXIV.2104.07857"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00045"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Olaf Ronneberger Philipp Fischer and Thomas Brox. 2015. U-Net: Convolutional Networks for Biomedical Image Segmentation. arxiv:1505.04597.","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"e_1_3_2_1_32_1","unstructured":"Samsung. 2021. Exynos 2100. https:\/\/www.samsung.com\/semiconductor\/minisite\/exynos\/products\/mobileprocessor\/exynos-2100\/"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Mark Sandler Andrew Howard Menglong Zhu Andrey Zhmoginov and Liang-Chieh Chen. 2019. MobileNetV2: Inverted Residuals and Linear Bottlenecks. arxiv:1801.04381.","DOI":"10.1109\/CVPR.2018.00474"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2019. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. https:\/\/doi.org\/10.48550\/ARXIV.1909.08053","DOI":"10.48550\/ARXIV.1909.08053"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/SBAC-PAD53543.2021.00020"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jonathon Shlens and Zbigniew Wojna. 2015. Rethinking the Inception Architecture for Computer Vision. arxiv:1512.00567.","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063431"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Sanket Tavarageri Alexander Heinecke Sasikanth Avancha Gagandeep Goyal Ramakrishna Upadrasta and Bharat Kaul. 2020. PolyDL: Polyhedral Optimizations for Creation of High Performance DL primitives.","DOI":"10.1145\/3433103"},{"key":"e_1_3_2_1_39_1","volume-title":"Unity: Accelerating DNN Training Through Joint Optimization of Algebraic Transformations and Parallelization. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Unger Colin","year":"2022","unstructured":"Colin Unger, Zhihao Jia, Wei Wu, Sina Lin, Mandeep Baines, Carlos Efrain Quintero Narvaez, Vinay Ramakrishnaiah, Nirmal Prajapati, Pat McCormick, Jamaludin Mohd-Yusof, Xi Luo, Dheevatsa Mudigere, Jongsoo Park, Misha Smelyanskiy, and Alex Aiken. 2022. Unity: Accelerating DNN Training Through Joint Optimization of Algebraic Transformations and Parallelization. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). USENIX Association, Carlsbad, CA. 267\u2013284. isbn:978-1-939133-28-1 https:\/\/www.usenix.org\/conference\/osdi22\/presentation\/unger"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00089"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00382"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2105.04663"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2016.7783723"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3058532"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404397.3404473"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2018.2858384"},{"key":"e_1_3_2_1_47_1","volume-title":"Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, and Koushik Sen.","author":"Zheng Lianmin","year":"2020","unstructured":"Lianmin Zheng, Chengfan Jia, Minmin Sun, Zhao Wu, Cody Hao Yu, Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, and Koushik Sen. 2020. Ansor: Generating high-performance tensor programs for deep learning. In 14th $USENIX$ Symposium on Operating Systems Design and Implementation ($OSDI$ 20). 863\u2013879."},{"key":"e_1_3_2_1_48_1","volume-title":"Alpa: Automating Inter- and Intra-Operator Parallelism for Distributed Deep Learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng Lianmin","year":"2022","unstructured":"Lianmin Zheng, Zhuohan Li, Hao Zhang, Yonghao Zhuang, Zhifeng Chen, Yanping Huang, Yida Wang, Yuanzhong Xu, Danyang Zhuo, Eric P. Xing, Joseph E. Gonzalez, and Ion Stoica. 2022. Alpa: Automating Inter- and Intra-Operator Parallelism for Distributed Deep Learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). USENIX Association, Carlsbad, CA. 559\u2013578. isbn:978-1-939133-28-1 https:\/\/www.usenix.org\/conference\/osdi22\/presentation\/zheng-lianmin"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA53966.2022.00042"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3318216.3363312"},{"key":"e_1_3_2_1_51_1","volume-title":"ROLLER: Fast and Efficient Tensor Compilation for Deep Learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zhu Hongyu","year":"2022","unstructured":"Hongyu Zhu, Ruofan Wu, Yijia Diao, Shanbin Ke, Haoyu Li, Chen Zhang, Jilong Xue, Lingxiao Ma, Yuqing Xia, Wei Cui, Fan Yang, Mao Yang, Lidong Zhou, Asaf Cidon, and Gennady Pekhimenko. 2022. ROLLER: Fast and Efficient Tensor Compilation for Deep Learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22). USENIX Association, Carlsbad, CA. 233\u2013248. isbn:978-1-939133-28-1 https:\/\/www.usenix.org\/conference\/osdi22\/presentation\/zhu"}],"event":{"name":"CGO '23: 21st ACM\/IEEE International Symposium on Code Generation and Optimization","location":"Montr\u00e9al QC Canada","acronym":"CGO '23","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing","SIGPLAN ACM Special Interest Group on Programming Languages","IEEE-CS Computer Society"]},"container-title":["Proceedings of the 21st ACM\/IEEE International Symposium on Code Generation and Optimization"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3579990.3580015","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3579990.3580015","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:08:19Z","timestamp":1750183699000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3579990.3580015"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,17]]},"references-count":51,"alternative-id":["10.1145\/3579990.3580015","10.1145\/3579990"],"URL":"https:\/\/doi.org\/10.1145\/3579990.3580015","relation":{},"subject":[],"published":{"date-parts":[[2023,2,17]]},"assertion":[{"value":"2023-02-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}