{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:46:01Z","timestamp":1787017561745,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":91,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,11,12]],"date-time":"2023-11-12T00:00:00Z","timestamp":1699747200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,11,12]]},"DOI":"10.1145\/3625687.3625797","type":"proceedings-article","created":{"date-parts":[[2024,4,26]],"date-time":"2024-04-26T08:07:18Z","timestamp":1714118838000},"page":"125-137","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["nnPerf: Demystifying DNN Runtime Inference Latency on Mobile Platforms"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-7184-3150","authenticated-orcid":false,"given":"Haolin","family":"Chu","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7950-6773","authenticated-orcid":false,"given":"Xiaolong","family":"Zheng","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5040-2468","authenticated-orcid":false,"given":"Liang","family":"Liu","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7199-5047","authenticated-orcid":false,"given":"Huadong","family":"Ma","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,4,26]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Adreno GPU OpenCL Support. https:\/\/en.wikipedia.org\/wiki\/Adreno."},{"key":"e_1_3_2_1_2_1","unstructured":"Adreno GPU Profiler. https:\/\/developer.qualcomm.com\/forums\/software\/adreno-gpu-profiler."},{"key":"e_1_3_2_1_3_1","unstructured":"Adreno GPU Profiler by Qualcomm. https:\/\/developer.qualcomm.com\/forums\/software\/adreno-gpu-profiler?order=last_updated&sort=asc."},{"key":"e_1_3_2_1_4_1","unstructured":"AI creates new levels for Doom and Super Mario games. https:\/\/www.bbc.com\/news\/technology-44040007."},{"key":"e_1_3_2_1_5_1","unstructured":"Android Debug Bridge. https:\/\/en.wikipedia.org\/wiki\/Android_Debug_Bridge."},{"key":"e_1_3_2_1_6_1","unstructured":"Android GPU Inspector. https:\/\/gpuinspector.dev\/."},{"key":"e_1_3_2_1_7_1","unstructured":"Android Studio Profiler. https:\/\/developer.android.com\/studio\/profile\/android-profiler."},{"key":"e_1_3_2_1_8_1","unstructured":"Chrono. https:\/\/cplusplus.com\/reference\/chrono\/."},{"key":"e_1_3_2_1_9_1","unstructured":"Deep Learning for Games. https:\/\/developer.nvidia.com\/deep-learning-games."},{"key":"e_1_3_2_1_10_1","unstructured":"Efficientnet-b0 TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/tensorflow\/lite-model\/efficientnet\/lite0\/fp32\/2."},{"key":"e_1_3_2_1_11_1","unstructured":"Emmagee - a practical handy performance test tool for specified Android App. https:\/\/github.com\/NetEase\/Emmagee."},{"key":"e_1_3_2_1_12_1","unstructured":"Esrgan TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/captain-pool\/esrgan-tf2\/l."},{"key":"e_1_3_2_1_13_1","unstructured":"High Voltage Power Monitor. https:\/\/www.msoon.com\/online-store\/High-Voltage-Power-Monitor-p90002590."},{"key":"e_1_3_2_1_14_1","unstructured":"MACE. https:\/\/github.com\/XiaoMi\/mace."},{"key":"e_1_3_2_1_15_1","unstructured":"Mali GPU OpenCL Support. https:\/\/en.wikipedia.org\/wiki\/Mali_(GPU)."},{"key":"e_1_3_2_1_16_1","unstructured":"MNasNet_1.0_224 TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/tensorflow\/lite-model\/mnasnet_1.0_224\/1\/default\/1."},{"key":"e_1_3_2_1_17_1","unstructured":"MNN. https:\/\/github.com\/alibaba\/MNN."},{"key":"e_1_3_2_1_18_1","unstructured":"MobileBert 101 TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/iree\/lite-model\/mobilebert\/fp32\/l."},{"key":"e_1_3_2_1_19_1","unstructured":"MobileNetV3-small-100-224 TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/iree\/lite-model\/mobilenet_v3_small_100_224\/fp32\/1."},{"key":"e_1_3_2_1_20_1","unstructured":"ncnn. https:\/\/github.com\/Tencent\/ncnn."},{"key":"e_1_3_2_1_21_1","unstructured":"NVIDIA Nsight Compute. https:\/\/developer.nvidia.com\/nsight-compute."},{"key":"e_1_3_2_1_22_1","unstructured":"NVIDIA Nsight Systems. https:\/\/developer.nvidia.com\/nsight-systems."},{"key":"e_1_3_2_1_23_1","unstructured":"Out-of-order execution. https:\/\/en.wikipedia.org\/wiki\/Out-of-order_execution."},{"key":"e_1_3_2_1_24_1","unstructured":"paddle-lite. https:\/\/github.com\/PaddlePaddle\/Paddle-Lite."},{"key":"e_1_3_2_1_25_1","unstructured":"Perfetto. https:\/\/ui.perfetto.dev\/."},{"key":"e_1_3_2_1_26_1","unstructured":"Portable Computing Language (PoCL). http:\/\/portablecl.org\/."},{"key":"e_1_3_2_1_27_1","unstructured":"ResnetV2 101 TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/tensorflow\/lite-model\/resnet_v2_101\/1\/default\/1."},{"key":"e_1_3_2_1_28_1","unstructured":"Snapdragon Profiler. https:\/\/developer.qualcomm.com\/software\/snapdragon-profiler."},{"key":"e_1_3_2_1_29_1","unstructured":"SoloPi. https:\/\/github.com\/alipay\/SoloPi."},{"key":"e_1_3_2_1_30_1","unstructured":"Squeezenet TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/tensorflow\/lite-model\/squeezenet\/1\/default\/1."},{"key":"e_1_3_2_1_31_1","unstructured":"SSDMobileV2 TFLite model TensorFLow HUb. https:\/\/tfhub.dev\/iree\/lite-model\/ssd_mobilenet_v2_100\/fp32\/default\/1."},{"key":"e_1_3_2_1_32_1","unstructured":"Tencent Perfdog. https:\/\/perfdog.qq.com\/."},{"key":"e_1_3_2_1_33_1","unstructured":"Tensorflow Benchmark tools. https:\/\/www.tensorflow.org\/lite\/performance\/measurement."},{"key":"e_1_3_2_1_34_1","unstructured":"TensorFlow Lite. https:\/\/www.tensorflow.org\/lite\/."},{"key":"e_1_3_2_1_35_1","unstructured":"vDSO. https:\/\/man7.org\/linux\/man-pages\/man7\/vdso.7.html."},{"key":"e_1_3_2_1_36_1","unstructured":"XCode Instrument. https:\/\/developer.apple.com\/documentation\/xcode-release-notes\/xcode-14_1-release-notes."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3487552.3487863"},{"key":"e_1_3_2_1_38_1","volume-title":"Automatic kernel generation for volta tensor cores. arXiv preprint arXiv:2006.12645","author":"Bhaskaracharya S. G.","year":"2020","unstructured":"S. G. Bhaskaracharya, J. Demouth, and V. Grover. Automatic kernel generation for volta tensor cores. arXiv preprint arXiv:2006.12645, 2020."},{"key":"e_1_3_2_1_39_1","first-page":"622","volume-title":"Asian Conference on Machine Learning","author":"Cai E.","year":"2017","unstructured":"E. Cai, D.-C. Juan, D. Stamoulis, and D. Marculescu. Neuralpower: Predict and deploy energy-efficient convolutional neural networks. In Asian Conference on Machine Learning, pages 622--637. PMLR, 2017."},{"key":"e_1_3_2_1_40_1","volume-title":"Once-for-all: Train one network and specialize it for efficient deployment. arXiv preprint arXiv","author":"Cai H.","year":"1908","unstructured":"H. Cai, C. Gan, T. Wang, Z. Zhang, and S. Han. Once-for-all: Train one network and specialize it for efficient deployment. arXiv preprint arXiv: 1908.09791, 2019."},{"key":"e_1_3_2_1_41_1","volume-title":"Proxylessnas: Direct neural architecture search on target task and hardware. arXiv preprint arXiv:1812.00332","author":"Cai H.","year":"2018","unstructured":"H. Cai, L. Zhu, and S. Han. Proxylessnas: Direct neural architecture search on target task and hardware. arXiv preprint arXiv:1812.00332, 2018."},{"key":"e_1_3_2_1_42_1","first-page":"578","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen T.","year":"2018","unstructured":"T. Chen, T. Moreau, Z. Jiang, L. Zheng, E. Yan, H. Shen, M. Cowan, L. Wang, Y. Hu, L. Ceze, et al. {TVM}: An automated {End-to-End} optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18), pages 578--594, 2018."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01166"},{"key":"e_1_3_2_1_44_1","volume-title":"A survey of on-device machine learning: An algorithms and learning theory perspective. 2(3)","author":"Dhar S.","year":"2021","unstructured":"S. Dhar, J. Guo, J. J. Liu, S. Tripathi, U. Kurup, and M. Shah. A survey of on-device machine learning: An algorithms and learning theory perspective. 2(3), 2021."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2018.8486285"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.5555\/3322706.3361996"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3241539.3241559"},{"key":"e_1_3_2_1_48_1","first-page":"379","article-title":"Riptide: Fast end-to-end binarized neural networks","volume":"2","author":"Fromm J.","year":"2020","unstructured":"J. Fromm, M. Cowan, M. Philipose, L. Ceze, and S. Patel. Riptide: Fast end-to-end binarized neural networks. Proceedings of Machine Learning and Systems, 2:379--389, 2020.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_49_1","volume-title":"Towards latency-aware dnn optimization with gpu runtime analysis and tail effect elimination. arXiv preprint arXiv:2011.03897","author":"Fuxun Yu T. S. D. S.-L. S. D. W. R. M. C. Z. X. L. N. K. D. L. A. L. C. L. Y. C. X. C.","year":"2020","unstructured":"T. S. D. S.-L. S. D. W. R. M. C. Z. X. L. N. K. D. L. A. L. C. L. Y. C. X. C. Fuxun Yu, Zirui Xu. Towards latency-aware dnn optimization with gpu runtime analysis and tail effect elimination. arXiv preprint arXiv:2011.03897, 2020."},{"key":"e_1_3_2_1_50_1","first-page":"544","volume-title":"Proceedings, Part XVI 16","author":"Guo Z.","year":"2020","unstructured":"Z. Guo, X. Zhang, H. Mu, W. Heng, Z. Liu, Y. Wei, and J. Sun. Single path one-shot neural architecture search with uniform sampling. In Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XVI 16, pages 544--560. Springer, 2020."},{"key":"e_1_3_2_1_51_1","first-page":"539","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Han M.","year":"2022","unstructured":"M. Han, H. Zhang, R. Chen, and H. Chen. Microsecond-scale preemption for concurrent {GPU-accelerated}{DNN} inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 539--558, 2022."},{"key":"e_1_3_2_1_52_1","first-page":"539","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Han M.","year":"2022","unstructured":"M. Han, H. Zhang, R. Chen, and H. Chen. Microsecond-scale preemption for concurrent {GPU-accelerated}{DNN} inferences. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 539--558, 2022."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_48"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.155"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00140"},{"key":"e_1_3_2_1_56_1","volume-title":"The OpenCL Specification 2.0","author":"Howes L.","year":"2015","unstructured":"L. Howes and A. Munshi. The OpenCL Specification 2.0. Khronos OpenCL Working Group, 2015."},{"key":"e_1_3_2_1_57_1","volume-title":"Squeezenet: Alexnet-level accuracy with 50x fewer parameters and < 0.5 mb model size. arXiv preprint arXiv:1602.07360","author":"Iandola F. N.","year":"2016","unstructured":"F. N. Iandola, S. Han, M. W. Moskewicz, K. Ashraf, W. J. Dally, and K. Keutzer. Squeezenet: Alexnet-level accuracy with 50x fewer parameters and < 0.5 mb model size. arXiv preprint arXiv:1602.07360, 2016."},{"key":"e_1_3_2_1_58_1","volume-title":"On-device training under 256kb memory. arXiv preprint arXiv:2206.15472","author":"Ji Lin W.-M. C. W.-C. W. C. G. S. H.","year":"2022","unstructured":"W.-M. C. W.-C. W. C. G. S. H. Ji Lin, Ligeng Zhu. On-device training under 256kb memory. arXiv preprint arXiv:2206.15472, 2022."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359630"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330648"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453483.3454038"},{"key":"e_1_3_2_1_62_1","first-page":"286","volume-title":"Stabilizing cpu frequency and voltage for temperature-aware dvfs in mobile devices","author":"Kim J. M.","year":"2015","unstructured":"J. M. Kim, Y. G. Kim, and S. W. Chung. Stabilizing cpu frequency and voltage for temperature-aware dvfs in mobile devices. volume 64, pages 286--292, 2015."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3529706.3529714"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"crossref","first-page":"25","DOI":"10.1145\/3469116.3470014","volume-title":"Proceedings of the 5th International Workshop on Embedded and Mobile Deep Learning","author":"Lee J.","year":"2021","unstructured":"J. Lee, Y. Liu, and Y. Lee. Parallelfusion: towards maximum utilization of mobile gpu for dnn inference. In Proceedings of the 5th International Workshop on Embedded and Mobile Deep Learning, pages 25--30, 2021."},{"key":"e_1_3_2_1_65_1","volume-title":"The 28th Annual International Conference On Mobile Computing And Networking (MobiCom 2022","author":"Liang R.","year":"2022","unstructured":"R. Liang, T. Cao, J. Wen, M. Wang, Y. Wang, J. Zou, and Y. Liu. Romou: Rapidly generate high-performance tensor kernels for mobile gpus. In The 28th Annual International Conference On Mobile Computing And Networking (MobiCom 2022). ACM, February 2022."},{"key":"e_1_3_2_1_66_1","first-page":"11711","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Lin J.","year":"2020","unstructured":"J. Lin, W.-M. Chen, Y. Lin, j. cohn, C. Gan, and S. Han. Mcunet: Tiny deep learning on iot devices. In H. Larochelle, M. Ranzato, R. Hadsell, M. Balcan, and H. Lin, editors, Advances in Neural Information Processing Systems, volume 33, pages 11711--11722. Curran Associates, Inc., 2020."},{"key":"e_1_3_2_1_67_1","volume-title":"Darts: Differentiable architecture search. arXiv preprint arXiv:1806.09055","author":"Liu H.","year":"2018","unstructured":"H. Liu, K. Simonyan, and Y. Yang. Darts: Differentiable architecture search. arXiv preprint arXiv:1806.09055, 2018."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00339"},{"key":"e_1_3_2_1_69_1","first-page":"881","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Ma L.","year":"2020","unstructured":"L. Ma, Z. Xie, Z. Yang, J. Xue, Y. Miao, W. Cui, W. Hu, F. Yang, L. Zhang, and L. Zhou. Rammer: Enabling holistic deep learning compiler optimizations with {rTasks}. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20), pages 881--897, 2020."},{"key":"e_1_3_2_1_70_1","first-page":"313","volume-title":"Linux Symposium","volume":"1","author":"Mochel P.","year":"2005","unstructured":"P. Mochel. The sysfs filesystem. In Linux Symposium, volume 1, pages 313--326. The Linux Foundation San Francisco, CA, USA, 2005."},{"key":"e_1_3_2_1_71_1","volume-title":"Pruning convolutional neural networks for resource efficient inference. arXiv preprint arXiv:1611.06440","author":"Molchanov P.","year":"2016","unstructured":"P. Molchanov, S. Tyree, T. Karras, T. Aila, and J. Kautz. Pruning convolutional neural networks for resource efficient inference. arXiv preprint arXiv:1611.06440, 2016."},{"key":"e_1_3_2_1_72_1","volume-title":"International Conference on Learning Representations.","author":"Qi H.","unstructured":"H. Qi, E. R. Sparks, and A. Talwalkar. Paleo: A performance model for deep neural networks. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_73_1","volume-title":"Qualcomm Snapdragon Mobile Platform OpenCL General Programming and Optimization","year":"2021","unstructured":"Qualcomm. Qualcomm Snapdragon Mobile Platform OpenCL General Programming and Optimization. Khronos OpenCL Working Group, 2021."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014780"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"crossref","first-page":"396","DOI":"10.1109\/ICASSP.2019.8682582","volume-title":"ICASSP 2019--2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Sharma B.","year":"2019","unstructured":"B. Sharma, C. Gupta, H. Li, and Y. Wang. Automatic lyrics-to-audio alignment on polyphonic music using singing-adapted acoustic models. In ICASSP 2019--2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 396--400. IEEE, 2019."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00293"},{"key":"e_1_3_2_1_77_1","first-page":"6105","volume-title":"International conference on machine learning","author":"Tan M.","year":"2019","unstructured":"M. Tan and Q. Le. Efficientnet: Rethinking model scaling for convolutional neural networks. In International conference on machine learning, pages 6105--6114. PMLR, 2019."},{"key":"e_1_3_2_1_78_1","first-page":"21","article-title":"To bridge neural network design and real-world performance: A behaviour study for neural networks","volume":"3","author":"Tang X.","year":"2021","unstructured":"X. Tang, S. Han, L. L. Zhang, T. Cao, and Y. Liu. To bridge neural network design and real-world performance: A behaviour study for neural networks. Proceedings of Machine Learning and Systems, 3:21--37, 2021.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_79_1","first-page":"37","volume-title":"15th USENIX Symposium on Operating Systems Design and Implementation (OSDI 21)","author":"Wang H.","year":"2021","unstructured":"H. Wang, J. Zhai, M. Gao, Z. Ma, S. Tang, L. Zheng, Y. Li, K. Rong, Y. Chen, and Z. Jia. PET: Optimizing tensor programs with partially equivalent transformations and automated corrections. In 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI 21), pages 37--54. USENIX Association, July 2021."},{"key":"e_1_3_2_1_80_1","volume-title":"Non-structured dnn weight pruning considered harmful. arXiv preprint arXiv:1907.02124, 2","author":"Wang Y.","year":"2019","unstructured":"Y. Wang, S. Ye, Z. He, X. Ma, L. Zhang, S. Lin, G. Yuan, S. H. Tan, Z. Li, D. Fan, et al. Non-structured dnn weight pruning considered harmful. arXiv preprint arXiv:1907.02124, 2, 2019."},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3241539.3241563"},{"key":"e_1_3_2_1_82_1","first-page":"1","volume-title":"CCF Transactions on Pervasive Computing and Interaction","author":"Yang S.","year":"2023","unstructured":"S. Yang, Y. He, and Y. Chen. Spatialgaze: towards spatial gaze tracking for extended reality. CCF Transactions on Pervasive Computing and Interaction, pages 1--17, 2023."},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01249-6_18"},{"key":"e_1_3_2_1_84_1","volume-title":"Towards latency-aware dnn optimization with gpu runtime analysis and tail effect elimination. arXiv preprint arXiv:2011.03897","author":"Yu F.","year":"2020","unstructured":"F. Yu, Z. Xu, T. Shen, D. Stamoulis, L. Shangguan, D. Wang, R. Madhok, C. Zhao, X. Li, N. Karianakis, et al. Towards latency-aware dnn optimization with gpu runtime analysis and tail effect elimination. arXiv preprint arXiv:2011.03897, 2020."},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458864.3467882"},{"key":"e_1_3_2_1_86_1","first-page":"7543","volume-title":"International conference on machine learning","author":"Zhao R.","year":"2019","unstructured":"R. Zhao, Y. Hu, J. Dotzel, C. De Sa, and Z. Zhang. Improving neural network quantization without retraining using outlier channel splitting. In International conference on machine learning, pages 7543--7552. PMLR, 2019."},{"key":"e_1_3_2_1_87_1","first-page":"863","volume-title":"14th USENIX symposium on operating systems design and implementation (OSDI 20)","author":"Zheng L.","year":"2020","unstructured":"L. Zheng, C. Jia, M. Sun, Z. Wu, C. H. Yu, A. Haj-Ali, Y. Wang, J. Yang, D. Zhuo, K. Sen, et al. Ansor: Generating {High-Performance} tensor programs for deep learning. In 14th USENIX symposium on operating systems design and implementation (OSDI 20), pages 863--879, 2020."},{"key":"e_1_3_2_1_88_1","first-page":"559","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng L.","year":"2022","unstructured":"L. Zheng, Z. Li, H. Zhang, Y. Zhuang, Z. Chen, Y. Huang, Y. Wang, Y. Xu, D. Zhuo, E. P. Xing, et al. Alpa: Automating inter-and {Intra-Operator} parallelism for distributed deep learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 559--578, 2022."},{"key":"e_1_3_2_1_89_1","first-page":"213","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng N.","year":"2022","unstructured":"N. Zheng, B. Lin, Q. Zhang, L. Ma, Y. Yang, F. Yang, Y. Wang, M. Yang, and L. Zhou. {SparTA}:{Deep-Learning} model sparsity via {Tensor-with-Sparsity-Attribute}. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 213--232, 2022."},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11623"},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00907"}],"event":{"name":"SenSys '23: 21st ACM Conference on Embedded Networked Sensor Systems","location":"Istanbul Turkiye","acronym":"SenSys '23","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems","SIGMETRICS ACM Special Interest Group on Measurement and Evaluation","SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing","SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 21st ACM Conference on Embedded Networked Sensor Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3625687.3625797","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3625687.3625797","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T13:49:11Z","timestamp":1750168151000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3625687.3625797"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,12]]},"references-count":91,"alternative-id":["10.1145\/3625687.3625797","10.1145\/3625687"],"URL":"https:\/\/doi.org\/10.1145\/3625687.3625797","relation":{},"subject":[],"published":{"date-parts":[[2023,11,12]]},"assertion":[{"value":"2024-04-26","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}