{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T18:16:20Z","timestamp":1782152180490,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,22]],"date-time":"2024-04-22T00:00:00Z","timestamp":1713744000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NSF CNS","award":["1816887"],"award-info":[{"award-number":["1816887"]}]},{"name":"NSF CCF","award":["1763747"],"award-info":[{"award-number":["1763747"]}]},{"name":"NSF IIS","award":["1833137"],"award-info":[{"award-number":["1833137"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,22]]},"DOI":"10.1145\/3642968.3654818","type":"proceedings-article","created":{"date-parts":[[2024,4,17]],"date-time":"2024-04-17T00:04:51Z","timestamp":1713312291000},"page":"31-36","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["A Benchmark for ML Inference Latency on Mobile Devices"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8308-0231","authenticated-orcid":false,"given":"Zhuojin","family":"Li","sequence":"first","affiliation":[{"name":"University of Southern California, Los Angeles, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5110-203X","authenticated-orcid":false,"given":"Marco","family":"Paolieri","sequence":"additional","affiliation":[{"name":"University of Southern California, Los Angeles, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8353-5040","authenticated-orcid":false,"given":"Leana","family":"Golubchik","sequence":"additional","affiliation":[{"name":"University of Southern California, Los Angeles, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,4,22]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2021. Sandbox for training deep learning networks. https:\/\/github.com\/osmr\/imgclsmob."},{"key":"e_1_3_2_1_2_1","unstructured":"2024. A Benchmark for ML Inference Latency on Mobile Devices. https:\/\/github.com\/qed-usc\/mobile-ml-benchmark.git."},{"key":"e_1_3_2_1_3_1","unstructured":"Apple. 2024. Optimizing GPU performance. https:\/\/developer.apple.com\/documentation\/xcode\/optimizing-gpu-performance."},{"key":"e_1_3_2_1_4_1","volume-title":"International Conference on Learning Representations.","author":"Cai Han","year":"2020","unstructured":"Han Cai, Chuang Gan, Tianzhe Wang, Zhekai Zhang, and Song Han. 2020. Once for All: Train One Network and Specialize it for Efficient Deployment. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_5_1","volume-title":"Nats-bench: Benchmarking nas algorithms for architecture topology and size","author":"Dong Xuanyi","year":"2021","unstructured":"Xuanyi Dong, Lu Liu, Katarzyna Musial, and Bogdan Gabrys. 2021. Nats-bench: Benchmarking nas algorithms for architecture topology and size. IEEE transactions on pattern analysis and machine intelligence 44, 7 (2021), 3634--3646."},{"key":"e_1_3_2_1_6_1","first-page":"10480","article-title":"BRP-NAS: Prediction-based NAS using GCNs","volume":"33","author":"Dudziak Lukasz","year":"2020","unstructured":"Lukasz Dudziak, Thomas Chau, Mohamed Abdelfattah, Royson Lee, Hyeji Kim, and Nicholas Lane. 2020. BRP-NAS: Prediction-based NAS using GCNs. In Advances in Neural Information Processing Systems, Vol. 33. 10480--10490.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_7_1","unstructured":"Google. 2022. TFLite Model Benchmark Tool. https:\/\/github.com\/tensorflow\/tensorflow\/tree\/master\/tensorflow\/lite\/tools\/benchmark."},{"key":"e_1_3_2_1_8_1","volume-title":"XNNPACK: High-efficiency floating-point neural network inference operators for mobile, server, and Web. https:\/\/github.com\/google\/XNNPACK.","year":"2023","unstructured":"Google. 2023. XNNPACK: High-efficiency floating-point neural network inference operators for mobile, server, and Web. https:\/\/github.com\/google\/XNNPACK."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2019.8737614"},{"key":"e_1_3_2_1_10_1","first-page":"352","article-title":"MLPerf mobile inference benchmark: An industry-standard open-source machine learning benchmark for on-device AI","volume":"4","author":"Reddi Vijay Janapa","year":"2022","unstructured":"Vijay Janapa Reddi, David Kanter, Peter Mattson, Jared Duke, Thai Nguyen, Ramesh Chukka, Ken Shiring, Koan-Sin Tan, Mark Charlebois, William Chou, et al. 2022. MLPerf mobile inference benchmark: An industry-standard open-source machine learning benchmark for on-device AI. Proceedings of Machine Learning and Systems 4 (2022), 352--369.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3093337.3037698"},{"key":"e_1_3_2_1_12_1","volume-title":"On-Device Neural Net Inference with Mobile GPUs. arXiv preprint arXiv:1907.01989","author":"Lee Juhyun","year":"2019","unstructured":"Juhyun Lee, Nikolay Chirkov, Ekaterina Ignasheva, Yury Pisarchyk, Mogan Shieh, Fabio Riccardi, Raman Sarokin, Andrei Kulik, and Matthias Grundmann. 2019. On-Device Neural Net Inference with Mobile GPUs. arXiv preprint arXiv:1907.01989 (2019)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578244.3583735"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"e_1_3_2_1_16_1","volume-title":"International Conference on Learning Representations.","author":"Mehta Sachin","year":"2021","unstructured":"Sachin Mehta and Mohammad Rastegari. 2021. MobileViT: Light-weight, Generalpurpose, and Mobile-friendly Vision Transformer. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_17_1","volume-title":"QNNPACK: Quantized Neural Networks PACKage. https:\/\/github.com\/pytorch\/pytorch\/tree\/main\/aten\/src\/ATen\/native\/quantized\/cpu\/qnnpack.","year":"2023","unstructured":"Meta. 2023. QNNPACK: Quantized Neural Networks PACKage. https:\/\/github.com\/pytorch\/pytorch\/tree\/main\/aten\/src\/ATen\/native\/quantized\/cpu\/qnnpack."},{"key":"e_1_3_2_1_18_1","unstructured":"PyTorch. 2023. PyTorch CPU Caching Allocator. https:\/\/github.com\/pytorch\/pytorch\/blob\/main\/c10\/mobile\/CPUCachingAllocator.h."},{"key":"e_1_3_2_1_19_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2019.2944584"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the 2020 conference on empirical methods in natural language processing: system demonstrations. 38--45","author":"Thomas","unstructured":"Thomas Wolf et al. 2020. Transformers: State-of-the-art natural language processing. In Proceedings of the 2020 conference on empirical methods in natural language processing: system demonstrations. 38--45."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01249-6_18"},{"key":"e_1_3_2_1_23_1","volume-title":"Neural architecture search with reinforcement learning. arXiv preprint arXiv:1611.01578","author":"Zoph Barret","year":"2016","unstructured":"Barret Zoph and Quoc V Le. 2016. Neural architecture search with reinforcement learning. arXiv preprint arXiv:1611.01578 (2016)."}],"event":{"name":"EuroSys '24: Nineteenth European Conference on Computer Systems","location":"Athens Greece","acronym":"EuroSys '24","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 7th International Workshop on Edge Systems, Analytics and Networking"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3642968.3654818","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3642968.3654818","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T18:43:20Z","timestamp":1755974600000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3642968.3654818"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,22]]},"references-count":23,"alternative-id":["10.1145\/3642968.3654818","10.1145\/3642968"],"URL":"https:\/\/doi.org\/10.1145\/3642968.3654818","relation":{},"subject":[],"published":{"date-parts":[[2024,4,22]]},"assertion":[{"value":"2024-04-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}