{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,28]],"date-time":"2026-04-28T04:10:14Z","timestamp":1777349414868,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,12]]},"DOI":"10.1145\/3673038.3673124","type":"proceedings-article","created":{"date-parts":[[2024,8,8]],"date-time":"2024-08-08T18:29:01Z","timestamp":1723141741000},"page":"367-376","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Arlo: Serving Transformer-based Language Models with Dynamic Input Lengths"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3785-9700","authenticated-orcid":false,"given":"Xin","family":"Tan","sequence":"first","affiliation":[{"name":"The Chinese University of Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8110-2436","authenticated-orcid":false,"given":"Jiamin","family":"Li","sequence":"additional","affiliation":[{"name":"Microsoft, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1332-7342","authenticated-orcid":false,"given":"Yitao","family":"Yang","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2519-2550","authenticated-orcid":false,"given":"Jingzong","family":"Li","sequence":"additional","affiliation":[{"name":"The Hang Seng University of Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9359-9571","authenticated-orcid":false,"given":"Hong","family":"Xu","sequence":"additional","affiliation":[{"name":"The Chinese University of Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,8,12]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Amazon autosacling. https:\/\/docs.aws.amazon.com\/autoscaling\/."},{"key":"e_1_3_2_1_2_1","unstructured":"ChatGPT. https:\/\/chat.openai.com\/."},{"key":"e_1_3_2_1_3_1","unstructured":"Dolly. https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm."},{"key":"e_1_3_2_1_4_1","unstructured":"FasterTransformer. https:\/\/github.com\/NVIDIA\/FasterTransformer."},{"key":"e_1_3_2_1_5_1","unstructured":"Gurobi Optimization. https:\/\/www.gurobi.com\/."},{"key":"e_1_3_2_1_6_1","unstructured":"HuggingFace. https:\/\/huggingface.co\/."},{"key":"e_1_3_2_1_7_1","unstructured":"HuggingfaceTokenizers. https:\/\/github.com\/huggingface\/tokenizers."},{"key":"e_1_3_2_1_8_1","unstructured":"TensorRT. https:\/\/developer.nvidia.com\/tensorrt."},{"key":"e_1_3_2_1_9_1","unstructured":"Triton inference server. https:\/\/github.com\/triton-inference-server\/server."},{"key":"e_1_3_2_1_10_1","unstructured":"TVM Unity. https:\/\/github.com\/apache\/tvm\/tree\/unity."},{"key":"e_1_3_2_1_11_1","unstructured":"Twitter streaming traces. https:\/\/archive.org\/details\/twitterstream."},{"key":"e_1_3_2_1_12_1","unstructured":"XLA: Accelerated Linear Algebra. https:\/\/github.com\/openxla\/xla."},{"key":"e_1_3_2_1_13_1","volume-title":"Proc. USENIX OSDI.","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen 2018. TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In Proc. USENIX OSDI."},{"key":"e_1_3_2_1_14_1","volume-title":"Proc. USENIX ATC.","author":"Choi Seungbeom","year":"2022","unstructured":"Seungbeom Choi 2022. Serving heterogeneous machine learning models on Multi-GPU servers with Spatio-Temporal sharing. In Proc. USENIX ATC."},{"key":"e_1_3_2_1_15_1","volume-title":"Proc. USENIX NSDI.","author":"Crankshaw Daniel","year":"2017","unstructured":"Daniel Crankshaw 2017. Clipper: A Low-Latency Online Prediction Serving System.. In Proc. USENIX NSDI."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421285"},{"key":"e_1_3_2_1_17_1","volume-title":"Proc. NAACL-HLT.","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proc. NAACL-HLT."},{"key":"e_1_3_2_1_18_1","volume-title":"The Efficiency Spectrum of Large Language Models: An Algorithmic Survey. arXiv preprint arXiv:2312.00678","author":"Tianyu Ding","year":"2023","unstructured":"Tianyu Ding 2023. The Efficiency Spectrum of Large Language Models: An Algorithmic Survey. arXiv preprint arXiv:2312.00678 (2023)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462990"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441578"},{"key":"e_1_3_2_1_21_1","volume-title":"Serving dnns like clockwork: Performance predictability from the bottom up. arXiv preprint arXiv:2006.02464","author":"Arpan Gujarati","year":"2020","unstructured":"Arpan Gujarati 2020. Serving dnns like clockwork: Performance predictability from the bottom up. arXiv preprint arXiv:2006.02464 (2020)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proc. USENIX NSDI.","author":"Gunasekaran Jashwant\u00a0Raj","year":"2022","unstructured":"Jashwant\u00a0Raj Gunasekaran 2022. Cocktail: A Multidimensional Optimization for Model Serving in Cloud. In Proc. USENIX NSDI."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3412747"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651335"},{"key":"e_1_3_2_1_27_1","volume-title":"High-Performance ML Serving. In Workshop on ML Systems at NIPS.","author":"Olston Christopher","year":"2017","unstructured":"Christopher Olston 2017. TensorFlow-Serving: Flexible, High-Performance ML Serving. In Workshop on ML Systems at NIPS."},{"key":"e_1_3_2_1_28_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_29_1","volume-title":"Exploring the limits of transfer learning with a unified text-to-text transformer. The Journal of Machine Learning Research","author":"Colin Raffel","year":"2020","unstructured":"Colin Raffel 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. The Journal of Machine Learning Research (2020)."},{"key":"e_1_3_2_1_30_1","volume-title":"Proc. USENIX ATC.","author":"Romero Francisco","year":"2021","unstructured":"Francisco Romero 2021. INFaaS: Automated model-less inference serving. In Proc. USENIX ATC."},{"key":"e_1_3_2_1_31_1","volume-title":"Glow: Graph lowering compiler techniques for neural networks. arXiv preprint arXiv:1805.00907","author":"Nadav Rotem","year":"2018","unstructured":"Nadav Rotem 2018. Glow: Graph lowering compiler techniques for neural networks. arXiv preprint arXiv:1805.00907 (2018)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359658"},{"key":"e_1_3_2_1_33_1","volume-title":"Proc. Machine Learning and Systems.","author":"Shen Haichen","year":"2021","unstructured":"Haichen Shen 2021. Nimble: Efficiently compiling dynamic neural networks for model inference. In Proc. Machine Learning and Systems."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigMM.2019.00-44"},{"key":"e_1_3_2_1_35_1","volume-title":"Serving DNN models with multi-instance gpus: A case of the reconfigurable machine scheduling problem. arXiv preprint arXiv:2109.11067","author":"Cheng Tan","year":"2021","unstructured":"Cheng Tan 2021. Serving DNN models with multi-instance gpus: A case of the reconfigurable machine scheduling problem. arXiv preprint arXiv:2109.11067 (2021)."},{"key":"e_1_3_2_1_36_1","volume-title":"Proc. ACM NIPS.","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani 2017. Attention is all you need. In Proc. ACM NIPS."},{"key":"e_1_3_2_1_37_1","volume-title":"Proc. Machine Learning and Systems.","author":"Wang Guanhua","year":"2021","unstructured":"Guanhua Wang 2021. sensai: Convnets decomposition via class parallelism for fast inference on live data. In Proc. Machine Learning and Systems."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3587438"},{"key":"e_1_3_2_1_39_1","volume-title":"Multi-passage bert: A globally normalized bert model for open-domain question answering. arXiv preprint arXiv:1908.08167","author":"Zhiguo Wang","year":"2019","unstructured":"Zhiguo Wang 2019. Multi-passage bert: A globally normalized bert model for open-domain question answering. arXiv preprint arXiv:1908.08167 (2019)."},{"key":"e_1_3_2_1_40_1","volume-title":"Fast Distributed Inference Serving for Large Language Models. arXiv preprint arXiv:2305.05920","author":"Bingyang Wu","year":"2023","unstructured":"Bingyang Wu 2023. Fast Distributed Inference Serving for Large Language Models. arXiv preprint arXiv:2305.05920 (2023)."},{"key":"e_1_3_2_1_41_1","volume-title":"igniter: Interference-aware gpu resource provisioning for predictable dnn inference in the cloud","author":"Fei Xu","year":"2022","unstructured":"Fei Xu 2022. igniter: Interference-aware gpu resource provisioning for predictable dnn inference in the cloud. IEEE TPDS (2022)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539090"},{"key":"e_1_3_2_1_43_1","volume-title":"Proc. USENIX OSDI.","author":"Yu Gyeong-In","year":"2022","unstructured":"Gyeong-In Yu 2022. Orca: A distributed serving system for Transformer-Based generative models. In Proc. USENIX OSDI."},{"key":"e_1_3_2_1_44_1","volume-title":"Proc. USENIX NSDI.","author":"Zaharia Matei","year":"2012","unstructured":"Matei Zaharia 2012. Resilient Distributed Datasets: A Fault-Tolerant Abstraction for in-Memory Cluster Computing. In Proc. USENIX NSDI."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS54959.2023.00042"},{"key":"e_1_3_2_1_46_1","volume-title":"Proc. USENIX ATC.","author":"Zhang Chengliang","year":"2019","unstructured":"Chengliang Zhang 2019. MArk: Exploiting Cloud Services for Cost-Effective, SLO-Aware Machine Learning Inference Serving.. In Proc. USENIX ATC."},{"key":"e_1_3_2_1_47_1","volume-title":"Proc. USENIX NSDI.","author":"Zhang Hong","year":"2023","unstructured":"Hong Zhang 2023. SHEPHERD: Serving DNNs in the Wild. In Proc. USENIX NSDI."},{"key":"e_1_3_2_1_48_1","volume-title":"Proc. USENIX OSDI.","author":"Zheng Ningxin","year":"2022","unstructured":"Ningxin Zheng 2022. { SparTA} :{ Deep-Learning} Model Sparsity via { Tensor-with-Sparsity-Attribute}. In Proc. USENIX OSDI."},{"key":"e_1_3_2_1_49_1","volume-title":"SparDA: Accelerating Dynamic Sparse Deep Neural Networks via Sparse-Dense Transformation. arXiv preprint arXiv:2301.10936","author":"Ningxin Zheng","year":"2023","unstructured":"Ningxin Zheng 2023. SparDA: Accelerating Dynamic Sparse Deep Neural Networks via Sparse-Dense Transformation. arXiv preprint arXiv:2301.10936 (2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. arXiv preprint arXiv:2401.09670","author":"Yinmin Zhong","year":"2024","unstructured":"Yinmin Zhong 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. arXiv preprint arXiv:2401.09670 (2024)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437984.3458838"},{"key":"e_1_3_2_1_52_1","volume-title":"Proc. Chinese National Conference on Computational Linguistics.","author":"Zhuang Liu","year":"2021","unstructured":"Liu Zhuang 2021. A Robustly Optimized BERT Pre-training Approach with Post-training. In Proc. Chinese National Conference on Computational Linguistics."}],"event":{"name":"ICPP '24: the 53rd International Conference on Parallel Processing","location":"Gotland Sweden","acronym":"ICPP '24"},"container-title":["Proceedings of the 53rd International Conference on Parallel Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673124","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3673038.3673124","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T17:29:19Z","timestamp":1758648559000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673038.3673124"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,12]]},"references-count":52,"alternative-id":["10.1145\/3673038.3673124","10.1145\/3673038"],"URL":"https:\/\/doi.org\/10.1145\/3673038.3673124","relation":{},"subject":[],"published":{"date-parts":[[2024,8,12]]},"assertion":[{"value":"2024-08-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}