{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T16:26:34Z","timestamp":1754151994532,"version":"3.41.2"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","funder":[{"name":"European High-Performance Computing Joint Undertaking (JU)","award":["956702"],"award-info":[{"award-number":["956702"]}]},{"name":"Swedish Research Council","award":["020-06735_3"],"award-info":[{"award-number":["020-06735_3"]}]},{"name":"Swedish Foundation for Strategic Research","award":["CHI19-0048"],"award-info":[{"award-number":["CHI19-0048"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,28]]},"DOI":"10.1145\/3719276.3725190","type":"proceedings-article","created":{"date-parts":[[2025,7,4]],"date-time":"2025-07-04T05:00:46Z","timestamp":1751605246000},"page":"159-167","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Accordion: A malleable pipeline scheduling approach for adaptive SLO-aware inference serving"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8654-3249","authenticated-orcid":false,"given":"Pirah Noor","family":"Soomro","sequence":"first","affiliation":[{"name":"Chalmers University of Technology and University of Gothenburg, Gothenburg, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2141-5654","authenticated-orcid":false,"given":"Nikela","family":"Papadopoulou","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, Scotland UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7583-6609","authenticated-orcid":false,"given":"Miquel","family":"Peric\u00e0s","sequence":"additional","affiliation":[{"name":"Chalmers University of Technology and University of Gothenburg, Gothenburg, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,4]]},"reference":[{"unstructured":"[n. d.]. Amazon SageMaker. https:\/\/aws.amazon.com\/sagemaker\/. Accessed: 2024.","key":"e_1_3_3_1_2_2"},{"unstructured":"[n. d.]. Keras Documentation. https:\/\/keras.io\/. Accessed: 2024.","key":"e_1_3_3_1_3_2"},{"unstructured":"[n. d.]. NVIDIA Triton Inference Server Documentation. https:\/\/docs.nvidia.com\/deeplearning\/triton-inference-server\/. Accessed: 2024.","key":"e_1_3_3_1_4_2"},{"volume-title":"Twitter Streaming Traces","year":"2018","unstructured":"2018. Twitter Streaming Traces. https:\/\/archive.org\/details\/archiveteam-twitter-stream-2018-04","key":"e_1_3_3_1_5_2"},{"unstructured":"2024. Azure Machine Learning. https:\/\/docs.microsoft.com\/en-us\/azure\/machine-learning\/.","key":"e_1_3_3_1_6_2"},{"unstructured":"2024. Google Cloud AI Platform. https:\/\/cloud.google.com\/ai-platform\/.","key":"e_1_3_3_1_7_2"},{"doi-asserted-by":"crossref","unstructured":"Sohaib Ahmad Hui Guan Brian\u00a0D Friedman Thomas Williams Ramesh\u00a0K Sitaraman and Thomas Woo. 2024. Proteus: A High-Throughput Inference-Serving System with Accuracy Scaling. (2024).","key":"e_1_3_3_1_8_2","DOI":"10.1145\/3617232.3624849"},{"unstructured":"Sherif Akoush Andrei Paleyes Arnaud Van\u00a0Looveren and Clive Cox. 2022. Desiderata for next generation of ML model serving. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.14665 (2022).","key":"e_1_3_3_1_9_2"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_10_2","DOI":"10.1109\/SC41405.2020.00073"},{"doi-asserted-by":"crossref","unstructured":"Ahsan Ali Riccardo Pinciroli Feng Yan and Evgenia Smirni. 2022. Optimizing inference serving on serverless platforms. Proceedings of the VLDB Endowment 15 10 (2022).","key":"e_1_3_3_1_11_2","DOI":"10.14778\/3547305.3547313"},{"doi-asserted-by":"crossref","unstructured":"Tal Ben-Nun and Torsten Hoefler. 2019. Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. ACM Computing Surveys (CSUR) 52 4 (2019) 1\u201343.","key":"e_1_3_3_1_12_2","DOI":"10.1145\/3320060"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_13_2","DOI":"10.1145\/3037697.3037700"},{"doi-asserted-by":"crossref","unstructured":"Quan Chen Hailong Yang Jason Mars and Lingjia Tang. 2016. Baymax: Qos awareness and increased utilization for non-preemptive accelerators in warehouse scale computers. ACM SIGPLAN Notices 51 4 (2016) 681\u2013696.","key":"e_1_3_3_1_14_2","DOI":"10.1145\/2954679.2872368"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_15_2","DOI":"10.1109\/HPCA51647.2021.00049"},{"key":"e_1_3_3_1_16_2","first-page":"613","volume-title":"14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17)","author":"Crankshaw Daniel","year":"2017","unstructured":"Daniel Crankshaw, Xin Wang, Guilio Zhou, Michael\u00a0J Franklin, Joseph\u00a0E Gonzalez, and Ion Stoica. 2017. Clipper: A { Low-Latency} online prediction serving system. In 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17). 613\u2013627."},{"doi-asserted-by":"crossref","unstructured":"Christina Delimitrou and Christos Kozyrakis. 2014. Quasar: Resource-efficient and QoS-aware cluster management. ACM Sigplan Notices 49 4 (2014) 127\u2013144.","key":"e_1_3_3_1_17_2","DOI":"10.1145\/2644865.2541941"},{"unstructured":"Facebook Engineering. [n. d.]. Building Meta\u2019s GenAI Infrastructure. https:\/\/engineering.fb.com\/2024\/03\/12\/data-center-engineering\/building-metas-genai-infrastructure\/. Accessed on April 22 2024.","key":"e_1_3_3_1_18_2"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_19_2","DOI":"10.1109\/ISCA45697.2020.00084"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_20_2","DOI":"10.1109\/HPCA47549.2020.00047"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_21_2","DOI":"10.1109\/HPCA.2018.00059"},{"unstructured":"Zhihao Jia Matei Zaharia and Alex Aiken. 2019. Beyond Data and Model Parallelism for Deep Neural Networks. Proceedings of Machine Learning and Systems 1 (2019) 1\u201313.","key":"e_1_3_3_1_22_2"},{"doi-asserted-by":"crossref","unstructured":"Fei Jiang Yong Jiang Hui Zhi Yi Dong Hao Li Sufeng Ma Yilong Wang Qiang Dong Haipeng Shen and Yongjun Wang. 2017. Artificial intelligence in healthcare: past present and future. Stroke and vascular neurology 2 4 (2017).","key":"e_1_3_3_1_23_2","DOI":"10.1136\/svn-2017-000101"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_24_2","DOI":"10.1145\/3230543.3230574"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_25_2","DOI":"10.1145\/3079856.3080246"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_26_2","DOI":"10.1109\/HPCA53966.2022.00019"},{"doi-asserted-by":"crossref","unstructured":"I-Ting\u00a0Angelina Lee Charles\u00a0E Leiserson Tao\u00a0B Schardl Zhunping Zhang and Jim Sukha. 2015. On-the-fly pipeline parallelism. ACM Transactions on Parallel Computing (TOPC) 2 3 (2015) 1\u201342.","key":"e_1_3_3_1_27_2","DOI":"10.1145\/2809808"},{"key":"e_1_3_3_1_28_2","series-title":"(OSDI\u201918)","first-page":"611","volume-title":"Proceedings of the 13th USENIX Conference on Operating Systems Design and Implementation","author":"Lee Yunseong","year":"2018","unstructured":"Yunseong Lee, Alberto Scolari, Byung-Gon Chun, Marco\u00a0Domenico Santambrogio, Markus Weimer, and Matteo Interlandi. 2018. Pretzel: opening the black box of machine learning prediction serving systems. In Proceedings of the 13th USENIX Conference on Operating Systems Design and Implementation (Carlsbad, CA, USA) (OSDI\u201918). USENIX Association, USA, 611\u2013626."},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_29_2","DOI":"10.1109\/IC2E48712.2020.00014"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_30_2","DOI":"10.1145\/3373376.3378522"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_31_2","DOI":"10.1145\/3341301.3359646"},{"volume-title":"Multi-Process Service","unstructured":"Nvidia. [n. d.]. Multi-Process Service. https:\/\/docs.nvidia.com\/pdf\/CUDA_Multi_Process_Service_Overview.pdf","key":"e_1_3_3_1_32_2"},{"doi-asserted-by":"publisher","unstructured":"Miquel Peric\u00e0s. 2018. Elastic Places: An Adaptive Resource Manager for Scalable and Portable Performance. ACM Trans. Archit. Code Optim. 15 2 Article 19 (May 2018) 26\u00a0pages. 10.1145\/3185458","key":"e_1_3_3_1_33_2","DOI":"10.1145\/3185458"},{"unstructured":"Francisco Romero Qian Li Neeraja\u00a0J Yadwadkar and Christos Kozyrakis. 2019. INFaaS: A model-less and managed inference serving system. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1905.13348 (2019).","key":"e_1_3_3_1_34_2"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_35_2","DOI":"10.1145\/3472883.3486972"},{"doi-asserted-by":"publisher","unstructured":"Rotem. et\u00a0al. 2022. Intel Alder Lake CPU Architectures. IEEE Micro 42 3 (2022) 13\u201319. 10.1109\/MM.2022.3164338","key":"e_1_3_3_1_36_2","DOI":"10.1109\/MM.2022.3164338"},{"doi-asserted-by":"crossref","unstructured":"Wonik Seo Sanghoon Cha Yeonjae Kim Jaehyuk Huh and Jongse Park. 2021. SLO-aware inference scheduler for heterogeneous processors in edge platforms. ACM Transactions on Architecture and Code Optimization (TACO) 18 4 (2021) 1\u201326.","key":"e_1_3_3_1_37_2","DOI":"10.1145\/3460352"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_38_2","DOI":"10.1145\/3341301.3359658"},{"volume-title":"PPAM 2022","author":"Soomro Pirah\u00a0Noor","unstructured":"Pirah\u00a0Noor Soomro, Mustafa Abduljabbar, Jeronimo Castrillon, and Miquel Peric\u00e0s. [n. d.]. Shisha: Online scheduling of CNN pipelines on heterogeneous architectures. In PPAM 2022.","key":"e_1_3_3_1_39_2"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_40_2","DOI":"10.1007\/978-3-031-39698-4_12"},{"doi-asserted-by":"publisher","key":"e_1_3_3_1_41_2","DOI":"10.1145\/3317550.3321443"},{"key":"e_1_3_3_1_42_2","first-page":"377","volume-title":"14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17)","author":"Zhang Haoyu","year":"2017","unstructured":"Haoyu Zhang, Ganesh Ananthanarayanan, Peter Bodik, Matthai Philipose, Paramvir Bahl, and Michael\u00a0J Freedman. 2017. Live video analytics at scale with approximation and { Delay-Tolerance}. In 14th USENIX Symposium on Networked Systems Design and Implementation (NSDI 17). 377\u2013392."},{"key":"e_1_3_3_1_43_2","first-page":"787","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Zhang Hong","year":"2023","unstructured":"Hong Zhang, Yupeng Tang, Anurag Khandelwal, and Ion Stoica. 2023. { SHEPHERD} : Serving { DNNs} in the wild. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). 787\u2013808."}],"event":{"sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"],"acronym":"CF '25","name":"CF '25: 22nd ACM International Conference on Computing Frontiers","location":"Cagliari Italy"},"container-title":["Proceedings of the 22nd ACM International Conference on Computing Frontiers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3719276.3725190","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,21]],"date-time":"2025-07-21T09:49:27Z","timestamp":1753091367000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3719276.3725190"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,28]]},"references-count":42,"alternative-id":["10.1145\/3719276.3725190","10.1145\/3719276"],"URL":"https:\/\/doi.org\/10.1145\/3719276.3725190","relation":{},"subject":[],"published":{"date-parts":[[2025,5,28]]},"assertion":[{"value":"2025-07-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}