{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,26]],"date-time":"2026-05-26T15:03:22Z","timestamp":1779807802042,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":10,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,8]]},"DOI":"10.1145\/3801489.3806875","type":"proceedings-article","created":{"date-parts":[[2026,5,26]],"date-time":"2026-05-26T14:22:48Z","timestamp":1779805368000},"page":"232-234","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SageServe: Optimizing LLM Serving on Cloud Data Centers with Forecast Aware Auto-Scaling"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2526-5780","authenticated-orcid":false,"given":"Shashwat","family":"Jaiswal","sequence":"first","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana-Champaign, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0941-5753","authenticated-orcid":false,"given":"Kunal","family":"Jain","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4140-7774","authenticated-orcid":false,"given":"Yogesh","family":"Simmhan","sequence":"additional","affiliation":[{"name":"Indian Institute of Science, Bangalore, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6296-0395","authenticated-orcid":false,"given":"Anjaly","family":"Parayil","sequence":"additional","affiliation":[{"name":"Microsoft, Bangalore, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7068-5627","authenticated-orcid":false,"given":"Ankur","family":"Mallick","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4019-5327","authenticated-orcid":false,"given":"Rujia","family":"Wang","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9387-5886","authenticated-orcid":false,"given":"Renee St.","family":"Amant","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0102-8139","authenticated-orcid":false,"given":"Chetan","family":"Bansal","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8957-7628","authenticated-orcid":false,"given":"Victor","family":"R\u00fchle","sequence":"additional","affiliation":[{"name":"Microsoft, Cambridge, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4412-1252","authenticated-orcid":false,"given":"Anoop","family":"Kulkarni","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8558-5954","authenticated-orcid":false,"given":"Steve","family":"Kofsky","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0204-7187","authenticated-orcid":false,"given":"Saravan","family":"Rajmohan","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,8]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"AWS. 2024. Introducing Fast Model Loader in SageMaker Inference: Accelerate autoscaling for your Large Language Models (LLMs). https:\/\/aws.amazon.com\/blogs\/machine-learning\/introducing-fast-model-loader-in-sagemaker-inference-accelerate-autoscaling-for-your-large-language-models-llms-part-1\/"},{"key":"e_1_3_2_1_2_1","unstructured":"Google Cloud. 2025. From LLMs to image generation: Accelerate inference workloads with AI Hypercomputer. https:\/\/cloud.google.com\/blog\/products\/compute\/ai-hypercomputer-inference-updates-for-google-cloud-tpu-and-gpu"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3771576"},{"key":"e_1_3_2_1_4_1","volume-title":"Towards efficient generative large language model serving: A survey from algorithms to systems. Comput. Surveys","author":"Miao Xupeng","year":"2023","unstructured":"Xupeng Miao, Gabriele Oliaro, Zhihao Zhang, Xinhao Cheng, Hongyi Jin, Tianqi Chen, and Zhihao Jia. 2023. Towards efficient generative large language model serving: A survey from algorithms to systems. Comput. Surveys (2023)."},{"key":"e_1_3_2_1_5_1","unstructured":"Archit Patke Dhemath Reddy Saurabh Jha Chandra Narayanaswami Zbigniew Kalbarczyk and Ravishankar Iyer. 2025. Hierarchical Autoscaling for Large Language Model Serving with Chiron. arXiv:2501.08090 [cs.DC]"},{"key":"e_1_3_2_1_6_1","unstructured":"Ye Qi. 2025. Scaling Large Language Model Serving Infrastructure at Meta. In QCon San Francisco. https:\/\/www.infoq.com\/presentations\/llm-meta\/"},{"key":"e_1_3_2_1_7_1","unstructured":"P. Schmid O. Sansevieroa P. Cuenca and L. Tunstall. [n.d.]. Llama 2 is here - Get it on Hugging Face [Online]. https:\/\/huggingface.co\/blog\/llama2"},{"key":"e_1_3_2_1_8_1","unstructured":"Top500. 2025. Top 500 Supercomputing List. https:\/\/www.top500.org\/system\/180236\/"},{"key":"e_1_3_2_1_9_1","unstructured":"HPC Wire. 2024. AWS Delivers the AI Heat: Project Rainier and GenAI Innovations Lead the Way. https:\/\/www.hpcwire.com\/2024\/12\/05\/aws-delivers-the-ai-heat-project-rainier-and-genai-innovations-lead-the-way\/"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3708530"}],"event":{"name":"SIGMETRICS '26: ACM SIGMETRICS International Conference on Measurement and Modeling of Computer Systems","location":"Ann Arbor MI USA","sponsor":["SIGMETRICS ACM Special Interest Group on Measurement and Evaluation"]},"container-title":["Abstracts of the 2026 ACM SIGMETRICS International Conference on Measurement and Modeling of Computer Systems"],"original-title":[],"deposited":{"date-parts":[[2026,5,26]],"date-time":"2026-05-26T14:23:16Z","timestamp":1779805396000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3801489.3806875"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,8]]},"references-count":10,"alternative-id":["10.1145\/3801489.3806875","10.1145\/3801489"],"URL":"https:\/\/doi.org\/10.1145\/3801489.3806875","relation":{},"subject":[],"published":{"date-parts":[[2026,6,8]]},"assertion":[{"value":"2026-06-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}