{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T16:57:04Z","timestamp":1775667424970,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,20]],"date-time":"2024-11-20T00:00:00Z","timestamp":1732060800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Science Foundation of China","award":["62232012"],"award-info":[{"award-number":["62232012"]}]},{"name":"National Key Research & Development (R&D) Plan","award":["2022YFB4501703"],"award-info":[{"award-number":["2022YFB4501703"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,20]]},"DOI":"10.1145\/3698038.3698556","type":"proceedings-article","created":{"date-parts":[[2024,11,14]],"date-time":"2024-11-14T06:32:43Z","timestamp":1731565963000},"page":"487-504","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["InferCool: Enhancing AI Inference Cooling through Transparent, Non-Intrusive Task Reassignment"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8870-4309","authenticated-orcid":false,"given":"Qiangyu","family":"Pei","sequence":"first","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China and National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7181-6128","authenticated-orcid":false,"given":"Lin","family":"Wang","sequence":"additional","affiliation":[{"name":"Paderborn University, Paderborn, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6574-6395","authenticated-orcid":false,"given":"Dong","family":"Zhang","sequence":"additional","affiliation":[{"name":"Inspur Data Co., Ltd., Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6255-1055","authenticated-orcid":false,"given":"Bingheng","family":"Yan","sequence":"additional","affiliation":[{"name":"Inspur Data Co., Ltd., Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0782-0450","authenticated-orcid":false,"given":"Chen","family":"Yu","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Wuhan, China and National Engineering Research Center for Big Data Technology and System, Services Computing Technology and System Lab, Cluster and Grid Computing Lab, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8570-1345","authenticated-orcid":false,"given":"Fangming","family":"Liu","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology, Peng Cheng Laboratory, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,11,20]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/2207222.2207227"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00073"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00040"},{"key":"e_1_3_2_1_4_1","volume-title":"Retrieved","author":"Team Archive","year":"2024","unstructured":"Archive Team. [n. d.]. Archive Team: The Twitter Stream Grab. Retrieved June 29, 2024 from https:\/\/archive.org\/details\/twitterstream"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2012.6169035"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3542929.3563498"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3396851.3402658"},{"key":"e_1_3_2_1_8_1","volume-title":"Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In 2022 USENIX Annual Technical Conference (USENIX ATC 22)","author":"Choi Seungbeom","year":"2022","unstructured":"Seungbeom Choi, Sunho Lee, Yeonjae Kim, Jongse Park, Youngjin Kwon, and Jaehyuk Huh. 2022. Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In 2022 USENIX Annual Technical Conference (USENIX ATC 22). 199--216."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476143"},{"key":"e_1_3_2_1_10_1","volume-title":"Compute and Energy Consumption Trends in Deep Learning Inference. arXiv preprint arXiv:2109.05472","author":"Desislavov Radosvet","year":"2021","unstructured":"Radosvet Desislavov, Fernando Mart\u00ednez-Plumed, and Jos\u00e9 Hern\u00e1ndez-Orallo. 2021. Compute and Energy Consumption Trends in Deep Learning Inference. arXiv preprint arXiv:2109.05472 (2021)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3419111.3421284"},{"key":"e_1_3_2_1_12_1","volume-title":"The Design and Operation of CloudLab. In 2019 USENIX Annual Technical Conference (USENIX ATC 19)","author":"Duplyakin Dmitry","year":"2019","unstructured":"Dmitry Duplyakin, Robert Ricci, Aleksander Maricq, Gary Wong, Jonathon Duerig, Eric Eide, Leigh Stoller, Mike Hibler, David Johnson, Kirk Webb, Aditya Akella, Kuangching Wang, Glenn Ricart, Larry Landweber, Chip Elliott, Michael Zink, Emmanuel Cecchet, Snigdhaswin Kar, and Prabodh Mishra. 2019. The Design and Operation of CloudLab. In 2019 USENIX Annual Technical Conference (USENIX ATC 19). 1--14."},{"key":"e_1_3_2_1_13_1","volume-title":"Retrieved","author":"Manual Energy Efficiency","year":"2015","unstructured":"Energy Efficiency Manual. 2015. Keep the chilled water supply temperature as high as possible. Retrieved October 15, 2024 from http:\/\/energybooks.com\/wp-content\/uploads\/2015\/07\/264266.pdf"},{"key":"e_1_3_2_1_14_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Gujarati Arpan","year":"2020","unstructured":"Arpan Gujarati, Reza Karimi, Safya Alzayat, Wei Hao, Antoine Kaufmann, Ymir Vigfusson, and Jonathan Mace. 2020. Serving DNNs like Clockwork: Performance Predictability from the Bottom Up. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). 443--462."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357223.3362714"},{"key":"e_1_3_2_1_16_1","volume-title":"d.]. Hugging Face. Retrieved","author":"Face Hugging","year":"2024","unstructured":"Hugging Face. [n. d.]. Hugging Face. Retrieved October 15, 2024 from https:\/\/huggingface.co\/"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.23919\/DATE.2019.8715033"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00055"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322236"},{"key":"e_1_3_2_1_20_1","volume-title":"Retrieved","author":"Kennedy Patrick","year":"2018","unstructured":"Patrick Kennedy. 2018. Baidu X-MAN Liquid Cooled 8-Way NVIDIA Tesla V100 Shelf. Retrieved October 15, 2024 from https:\/\/www.servethehome.com\/baidu-x-man-liquid-cooled-8-way-nvidia-tesla-v100-shelf\/"},{"key":"e_1_3_2_1_21_1","volume-title":"Retrieved","author":"Kevin Lee Mathew Oldham","year":"2024","unstructured":"Mathew Oldham Kevin Lee, Adi Gangidi. 2024. Building Meta's GenAI Infrastructure. Retrieved October 15, 2024 from https:\/\/engineering.fb.com\/2024\/03\/12\/data-center-engineering\/building-metas-genai-infrastructure\/"},{"key":"e_1_3_2_1_22_1","volume-title":"Retrieved","year":"2024","unstructured":"Kubernetes. [n. d.]. Production-Grade Container Orchestration. Retrieved October 15, 2024 from https:\/\/kubernetes.io\/"},{"key":"e_1_3_2_1_23_1","unstructured":"Kubernetes. [n.d.]. Scheduler Configuration. Retrieved October 15 2024 from https:\/\/kubernetes.io\/docs\/reference\/scheduling\/config\/"},{"key":"e_1_3_2_1_24_1","volume-title":"Characterizing Multi-Instance GPU for Machine Learning Workloads. In 2022 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW). IEEE, 724--731","author":"Li Baolin","year":"2022","unstructured":"Baolin Li, Viiay Gadepally, Siddharth Samsi, and Devesh Tiwari. 2022. Characterizing Multi-Instance GPU for Machine Learning Workloads. In 2022 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW). IEEE, 724--731."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3542929.3563510"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607034"},{"key":"e_1_3_2_1_27_1","volume-title":"Retrieved","author":"Linden Greg","year":"2024","unstructured":"Greg Linden. [n.d.]. Make Data Useful. Retrieved October 15, 2024 from https:\/\/www.scribd.com\/doc\/4970486\/"},{"key":"e_1_3_2_1_28_1","volume-title":"Retrieved","author":"Maggie Zhang Chetan Tekur","year":"2020","unstructured":"Chetan Tekur Maggie Zhang, James Sohn. 2020. Getting the Most Out of the NVIDIA A100 GPU with Multi-Instance GPU. Retrieved October 15, 2024 from https:\/\/developer.nvidia.com\/blog\/getting-the-most-out-of-the-a100-gpu-with-multi-instance-gpu\/"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2806777.2806938"},{"key":"e_1_3_2_1_30_1","volume-title":"USENIX Annual Technical Conference, General Track. 61--75","author":"Moore Justin D","year":"2005","unstructured":"Justin D Moore, Jeffrey S Chase, Parthasarathy Ranganathan, and Ratnesh K Sharma. 2005. Making Scheduling \"Cool\": Temperature-Aware Workload Placement in Data Centers. In USENIX Annual Technical Conference, General Track. 61--75."},{"key":"e_1_3_2_1_31_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. [n.d.]. NVIDIA Multi-Instance GPU. Retrieved October 15, 2024 from https:\/\/www.nvidia.com\/en-us\/technologies\/multi-instance-gpu\/"},{"key":"e_1_3_2_1_32_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2022","unstructured":"NVIDIA. 2022. NVIDIA Teams With Microsoft to Build Massive Cloud AI Computer. Retrieved October 15, 2024 from https:\/\/nvidianews.nvidia.com\/news\/nvidia-microsoft-accelerate-cloud-enterprise-ai"},{"key":"e_1_3_2_1_33_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. 2024. A100 MIG Profiles. Retrieved October 15, 2024 from https:\/\/docs.nvidia.com\/datacenter\/tesla\/mig-user-guide\/#a100-mig-profiles"},{"key":"e_1_3_2_1_34_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. 2024. About the NVIDIA GPU Operator. Retrieved October 15, 2024 from https:\/\/docs.nvidia.com\/datacenter\/cloud-native\/gpu-operator\/latest\/index.html"},{"key":"e_1_3_2_1_35_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. 2024. Multi-Process Service. Retrieved October 15, 2024 from https:\/\/docs.nvidia.com\/deploy\/mps\/index.html"},{"key":"e_1_3_2_1_36_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. 2024. NVIDIA DCGM. Retrieved October 15, 2024 from https:\/\/developer.nvidia.com\/dcgm"},{"key":"e_1_3_2_1_37_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. 2024. NVIDIA GB200 NVL72. Retrieved October 15, 2024 from https:\/\/www.nvidia.com\/en-us\/data-center\/gb200-nvl72\/"},{"key":"e_1_3_2_1_38_1","volume-title":"Retrieved","author":"Pei Qiangyu","year":"2024","unstructured":"Qiangyu Pei. 2024. Supplementary file for InferCool. Retrieved October 15, 2024 from https:\/\/qiangyupei.github.io\/files\/SoCC24_InferCool_supplementary.pdf"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507713"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620678.3624664"},{"key":"e_1_3_2_1_41_1","volume-title":"Language Models are Unsupervised Multitask Learners. OpenAI blog 1, 8","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, and Ilya Sutskever. 2019. Language Models are Unsupervised Multitask Learners. OpenAI blog 1, 8 (2019), 9."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDCS.2019.00070"},{"key":"e_1_3_2_1_43_1","volume-title":"INFaaS: Automated Model-Less Inference Serving. In 2021 USENIX Annual Technical Conference (USENIX ATC 21)","author":"Romero Francisco","year":"2021","unstructured":"Francisco Romero, Qian Li, Neeraja J Yadwadkar, and Christos Kozyrakis. 2021. INFaaS: Automated Model-Less Inference Serving. In 2021 USENIX Annual Technical Conference (USENIX ATC 21). 397--411."},{"key":"e_1_3_2_1_44_1","volume-title":"Retrieved","author":"Ryan Carol","year":"2024","unstructured":"Carol Ryan. 2024. Energy-Guzzling AI Is Also the Future of Energy Savings. Retrieved October 15, 2024 from https:\/\/www.wsj.com\/business\/energy-oil\/ai-data-centers-energy-savings-d602296e"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359658"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2749474"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00013"},{"key":"e_1_3_2_1_48_1","volume-title":"Retrieved","year":"2024","unstructured":"Vertiv. [n. d.]. Understanding the Cost of Data Center Downtime: An Analysis of the Financial Impact on Infrastructure Vulnerability. Retrieved October 15, 2024 from https:\/\/www.vertiv.com\/4a3537\/globalassets\/images\/about-images\/news-and-insights\/articles\/white-papers\/understanding-the-cost-of-data-center\/datacenter-downtime-wp-en-na-sl-24661_51225_1.pdf"},{"key":"e_1_3_2_1_49_1","volume-title":"High-Throughput Inference. In Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems. 768--781","author":"Yang Yanan","year":"2022","unstructured":"Yanan Yang, Laiping Zhao, Yiming Li, Huanyu Zhang, Jie Li, Mingyang Zhao, Xingzhen Chen, and Keqiu Li. 2022. INFless: A Native Serverless System for Low-Latency, High-Throughput Inference. In Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems. 768--781."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/2670979.2670996"},{"key":"e_1_3_2_1_51_1","volume-title":"Salus: Fine-Grained GPU Sharing Primitives for Deep Learning Applications. MLSys' 20","author":"Yu Peifeng","year":"2020","unstructured":"Peifeng Yu and Mosharaf Chowdhury. 2020. Salus: Fine-Grained GPU Sharing Primitives for Deep Learning Applications. MLSys' 20 (2020)."},{"key":"e_1_3_2_1_52_1","volume-title":"SLO-Aware Machine Learning Inference Serving. In 2019 USENIX Annual Technical Conference. 1049--1062","author":"Zhang Chengliang","year":"2019","unstructured":"Chengliang Zhang, Minchen Yu, Wei Wang, and Feng Yan. 2019. MArk: Exploiting Cloud Services for Cost-Effective, SLO-Aware Machine Learning Inference Serving. In 2019 USENIX Annual Technical Conference. 1049--1062."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507721"}],"event":{"name":"SoCC '24: ACM Symposium on Cloud Computing","location":"Redmond WA USA","acronym":"SoCC '24","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the ACM Symposium on Cloud Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3698038.3698556","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3698038.3698556","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T19:00:45Z","timestamp":1755889245000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3698038.3698556"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,20]]},"references-count":53,"alternative-id":["10.1145\/3698038.3698556","10.1145\/3698038"],"URL":"https:\/\/doi.org\/10.1145\/3698038.3698556","relation":{},"subject":[],"published":{"date-parts":[[2024,11,20]]},"assertion":[{"value":"2024-11-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}