{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T00:20:51Z","timestamp":1777422051689,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3721462.3770773","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T19:56:49Z","timestamp":1765223809000},"page":"256-269","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["xMem: A CPU-Based Approach for Accurate Estimation of GPU Memory in Deep Learning Training Workloads"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-8326-3663","authenticated-orcid":false,"given":"Jiabo","family":"Shi","sequence":"first","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0939-378X","authenticated-orcid":false,"given":"Dimitrios","family":"Pezaros","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4639-436X","authenticated-orcid":false,"given":"Yehia","family":"Elkhatib","sequence":"additional","affiliation":[{"name":"University of Glasgow, Glasgow, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the USENIX conference on Operating Systems Design and Implementation. USENIX Association, 265\u2013283","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, et al. 2016. TensorFlow: A System for Large-Scale Machine Learning. In Proceedings of the USENIX conference on Operating Systems Design and Implementation. USENIX Association, 265\u2013283. 10.5555\/3026877.3026899"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the IEEE International Symposium on Cluster, Cloud and Internet Computing (CCGrid). IEEE, 695\u2013705","author":"Albahar Hadeel","year":"2022","unstructured":"Hadeel Albahar, Shruti Dongare, Yanlin Du, Nannan Zhao, Arnab K. Paul, et al. 2022. SchedTune: A Heterogeneity-Aware GPU Scheduler for Deep Learning. In Proceedings of the IEEE International Symposium on Cluster, Cloud and Internet Computing (CCGrid). IEEE, 695\u2013705. 10.1109\/CCGrid54584.2022.00079"},{"key":"e_1_3_2_1_3_1","volume-title":"PyTorch vs TensorFlow - Which Framework is Best for Data Analysis","author":"Ana Crudu","year":"2025","unstructured":"Crudu Ana. 2025. PyTorch vs TensorFlow - Which Framework is Best for Data Analysis in 2025. https:\/\/moldstud.com\/articles\/p-pytorch-vs-tensornow-which-framework-is-best-for-data-analysis-in-2025"},{"key":"e_1_3_2_1_4_1","unstructured":"Anthropic. 2025. Comment on the framework for artificial intelligence diffusion. Technical Report. https:\/\/www-cdn.anthropic.com\/7449887b6715e3a35f362b1301e5b5d8a6b116e5.pdf"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the International Conference on Machine Learning. JMLR, 2397\u20132430","author":"Biderman Stella","year":"2023","unstructured":"Stella Biderman, Hailey Schoelkopf, Quentin Anthony, Herbie Bradley, Kyle O'Brien, et al. 2023. Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling. In Proceedings of the International Conference on Machine Learning. JMLR, 2397\u20132430."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3391896"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the ACM Symposium on Cloud Computing. ACM, 281\u2013297","author":"Cheng Runxiang","year":"2023","unstructured":"Runxiang Cheng, Chris Cai, Selman Yilmaz, Rahul Mitra, Malay Bag, et al. 2023. Towards GPU Memory Efficiency for Distributed Training at Scale. In Proceedings of the ACM Symposium on Cloud Computing. ACM, 281\u2013297. 10.1145\/3620678.3624661"},{"key":"e_1_3_2_1_8_1","unstructured":"Duncan Clark. 2025. The Great GPU Shortage of 2025: Why Graphics Cards Are So Hard to Find. https:\/\/thinglabs.io\/the-great-gpu-shortage-of-2025-why-graphics-cards-are-so-hard-to-find"},{"key":"e_1_3_2_1_9_1","volume-title":"TensorFlow vs PyTorch: A Comparative Analysis for","author":"Daniel Hayes","year":"2025","unstructured":"Hayes Daniel. 2025. TensorFlow vs PyTorch: A Comparative Analysis for 2025. https:\/\/leapcell.io\/blog\/tensorflow-vs-pytorch-a-comparative-analysis-for-2025"},{"key":"e_1_3_2_1_10_1","unstructured":"DeepSeek-AI Daya Guo Dejian Yang Haowei Zhang Junxiao Song et al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. arXiv:2501.12948 [cs.CL] https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"e_1_3_2_1_11_1","unstructured":"Nolan Dey Gurpreet Gosal Zhiming Chen Hemant Khachane et al. 2023. Cerebras-GPT: Open Compute-Optimal Language Models Trained on the Cere-bras Wafer-Scale Cluster. arXiv:2304.03208 [cs.LG] https:\/\/arxiv.org\/abs\/2304.03208"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering. ACM, 1342\u20131352","author":"Gao Yanjie","year":"2020","unstructured":"Yanjie Gao, Yu Liu, Hongyu Zhang, Zhengxian Li, Yonghao Zhu, et al. 2020. Estimating GPU memory consumption of deep learning models. In Proceedings of the ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering. ACM, 1342\u20131352. 10.1145\/3368089.3417050"},{"key":"e_1_3_2_1_13_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian et al. 2024. The Llama 3 Herd of Models. arXiv:2407.21783 [cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1002\/rob.21918"},{"key":"e_1_3_2_1_15_1","first-page":"11","article-title":"Liquid: Intelligent Resource Estimation and Network-Efficient Scheduling for Deep Learning Jobs on Distributed GPU Clusters","volume":"33","author":"Gu Rong","year":"2022","unstructured":"Rong Gu, Yuquan Chen, Shuai Liu, Haipeng Dai, Guihai Chen, et al. 2022. Liquid: Intelligent Resource Estimation and Network-Efficient Scheduling for Deep Learning Jobs on Distributed GPU Clusters. IEEE Transactions on Parallel and Distributed Systems 33, 11 (November 2022), 2808\u20132820. https:\/\/ieeexplore.ieee.org\/document\/9664375\/","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the IEEE\/ACM International Conference on Software Engineering: New Ideas and Emerging Results (ICSE-NIER). IEEE, 43\u201348","author":"Guerriero Antonio","year":"2023","unstructured":"Antonio Guerriero, Roberto Pietrantuono, and Stefano Russo. 2023. Iterative Assessment and Improvement of DNN Operational Accuracy. In Proceedings of the IEEE\/ACM International Conference on Software Engineering: New Ideas and Emerging Results (ICSE-NIER). IEEE, 43\u201348. 10.1109\/ICSE-NIER58687.2023.00014"},{"key":"e_1_3_2_1_17_1","first-page":"1","article-title":"A study of best-fit memory allocators","volume":"31","author":"Hasan Yusuf","year":"2005","unstructured":"Yusuf Hasan and Morris Chang. 2005. A study of best-fit memory allocators. Computer Languages, Systems & Structures 31, 1 (April 2005), 35\u201348. https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1477842404000211","journal-title":"Computer Languages, Systems & Structures"},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 770\u2013778","author":"He Kaiming","year":"2016","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 770\u2013778. 10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV). IEEE, 1314\u20131324","author":"Howard Andrew","year":"2019","unstructured":"Andrew Howard, Mark Sandler, Bo Chen, Weijun Wang, Liang-Chieh Chen, et al. 2019. Searching for MobileNetV3. In Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV). IEEE, 1314\u20131324. 10.1109\/ICCV.2019.00140"},{"key":"e_1_3_2_1_20_1","unstructured":"Intel. 2024. CPU Dispatcher Control. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/onednn\/developer-guide-reference\/2025-0\/cpu-dispatcher-control.html"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering. ACM, 510\u2013520","author":"Islam Md Johirul","year":"2019","unstructured":"Md Johirul Islam, Giang Nguyen, Rangeet Pan, and Hridesh Rajan. 2019. A comprehensive study on deep learning bug characteristics. In Proceedings of the ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering. ACM, 510\u2013520. 10.1145\/3338906.3338955"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the International Joint Conference on Artificial Intelligence. International Joint Conferences on Artificial Intelligence Organization, 6324\u20136332","author":"Kim Taeho","year":"2024","unstructured":"Taeho Kim, Yanming Wang, Vatshank Chaturvedi, Lokesh Gupta, Seyeon Kim, et al. 2024. LLMem: Estimating GPU Memory Usage for Fine-Tuning Pre-Trained LLMs. In Proceedings of the International Joint Conference on Artificial Intelligence. International Joint Conferences on Artificial Intelligence Organization, 6324\u20136332. 10.24963\/ijcai.2024\/699"},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the International Conference on Learning Representations. ICLR inc., 1\u201315","author":"Diederik","unstructured":"Diederik P. Kingma and Jimmy Ba. 2014. Adam: A Method for Stochastic Optimization. In Proceedings of the International Conference on Learning Representations. ICLR inc., 1\u201315. 10.48550\/ARXIV.1412.6980"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the IEEE International Symposium on High Performance Computer Architecture (HPCA). IEEE, 1259\u20131274","author":"Kokolis Apostolos","year":"2025","unstructured":"Apostolos Kokolis, Michael Kuchnik, John Hoffman, Adithya Kumar, Parth Malani, et al. 2025. Revisiting Reliability in Large-Scale Machine Learning Research Clusters. In Proceedings of the IEEE International Symposium on High Performance Computer Architecture (HPCA). IEEE, 1259\u20131274. 10.1109\/HPCA61900.2025.00096"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the Symposium on Cloud Computing. ACM, 173\u2013189","author":"Li Baolin","year":"2022","unstructured":"Baolin Li, Tirthak Patel, Siddharth Samsi, Vijay Gadepally, and Devesh Tiwari. 2022. MISO: Exploiting multi-instance GPU capability on multi-tenant GPU clusters. In Proceedings of the Symposium on Cloud Computing. ACM, 173\u2013189. 10.1145\/3542929.3563510"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the European Conference on Computer Systems (EuroSys). ACM, 835\u2013850","author":"Li Jiamin","year":"2023","unstructured":"Jiamin Li, Hong Xu, Yibo Zhu, Zherui Liu, Chuanxiong Guo, et al. 2023. Lyra: Elastic Scheduling for Deep Learning Clusters. In Proceedings of the European Conference on Computer Systems (EuroSys). ACM, 835\u2013850. 10.1145\/3552326.3587445"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","first-page":"39674","DOI":"10.1109\/ACCESS.2022.3164510","article-title":"TBEM: Testing-Based GPU-Memory Consumption Estimation for Deep Learning","volume":"10","author":"Liu Haiyi","year":"2022","unstructured":"Haiyi Liu, Shaoying Liu, Chenglong Wen, and W. Eric Wong. 2022. TBEM: Testing-Based GPU-Memory Consumption Estimation for Deep Learning. IEEE Access 10 (2022), 39674\u201339680. https:\/\/ieeexplore.ieee.org\/document\/9755116\/","journal-title":"IEEE Access"},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 11976\u201311986","author":"Liu Zhuang","year":"2022","unstructured":"Zhuang Liu, Hanzi Mao, Chao-Yuan Wu, Christoph Feichtenhofer, Trevor Darrell, et al. 2022. A ConvNet for the 2020s. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 11976\u201311986. 10.48550\/ARXIV.2201.03545"},{"key":"e_1_3_2_1_30_1","unstructured":"NVIDIA Corporation. 2024. NVML-NVIDIA Management Library. https:\/\/developer.nvidia.com\/management-library-nvml"},{"key":"e_1_3_2_1_31_1","unstructured":"NVIDIA Corporation. 2025. TensorFlow User Guide. https:\/\/docs.nvidia.com\/deeplearning\/frameworks\/tensorflow-user-guide\/index.html"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems (NIPS). ACM, 8026\u20138037","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, et al. 2019. PyTorch: An Imperative Style, High-Performance Deep Learning Library. In Proceedings of the International Conference on Neural Information Processing Systems (NIPS). ACM, 8026\u20138037. 10.5555\/3454287.3455008"},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the International Middleware Conference. ACM, 313\u2013326","author":"Pavlidakis Manos","year":"2024","unstructured":"Manos Pavlidakis, Giorgos Vasiliadis, Stelios Mavridis, Anargyros Argyros, Antony Chazapis, et al. 2024. Guardian: Safe GPU Sharing in Multi-Tenant Environments. In Proceedings of the International Middleware Conference. ACM, 313\u2013326. 10.1145\/3652892.3700768"},{"key":"e_1_3_2_1_34_1","unstructured":"PyTorch. 2023. Training with PyTorch. http:\/\/docs.pytorch.org\/tutorials\/beginner\/introyt\/trainingyt.html"},{"key":"e_1_3_2_1_35_1","unstructured":"PyTorch. 2024. CPP API. https:\/\/pytorch.org\/cppdocs"},{"key":"e_1_3_2_1_36_1","unstructured":"PyTorch. 2024. CUDA semantics. https:\/\/pytorch.org\/docs\/2.6\/notes\/cuda.html"},{"key":"e_1_3_2_1_37_1","unstructured":"PyTorch. 2024. GitHub-PyTorch Caching Allocator CPP Source Code. https:\/\/github.com\/pytorch\/pytorch\/blob\/release\/2.6\/c10\/cuda\/CUDACachingAllocator.cpp"},{"key":"e_1_3_2_1_38_1","unstructured":"PyTorch. 2024. PyTorch Profiler. https:\/\/pytorch.org\/tutorials\/recipes\/recipes\/profiler_recipe.html"},{"key":"e_1_3_2_1_39_1","unstructured":"PyTorch. 2024. Understanding CUDA Memory Usage. https:\/\/pytorch.org\/docs\/stable\/torch_cuda_memory.html"},{"key":"e_1_3_2_1_40_1","unstructured":"PyTorch. 2025. Docker Hub-pytorch\/pytorch. https:\/\/hub.docker.com\/r\/pytorch\/pytorch"},{"key":"e_1_3_2_1_41_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei et al. 2019. Language models are unsupervised multitask learners. https:\/\/cdn.openai.com\/better-language-models\/language_models_are_unsupervised_multitask_learners.pdf"},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 10425\u201310433","author":"Radosavovic Ilija","year":"2020","unstructured":"Ilija Radosavovic, Raj Prateek Kosaraju, Ross Girshick, Kaiming He, and Piotr Dollar. 2020. Designing Network Design Spaces. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 10425\u201310433. 10.1109\/CVPR42600.2020.01044"},{"key":"e_1_3_2_1_43_1","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, et al. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of Machine Learning Research 21, 140 (2020), 1\u201367. http:\/\/jmlr.org\/papers\/v21\/20-074.html","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_44_1","unstructured":"Habib Raza. 2023. OpenAI's plans according to Sam Altman. https:\/\/website-754fwhahs-humanloopml.vercel.app\/blog\/open_ai_talk"},{"key":"e_1_3_2_1_45_1","volume-title":"Kroese","author":"Rubinstein Reuven Y.","year":"2017","unstructured":"Reuven Y. Rubinstein and Dirk P. Kroese. 2017. Simulation and the Monte Carlo method (third ed.). Wiley, Hoboken, New Jersey."},{"key":"e_1_3_2_1_46_1","unstructured":"Sebastian Ruder. 2016. An overview of gradient descent optimization algorithms. arXiv:1609.04747 [cs.LG] https:\/\/arxiv.org\/abs\/1609.04747"},{"key":"e_1_3_2_1_47_1","volume-title":"PyTorch vs TensorFlow","author":"Ryan O'Connor","year":"2023","unstructured":"O'Connor Ryan. 2023. PyTorch vs TensorFlow in 2023. https:\/\/www.assemblyai.com\/blog\/pytorch-vs-tensorflow-in-2023"},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. IEEE, 4510\u20134520","author":"Sandler Mark","year":"2018","unstructured":"Mark Sandler, Andrew Howard, Menglong Zhu, Andrey Zhmoginov, and Liang-Chieh Chen. 2018. MobileNetV2: Inverted Residuals and Linear Bottlenecks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. IEEE, 4510\u20134520. 10.1109\/CVPR.2018.00474"},{"key":"e_1_3_2_1_49_1","unstructured":"Victor Sanh Lysandre Debut Julien Chaumond and Thomas Wolf. 2019. DistilBERT a distilled version of BERT: smaller faster cheaper and lighter. arXiv:1910.01108 [cs.CL] https:\/\/arxiv.org\/abs\/1910.01108"},{"key":"e_1_3_2_1_50_1","volume-title":"The Analysis of Variance","author":"Scheffe Henry","unstructured":"Henry Scheffe. 1999. The Analysis of Variance. Wiley-Interscience, New York."},{"key":"e_1_3_2_1_51_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR). 10","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very Deep Convolutional Networks for Large-Scale Image Recognition. In Proceedings of the International Conference on Learning Representations (ICLR). 10.48550\/ARXIV.1409.1556"},{"key":"e_1_3_2_1_52_1","volume-title":"Saqib Iqbal, Mohammed Alghobiri, Tassawar Iqbal, et al.","author":"Talha Mian Muhammad","year":"2023","unstructured":"Mian Muhammad Talha, Hikmat Ullah Khan, Saqib Iqbal, Mohammed Alghobiri, Tassawar Iqbal, et al. 2023. Deep learning in news recommender systems: A comprehensive survey, challenges and future trends. Neurocomputing 562 (December 2023), 126881. https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231223010044"},{"key":"e_1_3_2_1_53_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 2815\u20132823","author":"Tan Mingxing","year":"2019","unstructured":"Mingxing Tan, Bo Chen, Ruoming Pang, Vijay Vasudevan, Mark Sandler, et al. 2019. MnasNet: Platform-Aware Neural Architecture Search for Mobile. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). IEEE, 2815\u20132823. 10.1109\/CVPR.2019.00293"},{"key":"e_1_3_2_1_54_1","unstructured":"TensorFlow. 2025. TensorFlow GPU-BFC Allocator Source Code. https:\/\/github.com\/tensorflow\/tensorflow\/blob\/176771f3116b09076ff0e15c1687050f6c9e799c\/tensorflow\/core\/common_runtime\/gpu\/gpu_bfc_allocator.h"},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the International Conference on Neural Information Processing Systems. Curran Associates, Inc., 6000\u20136010","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, et al. 2017. Attention Is All You Need. In Proceedings of the International Conference on Neural Information Processing Systems. Curran Associates, Inc., 6000\u20136010. 10.5555\/3295222.3295349"},{"key":"e_1_3_2_1_56_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 8695\u20138704","author":"Wang Sheng-Yu","unstructured":"Sheng-Yu Wang, Oliver Wang, Richard Zhang, Andrew Owens, and Alexei A. Efros. 2020. CNN-Generated Images Are Surprisingly Easy to Spot... for Now. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 8695\u20138704."},{"key":"e_1_3_2_1_57_1","volume-title":"Proceedings of the USENIX symposium on networked systems design and implementation (NSDI 22)","author":"Weng Qizhen","year":"2022","unstructured":"Qizhen Weng, Wencong Xiao, Yinghao Yu, Wei Wang, Cheng Wang, et al. 2022. MLaaS in the wild: Workload analysis and scheduling in Large-Scale heterogeneous GPU clusters. In Proceedings of the USENIX symposium on networked systems design and implementation (NSDI 22). USENIX Association, 945\u2013960. https:\/\/www.usenix.org\/conference\/nsdi22\/presentation\/weng"},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of the USENIX annual technical conference (ATC 23)","author":"Weng Qizhen","year":"2023","unstructured":"Qizhen Weng, Lingyun Yang, Yinghao Yu, Wei Wang, Xiaochuan Tang, et al. 2023. Beware of fragmentation: Scheduling GPU-Sharing workloads with fragmentation gradient descent. In Proceedings of the USENIX annual technical conference (ATC 23). USENIX Association, 995\u20131008. https:\/\/www.usenix.org\/conference\/atc23\/presentation\/weng"},{"key":"e_1_3_2_1_59_1","volume-title":"Proceedings of the USENIX symposium on operating systems design and implementation (OSDI 18)","author":"Xiao Wencong","year":"2018","unstructured":"Wencong Xiao, Romil Bhardwaj, Ramachandran Ramjee, Muthian Sivathanu, Nipun Kwatra, et al. 2018. Gandiva: Introspective cluster scheduling for deep learning. In Proceedings of the USENIX symposium on operating systems design and implementation (OSDI 18). USENIX Association, 595\u2013610. https:\/\/www.usenix.org\/conference\/osdi18\/presentation\/xiao"},{"key":"e_1_3_2_1_60_1","volume-title":"Proceedings of the USENIX symposium on operating systems design and implementation (OSDI). USENIX Association, 533\u2013548","author":"Xiao Wencong","year":"2020","unstructured":"Wencong Xiao, Shiru Ren, Yong Li, Yang Zhang, Pengyang Hou, et al. 2020. AntMan: Dynamic scaling on GPU clusters for deep learning. In Proceedings of the USENIX symposium on operating systems design and implementation (OSDI). USENIX Association, 533\u2013548. 10.5555\/3488766.3488796"},{"key":"e_1_3_2_1_61_1","volume-title":"Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 5934\u20135938","author":"Xiong W.","year":"2018","unstructured":"W. Xiong, L. Wu, F. Alleva, J. Droppo, X. Huang, et al. 2018. The Microsoft 2017 Conversational Speech Recognition System. In Proceedings of the IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 5934\u20135938. 10.1109\/ICASSP.2018.8461870"},{"key":"e_1_3_2_1_62_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui et al. 2025. Qwen3 Technical Report. arXiv:2505.09388 [cs.CL] https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_2_1_63_1","first-page":"1","article-title":"Horus: Interference-Aware and Prediction-Based Scheduling in Deep Learning Systems","volume":"33","author":"Yeung Gingfung","year":"2022","unstructured":"Gingfung Yeung, Damian Borowiec, Renyu Yang, Adrian Friday, Richard Harper, et al. 2022. Horus: Interference-Aware and Prediction-Based Scheduling in Deep Learning Systems. IEEE Transactions on Parallel and Distributed Systems 33, 1 (January 2022), 88\u2013100. https:\/\/ieeexplore.ieee.org\/document\/9428512\/","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"e_1_3_2_1_64_1","volume-title":"Proceedings of the Machine Learning and Systems. ACM, 98\u2013111","author":"Yu Peifeng","year":"2020","unstructured":"Peifeng Yu and Mosharaf Chowdhury. 2020. Fine-grained GPU Sharing Primitives for Deep Learning Applications. In Proceedings of the Machine Learning and Systems. ACM, 98\u2013111. https:\/\/proceedings.mlsys.org\/paper_files\/paper\/2020\/file\/d9cd83bc91b8c36a0c7c0fcca59228f2-Paper.pdf"},{"key":"e_1_3_2_1_65_1","volume-title":"Proceedings of the USENIX symposium on networked systems design and implementation (NSDI 18)","author":"Zhang Kai","year":"2018","unstructured":"Kai Zhang, Bingsheng He, Jiayu Hu, Zeke Wang, Bei Hua, et al. 2018. G-NET: Effective GPU sharing in NFV systems. In Proceedings of the USENIX symposium on networked systems design and implementation (NSDI 18). USENIX Association, 187\u2013200. https:\/\/www.usenix.org\/conference\/nsdi18\/presentation\/zhang-kai"},{"key":"e_1_3_2_1_66_1","volume-title":"Proceedings of the ACM\/IEEE International Conference on Software Engineering. ACM, 1159\u20131170","author":"Zhang Ru","year":"2020","unstructured":"Ru Zhang, Wencong Xiao, Hongyu Zhang, Yu Liu, Haoxiang Lin, et al. 2020. An empirical study on program failures of deep learning jobs. In Proceedings of the ACM\/IEEE International Conference on Software Engineering. ACM, 1159\u20131170. 10.1145\/3377811.3380362"},{"key":"e_1_3_2_1_67_1","volume-title":"OPT: Open Pre-trained Transformer Language Models. arXiv:2205.01068 [cs.CL] https:\/\/arxiv.org\/abs\/2205.01068","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, et al. 2022. OPT: Open Pre-trained Transformer Language Models. arXiv:2205.01068 [cs.CL] https:\/\/arxiv.org\/abs\/2205.01068"},{"key":"e_1_3_2_1_68_1","volume-title":"Rubick: Exploiting Job Reconfigurability for Deep Learning Cluster Scheduling. arXiv:2408.08586 [cs.DC] https:\/\/arxiv.org\/abs\/2408.08586","author":"Zhang Xinyi","year":"2024","unstructured":"Xinyi Zhang, Hanyu Zhao, Wencong Xiao, Xianyan Jia, Fei Xu, et al. 2024. Rubick: Exploiting Job Reconfigurability for Deep Learning Cluster Scheduling. arXiv:2408.08586 [cs.DC] https:\/\/arxiv.org\/abs\/2408.08586"},{"key":"e_1_3_2_1_69_1","volume-title":"NLP-based Generation of Ontological System Descriptions for Composition of Smart Home Devices. In International Conference on Web Services (ICWS). IEEE. 10","author":"Zhang Ziyu","year":"2023","unstructured":"Ziyu Zhang, Yehia Elkhatib, and Abdessalam Elhabbash. 2023. NLP-based Generation of Ontological System Descriptions for Composition of Smart Home Devices. In International Conference on Web Services (ICWS). IEEE. 10.1109\/ICWS60048.2023.00055"}],"event":{"name":"MIDDLEWARE '25: 26th International Middleware Conference","location":"Vanderbilt University Nashville TN USA","acronym":"MIDDLEWARE '25","sponsor":["IFIP","Usenix"]},"container-title":["Proceedings of the 26th International Middleware Conference"],"original-title":[],"deposited":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T20:02:18Z","timestamp":1765224138000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3721462.3770773"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":69,"alternative-id":["10.1145\/3721462.3770773","10.1145\/3721462"],"URL":"https:\/\/doi.org\/10.1145\/3721462.3770773","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}