{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T00:07:49Z","timestamp":1755907669224,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":95,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,30]],"date-time":"2024-05-30T00:00:00Z","timestamp":1717027200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,30]]},"DOI":"10.1145\/3650200.3656599","type":"proceedings-article","created":{"date-parts":[[2024,6,3]],"date-time":"2024-06-03T14:11:54Z","timestamp":1717423914000},"page":"259-271","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Ymir: A Scheduler for Foundation Model Fine-tuning Workloads in Datacenters"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7048-1722","authenticated-orcid":false,"given":"Wei","family":"Gao","sequence":"first","affiliation":[{"name":"S-Lab, Nanyang Technological University, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8243-7772","authenticated-orcid":false,"given":"Weiming","family":"Zhuang","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore and Sony AI, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5278-5255","authenticated-orcid":false,"given":"Minghao","family":"Li","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8456-0491","authenticated-orcid":false,"given":"Peng","family":"Sun","sequence":"additional","affiliation":[{"name":"SenseTime, China and Shanghai AI Lab, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2751-5114","authenticated-orcid":false,"given":"Yonggang","family":"Wen","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6595-6650","authenticated-orcid":false,"given":"Tianwei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,6,3]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2022. HuggingFace Model Hub. https:\/\/huggingface. co\/models?sort=downloads.."},{"key":"e_1_3_2_1_2_1","unstructured":"2022. OpenAI Fine-tuning Service. https:\/\/beta.openai.com\/docs\/guides\/fine-tuning. ."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Alessandro Achille Michael Lam Rahul Tewari Avinash Ravichandran Subhransu Maji Charless\u00a0C Fowlkes Stefano Soatto and Pietro Perona. 2019. Task2vec: Task embedding for meta-learning. In CVPR. 6430\u20136439.","DOI":"10.1109\/ICCV.2019.00653"},{"key":"e_1_3_2_1_4_1","volume-title":"Ext5: Towards extreme multi-task scaling for transfer learning. arXiv preprint arXiv:2111.10952","author":"Aribandi Vamsi","year":"2021","unstructured":"Vamsi Aribandi, Yi Tay, Tal Schuster, Jinfeng Rao, Huaixiu\u00a0Steven Zheng, Sanket\u00a0Vaibhav Mehta, Honglei Zhuang, Vinh\u00a0Q Tran, Dara Bahri, Jianmo Ni, 2021. Ext5: Towards extreme multi-task scaling for transfer learning. arXiv preprint arXiv:2111.10952 (2021)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Sanjith Athlur Nitika Saran Muthian Sivathanu Ramachandran Ramjee and Nipun Kwatra. 2022. Varuna: scalable low-cost training of massive deep learning models. In Eurosys. 472\u2013487.","DOI":"10.1145\/3492321.3519584"},{"key":"e_1_3_2_1_6_1","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Yuhao Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Neurips 33 (2020), 12449\u201312460.","journal-title":"Neurips"},{"key":"e_1_3_2_1_7_1","volume-title":"PipeSwitch: Fast Pipelined Context Switching for Deep Learning Applications. In 14th USENIX Symposium on Operating Systems Design and Implementation(OSDI \u201920)","author":"Bai Zhihao","year":"2020","unstructured":"Zhihao Bai, Zhen Zhang, Yibo Zhu, and Xin Jin. 2020. PipeSwitch: Fast Pipelined Context Switching for Deep Learning Applications. In 14th USENIX Symposium on Operating Systems Design and Implementation(OSDI \u201920)."},{"key":"e_1_3_2_1_8_1","unstructured":"Mandeep Baines Shruti Bhosale Vittorio Caggiano Naman Goyal Siddharth Goyal Myle Ott Benjamin Lefaudeux Vitaliy Liptchinsky Mike Rabbat Sam Sheiffer Anjali Sridhar and Min Xu. 2021. FairScale: A general purpose modular PyTorch library for high performance and large scale training."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Zhengda Bian Shenggui Li Wei Wang and Yang You. 2021. Online evolutionary batch size orchestration for scheduling deep learning workloads in GPU clusters. In SC.","DOI":"10.1145\/3458817.3480859"},{"key":"e_1_3_2_1_10_1","first-page":"19301","article-title":"Scalable Diverse Model Selection for Accessible Transfer Learning","volume":"34","author":"Bolya Daniel","year":"2021","unstructured":"Daniel Bolya, Rohit Mittapalli, and Judy Hoffman. 2021. Scalable Diverse Model Selection for Accessible Transfer Learning. Neurips 34 (2021), 19301\u201319312.","journal-title":"Neurips"},{"key":"e_1_3_2_1_11_1","first-page":"19301","article-title":"Scalable Diverse Model Selection for Accessible Transfer Learning","volume":"34","author":"Bolya Daniel","year":"2021","unstructured":"Daniel Bolya, Rohit Mittapalli, and Judy Hoffman. 2021. Scalable Diverse Model Selection for Accessible Transfer Learning. Neurips 34 (2021), 19301\u201319312.","journal-title":"Neurips"},{"key":"e_1_3_2_1_12_1","volume-title":"On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258","author":"Bommasani Rishi","year":"2021","unstructured":"Rishi Bommasani, Drew\u00a0A Hudson, Ehsan Adeli, Russ Altman, Simran Arora, Sydney von Arx, Michael\u00a0S Bernstein, Jeannette Bohg, Antoine Bosselut, Emma Brunskill, 2021. On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258 (2021)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Lukas Bossard Matthieu Guillaumin and Luc Van\u00a0Gool. 2014. Food-101 \u2013 Mining Discriminative Components with Random Forests. In ECCV.","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"e_1_3_2_1_14_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared\u00a0D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell Sandhini Agarwal Ariel Herbert-Voss Gretchen Krueger Tom Henighan Rewon Child Aditya Ramesh Daniel Ziegler Jeffrey Wu Clemens Winter Chris Hesse Mark Chen Eric Sigler Mateusz Litwin Scott Gray Benjamin Chess Jack Clark Christopher Berner Sam McCandlish Alec Radford Ilya Sutskever and Dario Amodei. 2020. Language Models are Few-Shot Learners. In Neurips."},{"key":"e_1_3_2_1_15_1","volume-title":"Lessons Learned from Three Container-Management Systems over a Decade. Queue","author":"Burns Brendan","year":"2016","unstructured":"Brendan Burns, Brian Grant, David Oppenheimer, Eric Brewer, and John Wilkes. 2016. Borg, Omega, and Kubernetes: Lessons Learned from Three Container-Management Systems over a Decade. Queue (2016)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3342195.3387555"},{"key":"e_1_3_2_1_17_1","volume-title":"Conv-Adapter: Exploring Parameter Efficient Transfer Learning for ConvNets. arXiv preprint arXiv:2208.07463","author":"Chen Hao","year":"2022","unstructured":"Hao Chen, Ran Tao, Han Zhang, Yidong Wang, Wei Ye, Jindong Wang, Guosheng Hu, and Marios Savvides. 2022. Conv-Adapter: Exploring Parameter Efficient Transfer Learning for ConvNets. arXiv preprint arXiv:2208.07463 (2022)."},{"key":"e_1_3_2_1_18_1","volume-title":"Net2net: Accelerating learning via knowledge transfer. arXiv preprint arXiv:1511.05641","author":"Chen Tianqi","year":"2015","unstructured":"Tianqi Chen, Ian Goodfellow, and Jonathon Shlens. 2015. Net2net: Accelerating learning via knowledge transfer. arXiv preprint arXiv:1511.05641 (2015)."},{"key":"e_1_3_2_1_19_1","volume-title":"Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang, Zhuohan Li, Zi Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph\u00a0E. Gonzalez, Ion Stoica, and Eric\u00a0P. Xing. 2023. Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics(NAACL \u201919)","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics(NAACL \u201919)."},{"key":"e_1_3_2_1_21_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01267"},{"key":"e_1_3_2_1_23_1","unstructured":"Saar Eliad Ido Hakimi Alon\u00a0De Jagger Mark Silberstein and Assaf Schuster. 2021. Fine-tuning giant neural networks on commodity hardware with automatic pipeline model parallelism. In USENIX ATC."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Wei Gao Peng Sun Yonggang Wen and Tianwei Zhang. 2022. Titan: a scheduler for foundation model fine-tuning workloads. In ACM SoCC. 348\u2013354.","DOI":"10.1145\/3542929.3563460"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-5409"},{"key":"e_1_3_2_1_26_1","volume-title":"Tiresias: A GPU Cluster Manager for Distributed Deep Learning. In NSDI.","author":"Gu Juncheng","year":"2019","unstructured":"Juncheng Gu, Mosharaf Chowdhury, Kang\u00a0G. Shin, Yibo Zhu, Myeongjae Jeon, Junjie Qian, Hongqiang Liu, and Chuanxiong Guo. 2019. Tiresias: A GPU Cluster Manager for Distributed Deep Learning. In NSDI."},{"key":"e_1_3_2_1_27_1","volume-title":"Parameter-efficient transfer learning with diff pruning. arXiv preprint arXiv:2012.07463","author":"Guo Demi","year":"2020","unstructured":"Demi Guo, Alexander\u00a0M Rush, and Yoon Kim. 2020. Parameter-efficient transfer learning with diff pruning. arXiv preprint arXiv:2012.07463 (2020)."},{"key":"e_1_3_2_1_28_1","volume-title":"Sommelier: Curating DNN Models for the Masses. In ICDM. 1876\u20131890.","author":"Guo Peizhen","year":"2022","unstructured":"Peizhen Guo, Bo Hu, and Wenjun Hu. 2022. Sommelier: Curating DNN Models for the Masses. In ICDM. 1876\u20131890."},{"key":"e_1_3_2_1_29_1","unstructured":"Mingcong Han Hanze Zhang Rong Chen and Haibo Chen. 2022. Microsecond-scale preemption for concurrent { GPU-accelerated}{ DNN} inferences. In OSDI. 539\u2013558."},{"key":"e_1_3_2_1_30_1","volume-title":"Visualizing and understanding the effectiveness of BERT. arXiv preprint arXiv:1908.05620","author":"Hao Yaru","year":"2019","unstructured":"Yaru Hao, Li Dong, Furu Wei, and Ke Xu. 2019. Visualizing and understanding the effectiveness of BERT. arXiv preprint arXiv:1908.05620 (2019)."},{"key":"e_1_3_2_1_31_1","unstructured":"Junxian He Chunting Zhou Xuezhe Ma Taylor Berg-Kirkpatrick and Graham Neubig. 2022. Towards a Unified View of Parameter-Efficient Transfer Learning. In ICLR."},{"key":"e_1_3_2_1_32_1","unstructured":"Neil Houlsby Andrei Giurgiu Stanislaw Jastrzebski Bruna Morrone Quentin De\u00a0Laroussilhe Andrea Gesmundo Mona Attariyan and Sylvain Gelly. 2019. Parameter-efficient transfer learning for NLP. In ICML. PMLR 2790\u20132799."},{"key":"e_1_3_2_1_33_1","volume-title":"Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685","author":"Hu J","year":"2021","unstructured":"Edward\u00a0J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)."},{"key":"e_1_3_2_1_34_1","unstructured":"Long-Kai Huang Junzhou Huang Yu Rong Qiang Yang and Ying Wei. 2022. Frustratingly easy transferability estimation. In ICML. 9201\u20139225."},{"key":"e_1_3_2_1_35_1","volume-title":"Gpipe: Efficient training of giant neural networks using pipeline parallelism. Neurips","author":"Huang Yanping","year":"2019","unstructured":"Yanping Huang, Youlong Cheng, Ankur Bapna, Orhan Firat, Dehao Chen, Mia Chen, HyoukJoong Lee, Jiquan Ngiam, Quoc\u00a0V Le, Yonghui Wu, 2019. Gpipe: Efficient training of giant neural networks using pipeline parallelism. Neurips (2019)."},{"key":"e_1_3_2_1_36_1","unstructured":"Changho Hwang Taehyun Kim Sunghyun Kim Jinwoo Shin and KyoungSoo Park. 2021. Elastic Resource Sharing for Distributed Deep Learning. In NSDI."},{"key":"e_1_3_2_1_37_1","volume-title":"Oobleck: Resilient Distributed Training of Large Models Using Pipeline Templates. In SOSP. 382\u2013395.","author":"Jang Insu","year":"2023","unstructured":"Insu Jang, Zhenning Yang, Zhen Zhang, Xin Jin, and Mosharaf Chowdhury. 2023. Oobleck: Resilient Distributed Training of Large Models Using Pipeline Templates. In SOSP. 382\u2013395."},{"key":"e_1_3_2_1_38_1","unstructured":"Myeongjae Jeon Shivaram Venkataraman Amar Phanishayee Junjie Qian Wencong Xiao and Fan Yang. 2019. Analysis of Large-Scale Multi-Tenant GPU Clusters for DNN Training Workloads. In USENIX ATC."},{"key":"e_1_3_2_1_39_1","volume-title":"Scaling laws for neural language models. arXiv preprint arXiv:2001.08361","author":"Kaplan Jared","year":"2020","unstructured":"Jared Kaplan, Sam McCandlish, Tom Henighan, Tom\u00a0B Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeffrey Wu, and Dario Amodei. 2020. Scaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020)."},{"key":"e_1_3_2_1_40_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma P","year":"2014","unstructured":"Diederik\u00a0P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_41_1","unstructured":"Alex Krizhevsky. 2009. Learning multiple layers of features from tiny images. Technical Report."},{"key":"e_1_3_2_1_42_1","unstructured":"Fan Lai Yinwei Dai Harsha\u00a0V. Madhyastha and Mosharaf Chowdhury. 2023. ModelKeeper: Accelerating DNN Training via Automated Training Warmup. In NSDI."},{"key":"e_1_3_2_1_43_1","volume-title":"The power of scale for parameter-efficient prompt tuning. arXiv preprint arXiv:2104.08691","author":"Lester Brian","year":"2021","unstructured":"Brian Lester, Rami Al-Rfou, and Noah Constant. 2021. The power of scale for parameter-efficient prompt tuning. arXiv preprint arXiv:2104.08691 (2021)."},{"key":"e_1_3_2_1_44_1","unstructured":"Conglong Li Minjia Zhang and Yuxiong He. 2022. The Stability-Efficiency Dilemma: Investigating Sequence Length Warmup for Training GPT Models. In Neurips."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Junyi Li Tianyi Tang Gaole He Jinhao Jiang Xiaoxuan Hu Puzhao Xie Zhipeng Chen Zhuohao Yu Wayne\u00a0Xin Zhao and Ji-Rong Wen. 2021. TextBox: A Unified Modularized and Extensible Framework for Text Generation. In ACL. 30\u201339.","DOI":"10.18653\/v1\/2021.acl-demo.4"},{"volume-title":"Evaluating parameter-efficient transfer learning approaches on sure benchmark for speech understanding","author":"Li Yingting","key":"e_1_3_2_1_46_1","unstructured":"Yingting Li, Ambuj Mehrish, Rishabh Bhardwaj, Navonil Majumder, Bo Cheng, Shuai Zhao, Amir Zadeh, Rada Mihalcea, and Soujanya Poria. 2023. Evaluating parameter-efficient transfer learning approaches on sure benchmark for speech understanding. In ICASSP. IEEE, 1\u20135."},{"key":"e_1_3_2_1_47_1","volume-title":"Multi-task deep neural networks for natural language understanding. arXiv preprint arXiv:1901.11504","author":"Liu Xiaodong","year":"2019","unstructured":"Xiaodong Liu, Pengcheng He, Weizhu Chen, and Jianfeng Gao. 2019. Multi-task deep neural networks for natural language understanding. arXiv preprint arXiv:1901.11504 (2019)."},{"key":"e_1_3_2_1_48_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_1_49_1","volume-title":"Full Parameter Fine-tuning for Large Language Models with Limited Resources. arXiv preprint arXiv:2306.09782","author":"Lv Kai","year":"2023","unstructured":"Kai Lv, Yuqing Yang, Tengxiao Liu, Qinghui Gao, Qipeng Guo, and Xipeng Qiu. 2023. Full Parameter Fine-tuning for Large Language Models with Limited Resources. arXiv preprint arXiv:2306.09782 (2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"Themis: Fair and Efficient GPU Cluster Scheduling. In NSDI.","author":"Mahajan Kshiteej","year":"2020","unstructured":"Kshiteej Mahajan, Arjun Balasubramanian, Arjun Singhvi, Shivaram Venkataraman, Aditya Akella, Amar Phanishayee, and Shuchi Chawla. 2020. Themis: Fair and Efficient GPU Cluster Scheduling. In NSDI."},{"key":"e_1_3_2_1_51_1","volume-title":"Unipelt: A unified framework for parameter-efficient language model tuning. arXiv preprint arXiv:2110.07577","author":"Mao Yuning","year":"2021","unstructured":"Yuning Mao, Lambert Mathias, Rui Hou, Amjad Almahairi, Hao Ma, Jiawei Han, Wen-tau Yih, and Madian Khabsa. 2021. Unipelt: A unified framework for parameter-efficient language model tuning. arXiv preprint arXiv:2110.07577 (2021)."},{"key":"e_1_3_2_1_52_1","volume-title":"An empirical model of large-batch training. arXiv preprint arXiv:1812.06162","author":"McCandlish Sam","year":"2018","unstructured":"Sam McCandlish, Jared Kaplan, Dario Amodei, and OpenAI\u00a0Dota Team. 2018. An empirical model of large-batch training. arXiv preprint arXiv:1812.06162 (2018)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Nasrin Mostafazadeh Nathanael Chambers Xiaodong He Devi Parikh Dhruv Batra Lucy Vanderwende Pushmeet Kohli and James Allen. 2016. A Corpus and Cloze Evaluation for Deeper Understanding of Commonsense Stories. In ACL. 839\u2013849.","DOI":"10.18653\/v1\/N16-1098"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Deepak Narayanan Mohammad Shoeybi Jared Casper Patrick LeGresley Mostofa Patwary Vijay Korthikanti Dmitri Vainbrand Prethvi Kashinkunti Julie Bernauer Bryan Catanzaro Amar Phanishayee and Matei Zaharia. 2021. Efficient large-scale language model training on GPU clusters using megatron-LM. In SC.","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_2_1_55_1","volume-title":"International Conference on Machine Learning. PMLR, 7294\u20137305","author":"Nguyen Cuong","year":"2020","unstructured":"Cuong Nguyen, Tal Hassner, Matthias Seeger, and Cedric Archambeau. 2020. Leep: A new measure to evaluate transferability of learned representations. In International Conference on Machine Learning. PMLR, 7294\u20137305."},{"key":"e_1_3_2_1_56_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. arxiv:2303.08774\u00a0[cs.CL]"},{"key":"e_1_3_2_1_57_1","volume-title":"Parameter-efficient image-to-video transfer learning. arXiv e-prints","author":"Pan Junting","year":"2022","unstructured":"Junting Pan, Ziyi Lin, Xiatian Zhu, Jing Shao, and Hongsheng Li. 2022. Parameter-efficient image-to-video transfer learning. arXiv e-prints (2022), arXiv\u20132206."},{"volume-title":"PyTorch: An Imperative Style","author":"Paszke Adam","key":"e_1_3_2_1_58_1","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas Kopf, Edward Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: An Imperative Style, High-Performance Deep Learning Library. In Neurips, H.\u00a0Wallach, H.\u00a0Larochelle, A.\u00a0Beygelzimer, F.\u00a0d\u2019Alch\u00e9 Buc, E.\u00a0Fox, and R.\u00a0Garnett (Eds.). 8024\u20138035."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3190508.3190517"},{"key":"e_1_3_2_1_60_1","volume-title":"AdapterFusion: Non-destructive task composition for transfer learning. arXiv preprint arXiv:2005.00247","author":"Pfeiffer Jonas","year":"2020","unstructured":"Jonas Pfeiffer, Aishwarya Kamath, Andreas R\u00fcckl\u00e9, Kyunghyun Cho, and Iryna Gurevych. 2020. AdapterFusion: Non-destructive task composition for transfer learning. arXiv preprint arXiv:2005.00247 (2020)."},{"key":"e_1_3_2_1_61_1","volume-title":"What to pre-train on? efficient intermediate task selection. arXiv preprint arXiv:2104.08247","author":"Poth Clifton","year":"2021","unstructured":"Clifton Poth, Jonas Pfeiffer, Andreas R\u00fcckl\u00e9, and Iryna Gurevych. 2021. What to pre-train on? efficient intermediate task selection. arXiv preprint arXiv:2104.08247 (2021)."},{"key":"e_1_3_2_1_62_1","volume-title":"Pollux: Co-adaptive Cluster Scheduling for Goodput-Optimized Deep Learning. In OSDI.","author":"Qiao Aurick","year":"2021","unstructured":"Aurick Qiao, Sang\u00a0Keun Choe, Suhas\u00a0Jayaram Subramanya, Willie Neiswanger, Qirong Ho, Hao Zhang, Gregory\u00a0R. Ganger, and Eric\u00a0P. Xing. 2021. Pollux: Co-adaptive Cluster Scheduling for Goodput-Optimized Deep Learning. In OSDI."},{"key":"e_1_3_2_1_63_1","unstructured":"Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark 2021. Learning transferable visual models from natural language supervision. In ICML. 8748\u20138763."},{"key":"e_1_3_2_1_64_1","volume-title":"International Conference on Machine Learning. PMLR, 8748\u20138763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748\u20138763."},{"key":"e_1_3_2_1_65_1","volume-title":"Language models are unsupervised multitask learners. OpenAI blog 1, 8","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, Ilya Sutskever, 2019. Language models are unsupervised multitask learners. OpenAI blog 1, 8 (2019), 9."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00266"},{"key":"e_1_3_2_1_68_1","volume-title":"ImageNet Large Scale Visual Recognition Challenge. IJCV","author":"Russakovsky Olga","year":"2015","unstructured":"Olga Russakovsky, Jia Deng, Hao Su, Jonathan Krause, Sanjeev Satheesh, Sean Ma, Zhiheng Huang, Andrej Karpathy, Aditya Khosla, Michael Bernstein, Alexander\u00a0C. Berg, and Li Fei-Fei. 2015. ImageNet Large Scale Visual Recognition Challenge. IJCV (2015), 211\u2013252."},{"key":"e_1_3_2_1_69_1","volume-title":"Multitask prompted training enables zero-shot task generalization. arXiv preprint arXiv:2110.08207","author":"Sanh Victor","year":"2021","unstructured":"Victor Sanh, Albert Webson, Colin Raffel, Stephen\u00a0H Bach, Lintang Sutawika, Zaid Alyafeai, Antoine Chaffin, Arnaud Stiegler, Teven\u00a0Le Scao, Arun Raja, 2021. Multitask prompted training enables zero-shot task generalization. arXiv preprint arXiv:2110.08207 (2021)."},{"key":"e_1_3_2_1_70_1","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053","author":"Shoeybi Mohammad","year":"2019","unstructured":"Mohammad Shoeybi, Mostofa Patwary, Raul Puri, Patrick LeGresley, Jared Casper, and Bryan Catanzaro. 2019. Megatron-lm: Training multi-billion parameter language models using model parallelism. arXiv preprint arXiv:1909.08053 (2019)."},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"crossref","unstructured":"Reza Shokri Marco Stronati Congzheng Song and Vitaly Shmatikov. 2017. Membership inference attacks against machine learning models. In SP. 3\u201318.","DOI":"10.1109\/SP.2017.41"},{"key":"e_1_3_2_1_72_1","volume-title":"How to train your vit? data, augmentation, and regularization in vision transformers. arXiv preprint arXiv:2106.10270","author":"Steiner Andreas","year":"2021","unstructured":"Andreas Steiner, Alexander Kolesnikov, Xiaohua Zhai, Ross Wightman, Jakob Uszkoreit, and Lucas Beyer. 2021. How to train your vit? data, augmentation, and regularization in vision transformers. arXiv preprint arXiv:2106.10270 (2021)."},{"key":"e_1_3_2_1_73_1","unstructured":"Yusheng Su Xiaozhi Wang Yujia Qin Chi-Min Chan Yankai Lin Huadong Wang Kaiyue Wen Zhiyuan Liu Peng Li Juanzi Li Lei Hou Maosong Sun and Jie Zhou. 2022. On Transferability of Prompt Tuning for Natural Language Processing. In NAACL. 3949\u20133969."},{"key":"e_1_3_2_1_74_1","volume-title":"Vl-adapter: Parameter-efficient transfer learning for vision-and-language tasks. In CVPR. 5227\u20135237.","author":"Sung Yi-Lin","year":"2022","unstructured":"Yi-Lin Sung, Jaemin Cho, and Mohit Bansal. 2022. Vl-adapter: Parameter-efficient transfer learning for vision-and-language tasks. In CVPR. 5227\u20135237."},{"key":"e_1_3_2_1_75_1","volume-title":"Bamboo: Making Preemptible Instances Resilient for Affordable Training of Large { DNNs}. In NSDI. 497\u2013513.","author":"Thorpe John","year":"2023","unstructured":"John Thorpe, Pengzhan Zhao, Jonathan Eyolfson, Yifan Qiao, Zhihao Jia, Minjia Zhang, Ravi Netravali, and Guoqing\u00a0Harry Xu. 2023. Bamboo: Making Preemptible Instances Resilient for Affordable Training of Large { DNNs}. In NSDI. 497\u2013513."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"crossref","unstructured":"Anh\u00a0T Tran Cuong\u00a0V Nguyen and Tal Hassner. 2019. Transferability and hardness of supervised classification tasks. In ICCV. 1395\u20131405.","DOI":"10.1109\/ICCV.2019.00148"},{"key":"e_1_3_2_1_77_1","first-page":"7852","article-title":"On the theory of transfer learning: The importance of task diversity","volume":"33","author":"Tripuraneni Nilesh","year":"2020","unstructured":"Nilesh Tripuraneni, Michael Jordan, and Chi Jin. 2020. On the theory of transfer learning: The importance of task diversity. Neurips 33 (2020), 7852\u20137862.","journal-title":"Neurips"},{"volume-title":"Using adapters to overcome catastrophic forgetting in end-to-end automatic speech recognition","author":"Vander\u00a0Eeckt Steven","key":"e_1_3_2_1_78_1","unstructured":"Steven Vander\u00a0Eeckt and Hugo Van\u00a0Hamme. 2023. Using adapters to overcome catastrophic forgetting in end-to-end automatic speech recognition. In ICASSP. IEEE, 1\u20135."},{"key":"e_1_3_2_1_79_1","volume-title":"The shape of learning curves: a review. TPAMI","author":"Viering Tom","year":"2022","unstructured":"Tom Viering and Marco Loog. 2022. The shape of learning curves: a review. TPAMI (2022)."},{"key":"e_1_3_2_1_80_1","volume-title":"Exploring and predicting transferability across NLP tasks. arXiv preprint arXiv:2005.00770","author":"Vu Tu","year":"2020","unstructured":"Tu Vu, Tong Wang, Tsendsuren Munkhdalai, Alessandro Sordoni, Adam Trischler, Andrew Mattarella-Micke, Subhransu Maji, and Mohit Iyyer. 2020. Exploring and predicting transferability across NLP tasks. arXiv preprint arXiv:2005.00770 (2020)."},{"key":"e_1_3_2_1_81_1","volume-title":"GLUE: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461","author":"Wang Alex","year":"2018","unstructured":"Alex Wang, Amanpreet Singh, Julian Michael, Felix Hill, Omer Levy, and Samuel\u00a0R Bowman. 2018. GLUE: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461 (2018)."},{"key":"e_1_3_2_1_82_1","volume-title":"Proceedings of Machine Learning and Systems","author":"Wasay Abdul","year":"2020","unstructured":"Abdul Wasay, Brian Hentschel, Yuze Liao, Sanyuan Chen, and Stratos Idreos. 2020. Mothernets: Rapid deep ensemble learning. Proceedings of Machine Learning and Systems (2020), 199\u2013215."},{"key":"e_1_3_2_1_83_1","volume-title":"International conference on machine learning. PMLR, 564\u2013572","author":"Wei Tao","year":"2016","unstructured":"Tao Wei, Changhu Wang, Yong Rui, and Chang\u00a0Wen Chen. 2016. Network morphism. In International conference on machine learning. PMLR, 564\u2013572."},{"key":"e_1_3_2_1_84_1","volume-title":"When to Use Multi-Task Learning vs Intermediate Fine-Tuning for Pre-Trained Encoder Transfer Learning. arXiv preprint arXiv:2205.08124","author":"Weller Orion","year":"2022","unstructured":"Orion Weller, Kevin Seppi, and Matt Gardner. 2022. When to Use Multi-Task Learning vs Intermediate Fine-Tuning for Pre-Trained Encoder Transfer Learning. arXiv preprint arXiv:2205.08124 (2022)."},{"key":"e_1_3_2_1_85_1","volume-title":"Transformers: State-of-the-Art Natural Language Processing. In EMNLP. 38\u201345.","author":"Wolf Thomas","year":"2020","unstructured":"Thomas Wolf, Lysandre Debut, Victor Sanh, Julien Chaumond, Clement Delangue, Anthony Moi, Pierric Cistac, Tim Rault, R\u00e9mi Louf, Morgan Funtowicz, Joe Davison, Sam Shleifer, Patrick von Platen, Clara Ma, Yacine Jernite, Julien Plu, Canwen Xu, Teven\u00a0Le Scao, Sylvain Gugger, Mariama Drame, Quentin Lhoest, and Alexander\u00a0M. Rush. 2020. Transformers: State-of-the-Art Natural Language Processing. In EMNLP. 38\u201345."},{"key":"e_1_3_2_1_86_1","volume-title":"Gandiva: Introspective Cluster Scheduling for Deep Learning. In OSDI.","author":"Xiao Wencong","year":"2018","unstructured":"Wencong Xiao, Romil Bhardwaj, Ramachandran Ramjee, Muthian Sivathanu, Nipun Kwatra, Zhenhua Han, Pratyush Patel, Xuan Peng, Hanyu Zhao, Quanlu Zhang, Fan Yang, and Lidong Zhou. 2018. Gandiva: Introspective Cluster Scheduling for Deep Learning. In OSDI."},{"key":"e_1_3_2_1_87_1","volume-title":"Crossfit: A few-shot learning challenge for cross-task generalization in nlp. arXiv preprint arXiv:2104.08835","author":"Ye Qinyuan","year":"2021","unstructured":"Qinyuan Ye, Bill\u00a0Yuchen Lin, and Xiang Ren. 2021. Crossfit: A few-shot learning challenge for cross-task generalization in nlp. arXiv preprint arXiv:2104.08835 (2021)."},{"key":"e_1_3_2_1_88_1","volume-title":"How transferable are features in deep neural networks?Neurips 27","author":"Yosinski Jason","year":"2014","unstructured":"Jason Yosinski, Jeff Clune, Yoshua Bengio, and Hod Lipson. 2014. How transferable are features in deep neural networks?Neurips 27 (2014)."},{"key":"e_1_3_2_1_89_1","volume-title":"International Conference on Machine Learning. PMLR, 12133\u201312143","author":"You Kaichao","year":"2021","unstructured":"Kaichao You, Yong Liu, Jianmin Wang, and Mingsheng Long. 2021. Logme: Practical assessment of pre-trained models for transfer learning. In International Conference on Machine Learning. PMLR, 12133\u201312143."},{"key":"e_1_3_2_1_90_1","volume-title":"Towards a Unified View on Visual Parameter-Efficient Transfer Learning. arXiv preprint arXiv:2210.00788","author":"Yu XB","year":"2022","unstructured":"Bruce\u00a0XB Yu, Jianlong Chang, Lingbo Liu, Qi Tian, and Chang\u00a0Wen Chen. 2022. Towards a Unified View on Visual Parameter-Efficient Transfer Learning. arXiv preprint arXiv:2210.00788 (2022)."},{"key":"e_1_3_2_1_91_1","volume-title":"Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. arXiv preprint arXiv:2106.10199","author":"Zaken Elad\u00a0Ben","year":"2021","unstructured":"Elad\u00a0Ben Zaken, Shauli Ravfogel, and Yoav Goldberg. 2021. Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. arXiv preprint arXiv:2106.10199 (2021)."},{"key":"e_1_3_2_1_92_1","volume-title":"One Network","author":"Zeng Guangtao","year":"2023","unstructured":"Guangtao Zeng, Peiyuan Zhang, and Wei Lu. 2023. One Network, Many Masks: Towards More Parameter-Efficient Transfer Learning. arXiv preprint arXiv:2305.17682 (2023)."},{"key":"e_1_3_2_1_93_1","doi-asserted-by":"publisher","DOI":"10.1145\/3127479.3127490"},{"key":"e_1_3_2_1_94_1","volume-title":"Masking as an efficient alternative to finetuning for pretrained language models. arXiv preprint arXiv:2004.12406","author":"Zhao Mengjie","year":"2020","unstructured":"Mengjie Zhao, Tao Lin, Fei Mi, Martin Jaggi, and Hinrich Sch\u00fctze. 2020. Masking as an efficient alternative to finetuning for pretrained language models. arXiv preprint arXiv:2004.12406 (2020)."},{"key":"e_1_3_2_1_95_1","volume-title":"Shockwave: Fair and Efficient Cluster Scheduling for Dynamic Adaptation in Machine Learning. In NSDI.","author":"Zheng Pengfei","year":"2023","unstructured":"Pengfei Zheng, Rui Pan, Tarannum Khan, Shivaram Venkataraman, and Aditya Akella. 2023. Shockwave: Fair and Efficient Cluster Scheduling for Dynamic Adaptation in Machine Learning. In NSDI."}],"event":{"name":"ICS '24: 2024 International Conference on Supercomputing","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"],"location":"Kyoto Japan","acronym":"ICS '24"},"container-title":["Proceedings of the 38th ACM International Conference on Supercomputing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3650200.3656599","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3650200.3656599","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T15:24:18Z","timestamp":1755876258000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3650200.3656599"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,30]]},"references-count":95,"alternative-id":["10.1145\/3650200.3656599","10.1145\/3650200"],"URL":"https:\/\/doi.org\/10.1145\/3650200.3656599","relation":{},"subject":[],"published":{"date-parts":[[2024,5,30]]},"assertion":[{"value":"2024-06-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}