{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T13:04:11Z","timestamp":1780664651521,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,26]],"date-time":"2026-04-26T00:00:00Z","timestamp":1777161600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,27]]},"DOI":"10.1145\/3767295.3769389","type":"proceedings-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T20:20:04Z","timestamp":1777062004000},"page":"2292-2307","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Efficient Multimodal Serving via Module Multiplexing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5689-382X","authenticated-orcid":false,"given":"Zicong","family":"Hong","sequence":"first","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7738-640X","authenticated-orcid":false,"given":"Yuyan","family":"Chen","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Zhuhai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8308-3827","authenticated-orcid":false,"given":"Haoyue","family":"Zhang","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5303-0700","authenticated-orcid":false,"given":"Peng","family":"Li","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4430-7904","authenticated-orcid":false,"given":"Wuhui","family":"Chen","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, Zhuhai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9831-2202","authenticated-orcid":false,"given":"Song","family":"Guo","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5420-583X","authenticated-orcid":false,"given":"Xiaowei","family":"Shen","sequence":"additional","affiliation":[{"name":"MetaX, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,26]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"VQA: Visual Question Answering. In International Conference on Computer Vision.","author":"Antol Stanislaw","year":"2015","unstructured":"Stanislaw Antol, Aishwarya Agrawal, Jiasen Lu, Margaret Mitchell, Dhruv Batra, C. Lawrence Zitnick, and Devi Parikh. 2015. VQA: Visual Question Answering. In International Conference on Computer Vision."},{"key":"e_1_3_2_1_2_1","unstructured":"Jiuhai Chen Zhiyang Xu Xichen Pan Shusheng Yang Can Qin An Yan Honglu Zhou Zeyuan Chen Tianyi Zhou Silvio Savarese Le Xue Caiming Xiong and Ran Xu. 2025. BLIP3o-NEXT: A Next-Generation Multimodal Foundation Model. https:\/\/jiuhaichen.github.io\/BLIP3oNEXT.github.io\/"},{"key":"e_1_3_2_1_3_1","volume-title":"Topological Planning with Transformers for Vision-and-Language Navigation. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 11271\u201311281","author":"Chen Kevin","year":"2021","unstructured":"Kevin Chen, Junshen K. Chen, Jo Chuang, Marynel V\u00e1zquez, and Silvio Savarese. 2021. Topological Planning with Transformers for Vision-and-Language Navigation. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 11271\u201311281."},{"key":"e_1_3_2_1_4_1","volume-title":"European Conference on Computer Vision.","author":"Chen Liang","year":"2024","unstructured":"Liang Chen, Haozhe Zhao, Tianyu Liu, Shuai Bai, Junyang Lin, Chang Zhou, and Baobao Chang. 2024. An Image is Worth 1\/2 Tokens After Layer 2: Plug-and-Play Inference Acceleration for Large Vision-Language Models. In European Conference on Computer Vision."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2872362.2872368"},{"key":"e_1_3_2_1_6_1","volume-title":"International Conference on Learning Representations.","author":"Chen Wenhu","unstructured":"Wenhu Chen, Ming-Wei Chang, Eva Schlinger, William Yang Wang, and William W. Cohen. 2021. Open Question Answering over Tables and Text. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581784.3607060"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185\u201324198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al. 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185\u201324198."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3337821.3337892"},{"key":"e_1_3_2_1_10_1","volume-title":"Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In USENIX Annual Technical Conference. 199\u2013216","author":"Choi Seungbeom","year":"2022","unstructured":"Seungbeom Choi, Sunho Lee, Yeonjae Kim, Jongse Park, Youngjin Kwon, and Jaehyuk Huh. 2022. Serving Heterogeneous Machine Learning Models on Multi-GPU Servers with Spatio-Temporal Sharing. In USENIX Annual Technical Conference. 199\u2013216."},{"key":"e_1_3_2_1_11_1","unstructured":"Hyung Won Chung Le Hou Shayne Longpre Barret Zoph Yi Tay William Fedus Yunxuan Li Xuezhi Wang Mostafa Dehghani Siddhartha Brahma Albert Webson Shixiang Shane Gu Zhuyun Dai Mirac Suzgun Xinyun Chen Aakanksha Chowdhery Alex Castro-Ros Marie Pellat Kevin Robinson Dasha Valter Sharan Narang Gaurav Mishra Adams Yu Vincent Zhao Yanping Huang Andrew Dai Hongkun Yu Slav Petrov Ed H. Chi Jeff Dean Jacob Devlin Adam Roberts Denny Zhou Quoc V. Le and Jason Wei. 2022. Scaling Instruction-Finetuned Language Models. arXiv:2210.11416 [cs.LG] https:\/\/arxiv.org\/abs\/2210.11416"},{"key":"e_1_3_2_1_12_1","volume-title":"Clipper: A Low-Latency Online Prediction Serving System. In USENIX Symposium on Networked Systems Design and Implementation. 613\u2013627","author":"Crankshaw Daniel","year":"2017","unstructured":"Daniel Crankshaw, Xin Wang, Guilio Zhou, Michael J. Franklin, Joseph E. Gonzalez, and Ion Stoica. 2017. Clipper: A Low-Latency Online Prediction Serving System. In USENIX Symposium on Networked Systems Design and Implementation. 613\u2013627."},{"key":"e_1_3_2_1_13_1","volume-title":"DVABatch: Diversity-aware Multi-Entry Multi-Exit Batching for Efficient Processing of DNN Services on GPUs. In USENIX Annual Technical Conference. 183\u2013198","author":"Cui Weihao","year":"2022","unstructured":"Weihao Cui, Han Zhao, Quan Chen, Hao Wei, Zirui Li, Deze Zeng, Chao Li, and Minyi Guo. 2022. DVABatch: Diversity-aware Multi-Entry Multi-Exit Batching for Efficient Processing of DNN Services on GPUs. In USENIX Annual Technical Conference. 183\u2013198."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). 4171\u20134186."},{"key":"e_1_3_2_1_15_1","volume-title":"ACM Symposium on Cloud Computing. 492\u2013506","author":"Dhakal Aditya","unstructured":"Aditya Dhakal, Sameer G Kulkarni, and K. K. Ramakrishnan. 2020. GSLICE: controlled spatial sharing of GPUs for a scalable inference platform. In ACM Symposium on Cloud Computing. 492\u2013506."},{"key":"e_1_3_2_1_16_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ICLR","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. ICLR (2021)."},{"key":"e_1_3_2_1_17_1","volume-title":"MuxServe: Flexible Spatial-Temporal Multiplexing for Multiple LLM Serving. In Forty-first International Conference on Machine Learning.","author":"Duan Jiangfei","year":"2024","unstructured":"Jiangfei Duan, Runyu Lu, Haojie Duanmu, Xiuhong Li, Xingcheng Zhang, Dahua Lin, Ion Stoica, and Hao Zhang. 2024. MuxServe: Flexible Spatial-Temporal Multiplexing for Multiple LLM Serving. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437801.3441578"},{"key":"e_1_3_2_1_19_1","volume-title":"Optimus: Accelerating Large-Scale Multi-Modal LLM Training by Bubble Exploitation. arXiv:2408.03505 [cs.CL] https:\/\/arxiv.org\/abs\/2408.03505","author":"Feng Weiqi","year":"2024","unstructured":"Weiqi Feng, Yangrui Chen, Shaoyu Wang, Yanghua Peng, Haibin Lin, and Minlan Yu. 2024. Optimus: Accelerating Large-Scale Multi-Modal LLM Training by Bubble Exploitation. arXiv:2408.03505 [cs.CL] https:\/\/arxiv.org\/abs\/2408.03505"},{"key":"e_1_3_2_1_20_1","volume-title":"Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering. In Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Goyal Yash","year":"2017","unstructured":"Yash Goyal, Tejas Khot, Douglas Summers-Stay, Dhruv Batra, and Devi Parikh. 2017. Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering. In Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_1_21_1","unstructured":"Jiaming Han Renrui Zhang Wenqi Shao Peng Gao Peng Xu Han Xiao Kaipeng Zhang Chris Liu Song Wen Ziyu Guo Xudong Lu Shuai Ren Yafei Wen Xiaoxin Chen Xiangyu Yue Hongsheng Li and Yu Qiao. 2023. ImageBind-LLM: Multi-modality Instruction Tuning. arXiv:2309.03905 [cs.MM] https:\/\/arxiv.org\/abs\/2309.03905"},{"key":"e_1_3_2_1_22_1","volume-title":"VLNBERT: A Recurrent Vision-and-Language BERT for Navigation. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 1643\u20131653","author":"Hong Yicong","year":"2021","unstructured":"Yicong Hong, Qi Wu, Yuankai Qi, Cristian Rodriguez-Opazo, and Stephen Gould. 2021. VLNBERT: A Recurrent Vision-and-Language BERT for Navigation. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 1643\u20131653."},{"key":"e_1_3_2_1_23_1","volume-title":"DISTMM: Accelerating Distributed Multimodal Model Training. In Networked Systems Design and Implementation. 1157\u20131171.","author":"Huang Jun","year":"2024","unstructured":"Jun Huang, Zhen Zhang, Shuai Zheng, Feng Qin, and Yida Wang. 2024. DISTMM: Accelerating Distributed Multimodal Model Training. In Networked Systems Design and Implementation. 1157\u20131171."},{"key":"e_1_3_2_1_24_1","volume-title":"MADTP: Multimodal Alignment-Guided Dynamic Token Pruning for Accelerating Vision-Language Transformer. IEEE Conference on Computer Vision and Pattern Recognition","author":"Jianjian Cao","year":"2024","unstructured":"Cao Jianjian, Ye Peng, Li Shengze, Yu Chong, Tang Yansong, Lu Jiwen, and Chen Tao. 2024. MADTP: Multimodal Alignment-Guided Dynamic Token Pruning for Accelerating Vision-Language Transformer. IEEE Conference on Computer Vision and Pattern Recognition (2024)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.302"},{"key":"e_1_3_2_1_26_1","unstructured":"Yizhang Jin Jian Li Yexin Liu Tianjun Gu Kai Wu Zhengkai Jiang Muyang He Bo Zhao Xin Tan Zhenye Gan Yabiao Wang Chengjie Wang and Lizhuang Ma. 2024. Efficient Multimodal Large Language Models: A Survey. arXiv:2405.10739 [cs.CV]"},{"key":"e_1_3_2_1_27_1","volume-title":"OpenVLA: An Open-Source Vision-Language-Action Model. arXiv preprint arXiv:2406.09246","author":"Kim Moo Jin","year":"2024","unstructured":"Moo Jin Kim, Karl Pertsch, Siddharth Karamcheti, Ted Xiao, Ashwin Balakrishna, Suraj Nair, Rafael Rafailov, Ethan Foster, Grace Lam, Pannag Sanketi, Quan Vuong, Thomas Kollar, Benjamin Burchfiel, Russ Tedrake, Dorsa Sadigh, Sergey Levine, Percy Liang, and Chelsea Finn. 2024. OpenVLA: An Open-Source Vision-Language-Action Model. arXiv preprint arXiv:2406.09246 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"International Conference on Machine Learning. 5583\u20135594","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision. In International Conference on Machine Learning. 5583\u20135594."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Yejin Lee Anna Sun Basil Hosmer Bilge Acun Can Balioglu Changhan Wang Charles David Hernandez Christian Puhrsch Daniel Haziza Driss Guessous Francisco Massa Jacob Kahn Jeffrey Wan Jeremy Reizenstein Jiaqi Zhai Joe Isaacson Joel Schlosser Juan Pino Kaushik Ram Sadagopan Leonid Shamis Linjian Ma Min-Jae Hwang Mingda Chen Mostafa Elhoushi Pedro Rodriguez Ram Pasunuru Scott Yih Sravya Popuri Xing Liu and Carole-Jean Wu. 2025. Characterizing and Efficiently Accelerating Multimodal Generation Model Inference. arXiv:2410.00215 [cs.LG] https:\/\/arxiv.org\/abs\/2410.00215","DOI":"10.1109\/MM.2025.3596539"},{"key":"e_1_3_2_1_31_1","unstructured":"Junnan Li Dongxu Li Silvio Savarese and Steven Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. arXiv:2301.12597 [cs.CV]"},{"key":"e_1_3_2_1_32_1","volume-title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. arXiv:2201.12086 [cs.CV]","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. arXiv:2201.12086 [cs.CV]"},{"key":"e_1_3_2_1_33_1","volume-title":"Abdelaali Hassaine, Rema Ramakrishnan, Dexter Canoy, Yajie Zhu, Kazem Rahimi, and Gholamreza Salimi-Khorshidi.","author":"Li Yikuan","year":"2020","unstructured":"Yikuan Li, Shishir Rao, Jos\u00e9 Roberto Ayala Solares, Abdelaali Hassaine, Rema Ramakrishnan, Dexter Canoy, Yajie Zhu, Kazem Rahimi, and Gholamreza Salimi-Khorshidi. 2020. BEHRT: transformer for electronic health records. Scientific reports 10, 1 (2020), 7155."},{"key":"e_1_3_2_1_34_1","unstructured":"Tsung-Yi Lin Michael Maire Serge Belongie Lubomir Bourdev Ross Girshick James Hays Pietro Perona Deva Ramanan C. Lawrence Zitnick and Piotr Doll\u00e1r. 2015. Microsoft COCO: Common Objects in Context. arXiv:1405.0312 [cs.CV] https:\/\/arxiv.org\/abs\/1405.0312"},{"key":"e_1_3_2_1_35_1","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong Jae Lee. 2023. Visual Instruction Tuning. In Neural Information Processing Systems."},{"key":"e_1_3_2_1_36_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. 2025. NVIDIA Multi-Instance GPU (MIG). Retrieved March 15, 2025 from https:\/\/docs.nvidia.com\/datacenter\/tesla\/mig-user-guide\/index.html"},{"key":"e_1_3_2_1_37_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. 2025. NVIDIA Multi-Process Service (MPS). Retrieved March 15, 2025 from https:\/\/docs.nvidia.com\/deploy\/mps\/index.html"},{"key":"e_1_3_2_1_38_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. 2025. NVIDIA Nsight Systems. Retrieved March 15, 2025 from https:\/\/developer.nvidia.com\/nsight-systems"},{"key":"e_1_3_2_1_39_1","volume-title":"Retrieved","author":"NVIDIA.","year":"2025","unstructured":"NVIDIA. 2025. NVIDIA Triton Inference Server. Retrieved March 15, 2025 from https:\/\/www.nvidia.com\/en-us\/ai-data-science\/products\/triton-inference-server\/"},{"key":"e_1_3_2_1_40_1","volume-title":"Tensorflow-serving: Flexible, high-performance ml serving. arXiv preprint arXiv:1712.06139","author":"Olston Christopher","year":"2017","unstructured":"Christopher Olston, Noah Fiedel, Kiril Gorovoy, Jeremiah Harmsen, Li Lao, Fangwei Li, Vinu Rajashekhar, Sukriti Ramesh, and Jordan Soyke. 2017. Tensorflow-serving: Flexible, high-performance ml serving. arXiv preprint arXiv:1712.06139 (2017)."},{"key":"e_1_3_2_1_41_1","volume-title":"PyTorch: an imperative style, high-performance deep learning library","author":"Paszke Adam","unstructured":"Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas K\u00f6pf, Edward Yang, Zach DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: an imperative style, high-performance deep learning library. Curran Associates Inc."},{"key":"e_1_3_2_1_42_1","volume-title":"Splitwise: Efficient generative LLM inference using phase splitting. In ISCA.","author":"Patel Pratyush","year":"2024","unstructured":"Pratyush Patel, Esha Choukse, Chaojie Zhang, Aashaka Shah, Inigo Goiri, Saeed Maleki, and Ricardo Bianchini. 2024. Splitwise: Efficient generative LLM inference using phase splitting. In ISCA."},{"key":"e_1_3_2_1_43_1","unstructured":"Haoran Qiu Anish Biswas Zihan Zhao Jayashree Mohan Alind Khare Esha Choukse \u00cd\u00f1igo Goiri Zeyu Zhang Haiying Shen Chetan Bansal Ramachandran Ramjee and Rodrigo Fonseca. 2025. Mod-Serve: Scalable and Resource-Efficient Large Multimodal Model Serving. arXiv:2502.00937 [cs.DC] https:\/\/arxiv.org\/abs\/2502.00937"},{"key":"e_1_3_2_1_44_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. arXiv:2103.00020 [cs.CV]"},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning. 31292\u201331311","author":"Shi Dachuan","year":"2023","unstructured":"Dachuan Shi, Chaofan Tao, Ying Jin, Zhendong Yang, Chun Yuan, and Jiaqi Wang. 2023. UPop: Unified and Progressive Pruning for Compressing Vision-Language Transformers. In Proceedings of the 40th International Conference on Machine Learning. 31292\u201331311."},{"key":"e_1_3_2_1_46_1","volume-title":"USHER: Holistic Interference Avoidance for Resource Optimized ML Inference. In USENIX Symposium on Operating Systems Design and Implementation. 947\u2013964","author":"Shubha Sudipta Saha","year":"2024","unstructured":"Sudipta Saha Shubha, Haiying Shen, and Anand Iyer. 2024. USHER: Holistic Interference Avoidance for Resource Optimized ML Inference. In USENIX Symposium on Operating Systems Design and Implementation. 947\u2013964."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627703.3629578"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1644"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.235"},{"key":"e_1_3_2_1_50_1","unstructured":"Yujie Wang Shenhan Zhu Fangcheng Fu Xupeng Miao Jie Zhang Juan Zhu Fan Hong Yong Li and Bin Cui. 2024. Efficient Multi-Task Large Model Training via Data Heterogeneity-aware Model Management. arXiv:2409.03365 [cs.DC] https:\/\/arxiv.org\/abs\/2409.03365"},{"key":"e_1_3_2_1_51_1","unstructured":"Shengqiong Wu Hao Fei Leigang Qu Wei Ji and Tat-Seng Chua. 2023. NExT-GPT: Any-to-Any Multimodal LLM."},{"key":"e_1_3_2_1_52_1","unstructured":"Dejing Xu Zhou Zhao Jun Xiao Fei Wu Hanwang Zhang Xiangnan He and Yueting Zhuang. 2017. Video Question Answering via Gradually Refined Attention over Appearance and Motion. In ACM Multimedia."},{"key":"e_1_3_2_1_53_1","unstructured":"Shukang Yin Chaoyou Fu Sirui Zhao Ke Li Xing Sun Tong Xu and Enhong Chen. 2024. A Survey on Multimodal Large Language Models. arXiv:2306.13549 [cs.CV] https:\/\/arxiv.org\/abs\/2306.13549"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649361"},{"key":"e_1_3_2_1_55_1","volume-title":"MIGER: Integrating Multi-Instance GPU and Multi-Process Service for Deep Learning Clusters. In International Conference on Parallel Processing. 504\u2013513","author":"Zhang Bowen","year":"2024","unstructured":"Bowen Zhang, Shuxin Li, and Zhuozhao Li. 2024. MIGER: Integrating Multi-Instance GPU and Multi-Process Service for Deep Learning Clusters. In International Conference on Parallel Processing. 504\u2013513."},{"key":"e_1_3_2_1_56_1","volume-title":"Todor Mihaylov, Myle Ott, Sam Shleifer, Kurt Shuster, Daniel Simig, Punit Singh Koura, Anjali Sridhar, Tianlu Wang, and Luke Zettlemoyer.","author":"Zhang Susan","year":"2022","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, Todor Mihaylov, Myle Ott, Sam Shleifer, Kurt Shuster, Daniel Simig, Punit Singh Koura, Anjali Sridhar, Tianlu Wang, and Luke Zettlemoyer. 2022. OPT: Open Pre-trained Transformer Language Models. arXiv:2205.01068 [cs.CL] https:\/\/arxiv.org\/abs\/2205.01068"},{"key":"e_1_3_2_1_57_1","unstructured":"Zili Zhang Yinmin Zhong Ranchen Ming Hanpeng Hu Jianjian Sun Zheng Ge Yibo Zhu and Xin Jin. 2024. DistTrain: Addressing Model and Data Heterogeneity with Disaggregated Training for Multimodal Large Language Models. arXiv:2408.04275 [cs.DC] https:\/\/arxiv.org\/abs\/2408.04275"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00137"}],"event":{"name":"EUROSYS '26: 21st European Conference on Computer Systems","location":"McEwan Hall\/The University of Edinburgh Edinburgh Scotland UK","acronym":"EUROSYS '26","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 21st European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3767295.3769389","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:09:50Z","timestamp":1780661390000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3767295.3769389"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,26]]},"references-count":58,"alternative-id":["10.1145\/3767295.3769389","10.1145\/3767295"],"URL":"https:\/\/doi.org\/10.1145\/3767295.3769389","relation":{},"subject":[],"published":{"date-parts":[[2026,4,26]]},"assertion":[{"value":"2026-04-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}