{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:58:18Z","timestamp":1782835098066,"version":"3.54.5"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,5,20]],"date-time":"2024-05-20T00:00:00Z","timestamp":1716163200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,5,20]],"date-time":"2024-05-20T00:00:00Z","timestamp":1716163200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100003012","name":"Impact Fund","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100003012","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,5,20]]},"DOI":"10.1109\/infocom52122.2024.10621087","type":"proceedings-article","created":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T13:25:41Z","timestamp":1723469141000},"page":"1021-1030","source":"Crossref","is-referenced-by-count":4,"title":["OTAS: An Elastic Transformer Serving System via Token Adaptation"],"prefix":"10.1109","author":[{"given":"Jinyu","family":"Chen","sequence":"first","affiliation":[{"name":"The Hong Kong Polytechnic University,Department of Computing"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenchao","family":"Xu","sequence":"additional","affiliation":[{"name":"The Hong Kong Polytechnic University,Department of Computing"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zicong","family":"Hong","sequence":"additional","affiliation":[{"name":"The Hong Kong Polytechnic University,Department of Computing"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Song","family":"Guo","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology,Department of Computer Science and Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haozhao","family":"Wang","sequence":"additional","affiliation":[{"name":"Huazhong University of Science and Technology,School of Computer Science and Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Zhang","sequence":"additional","affiliation":[{"name":"The Hong Kong Polytechnic University,Department of Computing"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deze","family":"Zeng","sequence":"additional","affiliation":[{"name":"China University of Geosciences,School of Computer Science"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Surpassing nvidia fastertransformer\u2019s inference performance by 50%, open source project powers into the future of large models industrialization. HPC-AI Tech","year":"2022"},{"key":"ref2","volume-title":"Github copilot: Your ai pair programmer. Github","year":"2023"},{"key":"ref3","volume-title":"Introducing chatgpt. OpenAI","year":"2022"},{"key":"ref4","volume-title":"Chatgpt\u2019s growth begins to flatten, up 12.6% from march to april. Similarweb","year":"2023"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.14778\/3570690.3570692"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3587438"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3575698"},{"key":"ref8","first-page":"397","article-title":"INFaaS: Automated Model-less Inference Serving","volume-title":"USENIX Annual Technical Conference","author":"Romero"},{"key":"ref9","article-title":"High-throughput generative inference of large language models with a single gpu","author":"Sheng","year":"2023"},{"key":"ref10","article-title":"What are tokens?","volume-title":"Microsoft","year":"2023"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref12","article-title":"Token Merging: Your ViT but Faster","volume-title":"International Conference on Learning Representations","author":"Bolya"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"ref14","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref15","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"issue":"1","key":"ref16","first-page":"5232","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"The Journal of Machine Learning Research"},{"key":"ref17","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref18","article-title":"Scaling vision transformers to 22 billion parameters","author":"Dehghani","year":"2023"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"ref20","article-title":"On the opportunities and risks of foundation models","author":"Bommasani","year":"2021"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-demo.10"},{"key":"ref22","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref23","article-title":"Imagenet-21k pretraining for the masses","author":"Ridnik","year":"2021"},{"key":"ref24","first-page":"613","article-title":"Clipper: A Low-Latency Online Prediction Serving System","volume":"17","author":"Crankshaw","year":"2017","journal-title":"NSDI"},{"key":"ref25","first-page":"183","article-title":"DVABatch: Diversity-aware Multi-Entry Multi-Exit Batching for Efficient Processing of DNN Services on GPUs","volume-title":"2022 USENIX Annual Technical Conference (USENIX ATC 22)","author":"Cui"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1002\/nav.20231"},{"key":"ref27","first-page":"489","article-title":"PetS: A unified framework for Parameter-Efficient transformers serving","volume-title":"2022 USENIX Annual Technical Conference (USENIX ATC 22)","author":"Zhou"},{"key":"ref28","article-title":"PyTorch Image Models","author":"Wightman","year":"2019"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2019.2918242"},{"key":"ref30","first-page":"724","volume-title":"Faster and Cheaper Serverless Computing on Harvested Resources","author":"Zhang"},{"key":"ref31","first-page":"787","article-title":"SHEPHERD: Serving DNNs in the wild","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Zhang"},{"key":"ref32","article-title":"AlpaServe: Statistical Multiplexing with Model Parallelism for Deep Learning Serving","author":"Li","year":"2023"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00073"},{"key":"ref34","first-page":"521","article-title":"Orca: A Distributed Serving System for Transformer-Based Generative Models","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM48880.2022.9796853"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2022.3189186"},{"key":"ref37","article-title":"Nvidia fastertransformer","volume-title":"NVIDIA","year":"2019"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM48880.2022.9796939"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM48880.2022.9796884"}],"event":{"name":"IEEE INFOCOM 2024 - IEEE Conference on Computer Communications","location":"Vancouver, BC, Canada","start":{"date-parts":[[2024,5,20]]},"end":{"date-parts":[[2024,5,23]]}},"container-title":["IEEE INFOCOM 2024 - IEEE Conference on Computer Communications"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10621050\/10621073\/10621087.pdf?arnumber=10621087","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,26]],"date-time":"2025-08-26T19:02:26Z","timestamp":1756234946000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10621087\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,20]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/infocom52122.2024.10621087","relation":{},"subject":[],"published":{"date-parts":[[2024,5,20]]}}}