{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T07:09:20Z","timestamp":1761894560954,"version":"build-2065373602"},"reference-count":44,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/icme59968.2025.11209145","type":"proceedings-article","created":{"date-parts":[[2025,10,30]],"date-time":"2025-10-30T17:57:42Z","timestamp":1761847062000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["InterLayer: Efficient Inference with Interleaved Scheduling and Layer-Specific Optimization"],"prefix":"10.1109","author":[{"given":"Limin","family":"Cheng","sequence":"first","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Software"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hang","family":"Qin","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Software"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shouxu","family":"Kuang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Software"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinyu","family":"Wang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Software"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ling","family":"Li","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Software"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanjun","family":"Wu","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Software"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chen","family":"Zhao","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences,Institute of Software"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME55011.2023.00014"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICME57554.2024.10687572"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW63481.2024.10645431"},{"key":"ref4","article-title":"Technical blog"},{"key":"ref5","article-title":"Run ai with an api"},{"key":"ref6","article-title":"Open source cloud functions for data teams"},{"key":"ref7","article-title":"Language model inference"},{"article-title":"Faaswap: Slo-aware, gpu-efficient serverless inference via model swapping","year":"2023","author":"Yu","key":"ref8"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3317550.3321443"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3581791.3596842"},{"key":"ref11","first-page":"59","article-title":"Streambox: A lightweight gpu sandbox for serverless inference workflow","volume-title":"USENIX ATC","author":"Wu"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3638757"},{"key":"ref13","first-page":"499","article-title":"Pipeswitch: Fast pipelined context switching for deep learning applications","volume-title":"OSDI","author":"Bai"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3552326.3567508"},{"key":"ref15","first-page":"135","article-title":"Serverlessllm: Low-latency serverless inference for large language models","volume-title":"OSDI","author":"Fu"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3698038.3698510"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref18","article-title":"Easy, fast, and cheap llm serving for everyone"},{"key":"ref19","first-page":"521","article-title":"Orca: A distributed serving system for transformer-based generative models","volume-title":"OSDI","author":"Yu"},{"key":"ref20","first-page":"173","article-title":"Llumnix: Dynamic scheduling for large language model serving","volume-title":"OSDI","author":"Sun"},{"key":"ref21","first-page":"117","article-title":"Taming throughput-latency tradeoff in llm inference with sarathiserve","volume-title":"OSDI","author":"Agrawal"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695964"},{"key":"ref23","first-page":"31094","article-title":"Flexgen: High-throughput generative inference of large language models with a single gpu","volume-title":"ICML","author":"Sheng"},{"key":"ref24","first-page":"5209","article-title":"Medusa: Simple llm inference acceleration framework with multiple decoding heads","volume-title":"ICML","author":"Cai"},{"article-title":"Tandem transformers for inference efficient llms","volume-title":"ICML","author":"Aishwarya","key":"ref25"},{"key":"ref26","first-page":"8578","article-title":"Kv-runahead: scalable causal llm inference by parallel key-value cache generation","volume-title":"ICML","author":"Cho"},{"key":"ref27","first-page":"613","article-title":"Clipper: A low-latency online prediction serving system","volume-title":"NSDI","author":"Crankshaw"},{"key":"ref28","article-title":"Serving models"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3673038.3673122"},{"key":"ref30","article-title":"llama"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1093\/nsr\/nwae403"},{"key":"ref32","article-title":"Kvquant: Towards 10 million context length llm inference with kv cache quantization","author":"Hooper","year":"2024","journal-title":"NeurIPS"},{"key":"ref33","article-title":"vicuna-13b-v1.5"},{"key":"ref34","article-title":"Llama-2-7b-chat-hf"},{"key":"ref35","article-title":"Meta-Llama-3-8B"},{"key":"ref36","article-title":"Nvlink and nvlink switch"},{"key":"ref37","article-title":"opt-6.7b"},{"key":"ref38","article-title":"gpt3-finnish-3b"},{"key":"ref39","article-title":"deepseek-llm-7b-chat"},{"key":"ref40","article-title":"gemma"},{"key":"ref41","article-title":"llava-1.5-13b-hf"},{"key":"ref42","article-title":"llava-v1.6-mistral-7b-hf"},{"key":"ref43","first-page":"133","article-title":"Peeking behind the curtains of serverless platforms","author":"Wang","year":"2018","journal-title":"USENIX ATC"},{"key":"ref44","first-page":"1049","article-title":"Mark: Exploiting cloud services for cost-effective, slo-aware machine learning inference serving","author":"Zhang","year":"2019","journal-title":"USENIX ATC"}],"event":{"name":"2025 IEEE International Conference on Multimedia and Expo (ICME)","start":{"date-parts":[[2025,6,30]]},"location":"Nantes, France","end":{"date-parts":[[2025,7,4]]}},"container-title":["2025 IEEE International Conference on Multimedia and Expo (ICME)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11208895\/11208897\/11209145.pdf?arnumber=11209145","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T05:30:37Z","timestamp":1761888637000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11209145\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":44,"URL":"https:\/\/doi.org\/10.1109\/icme59968.2025.11209145","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}