{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T01:54:45Z","timestamp":1772934885986,"version":"3.50.1"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T00:00:00Z","timestamp":1765152000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T00:00:00Z","timestamp":1765152000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,12,8]]},"DOI":"10.1109\/bigdata66926.2025.11402259","type":"proceedings-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T20:57:57Z","timestamp":1772830677000},"page":"6535-6544","source":"Crossref","is-referenced-by-count":0,"title":["APSys: Disaggregated LLM Serving System with an Adaptive Parallel Strategy"],"prefix":"10.1109","author":[{"given":"Zhaochen","family":"Li","sequence":"first","affiliation":[{"name":"China Telecom Cloud Technology Co., Ltd,Intelligent Computing Platform Division,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huaxia","family":"Wang","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co., Ltd,Intelligent Computing Platform Division,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anfa","family":"Zhang","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co., Ltd,Intelligent Computing Platform Division,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiyong","family":"Yan","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co., Ltd,Intelligent Computing Platform Division,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Runhuai","family":"Huang","sequence":"additional","affiliation":[{"name":"China Telecom Cloud Technology Co., Ltd,Intelligent Computing Platform Division,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Deepseek-r1: Incentivizing reasoning capability in 11 ms via reinforcement learning","author":"Guo","year":"2025","journal-title":"arXiv preprint arXiv"},{"key":"ref2","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"ref4","article-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2019","journal-title":"arXiv preprint arXiv"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"ref6","article-title":"Inference without interference: Disaggregate 11 m inference for mixed downstream workloads","author":"Hu","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref7","first-page":"193","article-title":"\\{DistServe\\}: Disaggregating prefill and decoding for goodput-optimized large language model serving","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Zhong","year":"2024"},{"key":"ref8","article-title":"P\/d-serve: Serving disaggregated large language model at scale","author":"Jin","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref9","article-title":"Apex: An extensible and dynamism-aware simulator for automated parallel execution in 11 m serving","author":"Lin","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref10","article-title":"Seesaw: High-throughput 11m inference via model re-sharding","author":"Su","year":"2025","journal-title":"arXiv preprint arXiv"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/s10115-024-02167-7"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1017\/9781009089517"},{"key":"ref13","article-title":"Opt: Open pre-trained transformer language models","author":"Zhang","year":"2022","journal-title":"arXiv preprint arXiv"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref15","article-title":"Sharegpt","volume-title":"URL","author":"Team","year":"2023"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.1995.598994"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1002\/widm.8"},{"key":"ref18","article-title":"Support vector regression machines","volume":"9","author":"Drucker","year":"1996","journal-title":"Advances in neural information processing systems"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1038\/323533a0"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1214\/aos\/1013203451"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1006\/jcss.1997.1504"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295070"},{"key":"ref23","first-page":"155","article-title":"Mooncake: Trading more storage for less computation-a \\{KVCache-centric\\} architecture for serving \\{LLM\\} chatbot","volume-title":"23rd USENIX Conference on File and Storage Technologies (FAST 25)","author":"Qin","year":"2025"},{"key":"ref24","doi-asserted-by":"crossref","first-page":"115","DOI":"10.1145\/3458817.3476209","article-title":"Efficient large-scale language model training on gpu clusters using megatron-lm","volume-title":"Proceedings of the international conference for high performance computing, networking, storage and analysis","author":"Narayanan","year":"2021"},{"key":"ref25","first-page":"559","article-title":"Alpa: Automating inter-and \\{IntraOperator\\} parallelism for distributed deep learning","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng","year":"2022"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0480"}],"event":{"name":"2025 IEEE International Conference on Big Data (BigData)","location":"Macau, China","start":{"date-parts":[[2025,12,8]]},"end":{"date-parts":[[2025,12,11]]}},"container-title":["2025 IEEE International Conference on Big Data (BigData)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11400704\/11400712\/11402259.pdf?arnumber=11402259","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T07:19:30Z","timestamp":1772867970000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11402259\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,8]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/bigdata66926.2025.11402259","relation":{},"subject":[],"published":{"date-parts":[[2025,12,8]]}}}