{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:40:29Z","timestamp":1782834029070,"version":"3.54.5"},"publisher-location":"Cham","reference-count":18,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031998713","type":"print"},{"value":"9783031998720","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:00:00Z","timestamp":1755820800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:00:00Z","timestamp":1755820800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-99872-0_23","type":"book-chapter","created":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T11:08:47Z","timestamp":1755774527000},"page":"327-340","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["ScheInfer: Efficient Inference of\u00a0Large Language Models with\u00a0Task Scheduling on\u00a0Moderate GPUs"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5147-0844","authenticated-orcid":false,"given":"Wenxiang","family":"Lin","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1172-9935","authenticated-orcid":false,"given":"Xinglin","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1418-5160","authenticated-orcid":false,"given":"Shaohuai","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9745-4372","authenticated-orcid":false,"given":"Xiaowen","family":"Chu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,8,22]]},"reference":[{"key":"23_CR1","unstructured":"Abdin, M.I., Jacobs, S.A., Awan, A.A., et\u00a0al.: Phi-3 technical report: a highly capable language model locally on your phone. CoRR abs\/2404.14219 (2024)"},{"key":"23_CR2","doi-asserted-by":"crossref","unstructured":"Aminabadi, R.Y., Rajbhandari, et\u00a0al.: Deepspeed-inference: enabling efficient inference of transformer models at unprecedented scale. In: SC22: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201315. IEEE (2022)","DOI":"10.1109\/SC41404.2022.00051"},{"key":"23_CR3","unstructured":"Brown, T.B., Mann, B., Ryder, N., et\u00a0al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020, 6\u201312 December 2020, virtual (2020)"},{"key":"23_CR4","unstructured":"Dean, J., Corrado, G., Monga, R., et\u00a0al.: Large scale distributed deep networks. In: Advances in Neural Information Processing Systems, vol. 25 (2012)"},{"key":"23_CR5","doi-asserted-by":"crossref","unstructured":"Geng, X., Liu, S., Liu, L., Han, J., Jiang, H.: QUQ: quadruplet uniform quantization for efficient vision transformer inference. In: Proceedings of the 61st ACM\/IEEE Design Automation Conference, pp.\u00a01\u20136 (2024)","DOI":"10.1145\/3649329.3656516"},{"key":"23_CR6","unstructured":"Gerganov, G.: ggerganov\/llama.cpp: Port of Facebook\u2019s llama model in C\/C++. https:\/\/github.com\/ggerganov\/llama.cpp"},{"key":"23_CR7","unstructured":"Jiang, A.Q., Sablayrolles, A., Roux, A., et\u00a0al.: Mixtral of experts. arXiv preprint arXiv:2401.04088 (2024)"},{"key":"23_CR8","doi-asserted-by":"publisher","unstructured":"JosefAlbers, R\u00e9mi: Josefalbers\/phi-3-vision-mlx: Phi-3.5-mlx (2024). https:\/\/doi.org\/10.5281\/zenodo.13352415","DOI":"10.5281\/zenodo.13352415"},{"key":"23_CR9","unstructured":"Kamahori, K., Gu, Y., Zhu, K., Kasikci, B.: Fiddler: CPU-GPU orchestration for fast inference of mixture-of-experts models. CoRR abs\/2402.07033 (2024)"},{"key":"23_CR10","unstructured":"Lepikhin, D., et al.: Gshard: scaling giant models with conditional computation and automatic sharding. In: International Conference on Learning Representations (2021)"},{"key":"23_CR11","doi-asserted-by":"crossref","unstructured":"Lyu, H., Jiang, S., Zeng, H., et\u00a0al.: LLM-rec: personalized recommendation via prompting large language models. In: Findings of the Association for Computational Linguistics: NAACL 2024, Mexico City, Mexico, 16\u201321 June 2024, pp. 583\u2013612. Association for Computational Linguistics (2024)","DOI":"10.18653\/v1\/2024.findings-naacl.39"},{"key":"23_CR12","doi-asserted-by":"crossref","unstructured":"Ma, X., Fang, G., Wang, X.: LLM-pruner: on the structural pruning of large language models. In: Advances in Neural Information Processing Systems, vol. 36, pp. 21702\u201321720 (2023)","DOI":"10.52202\/075280-0950"},{"key":"23_CR13","doi-asserted-by":"crossref","unstructured":"Shi, S., Pan, X., Chu, X., Li, B.: PipeMoE: accelerating mixture-of-experts through adaptive pipelining. In: IEEE INFOCOM 2023-IEEE Conference on Computer Communications (2023)","DOI":"10.1109\/INFOCOM53939.2023.10228874"},{"key":"23_CR14","unstructured":"MLC team: MLC-LLM (2023). https:\/\/github.com\/mlc-ai\/mlc-llm"},{"key":"23_CR15","unstructured":"Touvron, H., Lavril, T., Izacard, G., et\u00a0al.: Llama: open and efficient foundation language models. CoRR abs\/2302.13971 (2023)"},{"key":"23_CR16","unstructured":"Vaswani, A.: Attention is all you need. In: Advances in Neural Information Processing Systems (2017)"},{"key":"23_CR17","unstructured":"Yang, A., Yang, B., Hui, B., et\u00a0al.: Qwen2 technical report. CoRR abs\/2407.10671 (2024)"},{"key":"23_CR18","doi-asserted-by":"crossref","unstructured":"Yuan, S., Chen, J., Fu, Z., et\u00a0al.: Distilling script knowledge from large language models for constrained language planning. arXiv preprint arXiv:2305.05252 (2023)","DOI":"10.18653\/v1\/2023.acl-long.236"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2025: Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-99872-0_23","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T13:24:12Z","timestamp":1780406652000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-99872-0_23"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,22]]},"ISBN":["9783031998713","9783031998720"],"references-count":18,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-99872-0_23","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,8,22]]},"assertion":[{"value":"22 August 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interest"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dresden","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 April 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 April 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2025.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}