{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T21:44:26Z","timestamp":1757627066960,"version":"3.44.0"},"publisher-location":"Cham","reference-count":19,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031998539"},{"type":"electronic","value":"9783031998546"}],"license":[{"start":{"date-parts":[[2025,8,27]],"date-time":"2025-08-27T00:00:00Z","timestamp":1756252800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,27]],"date-time":"2025-08-27T00:00:00Z","timestamp":1756252800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-031-99854-6_10","type":"book-chapter","created":{"date-parts":[[2025,8,26]],"date-time":"2025-08-26T05:09:15Z","timestamp":1756184955000},"page":"145-158","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Leveraging Expert Usage to\u00a0Speed up\u00a0LLM Inference with\u00a0Expert Parallelism"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2741-6228","authenticated-orcid":false,"given":"Olivier","family":"Beaumont","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rapha\u00ebl","family":"Bourgouin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Maxime","family":"Darrin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5519-9913","authenticated-orcid":false,"given":"Loris","family":"Marchal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8717-2117","authenticated-orcid":false,"given":"Pablo","family":"Piantanida","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,27]]},"reference":[{"issue":"3","key":"10_CR1","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1016\/0020-0190(92)90140-Q","volume":"42","author":"TN Bui","year":"1992","unstructured":"Bui, T.N., Jones, C.: Finding good approximate vertex and edge partitions is np-hard. Inf. Process. Lett. 42(3), 153\u2013159 (1992)","journal-title":"Inf. Process. Lett."},{"key":"10_CR2","doi-asserted-by":"crossref","unstructured":"Cai, W., Jiang, J., Wang, F., Tang, J., Kim, S., Huang, J.: A survey on mixture of experts (2024). arXiv preprint arXiv:2407.06204","DOI":"10.36227\/techrxiv.172055626.64129172\/v1"},{"key":"10_CR3","doi-asserted-by":"crossref","unstructured":"Dai, D., et al.: Deepseekmoe: towards ultimate expert specialization in mixture-of-experts language models (2024). arXiv preprint arXiv:2401.06066","DOI":"10.18653\/v1\/2024.acl-long.70"},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Darrin, M., Beaumont, O., Bourgouin, R., Marchal, L., Piantanida, P.: Leveraging expert usage to speed up LLM inference with expert parallelism \u2013 extended version (2025). https:\/\/hal.science\/hal-04994839","DOI":"10.1007\/978-3-031-99854-6_10"},{"key":"10_CR5","unstructured":"Introducing DBRX: A new state-of-the-art open LLM (2024). https:\/\/www.databricks.com\/blog\/introducing-dbrx-new-state-art-open-llm"},{"key":"10_CR6","unstructured":"Du, N., et al.: Glam: efficient scaling of language models with mixture-of-experts. In: ICML 2022 (2022)"},{"key":"10_CR7","unstructured":"Eliseev, A., Mazur, D.: Fast inference of mixture-of-experts language models with offloading (2023). https:\/\/arxiv.org\/abs\/2312.17238"},{"key":"10_CR8","unstructured":"Fedus, W., Zoph, B., Shazeer, N.: Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. J. Mach. Learn. Res. 23 (2022)"},{"key":"10_CR9","volume-title":"Computers and Intractability, a Guide to the Theory of NP-Completeness","author":"MR Garey","year":"1979","unstructured":"Garey, M.R., Johnson, D.S.: Computers and Intractability, a Guide to the Theory of NP-Completeness. W. H, Freeman and Company (1979)"},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"He, J., et al.: Fastermoe: modeling and optimizing training of large-scale dynamic pre-trained models. In: Proceedings of the 27th ACM SIGPLAN Symposium on PPOPP, pp. 120\u2013134 (2022)","DOI":"10.1145\/3503221.3508418"},{"issue":"4","key":"10_CR11","doi-asserted-by":"publisher","first-page":"225","DOI":"10.1137\/0202019","volume":"2","author":"JE Hopcroft","year":"1973","unstructured":"Hopcroft, J.E., Karp, R.M.: An n$$\\hat{\\,}$$5\/2 algorithm for maximum matchings in bipartite graphs. SIAM J. Comput. 2(4), 225\u2013231 (1973)","journal-title":"SIAM J. Comput."},{"key":"10_CR12","unstructured":"Huang, H., et al.: Towards moe deployment: mitigating inefficiencies in mixture-of-expert (moe) inference (2023). https:\/\/arxiv.org\/abs\/2303.06182"},{"key":"10_CR13","first-page":"269","volume":"5","author":"C Hwang","year":"2023","unstructured":"Hwang, C., et al.: Tutel: adaptive mixture-of-experts at scale. Proc. Mach. Learn. Syst. 5, 269\u2013287 (2023)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"10_CR14","unstructured":"Jiang, A.Q., et al.: Mixtral of experts (2024). arXiv preprint arXiv:2401.04088"},{"key":"10_CR15","unstructured":"Lepikhin, D., et al.: Gshard: scaling giant models with conditional computation and automatic sharding (2020). arXiv preprint arXiv:2006.16668"},{"issue":"2","key":"10_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3605943","volume":"56","author":"B Min","year":"2023","unstructured":"Min, B., et al.: Recent advances in natural language processing via large pre-trained language models: a survey. ACM Comput. Surv. 56(2), 1\u201340 (2023)","journal-title":"ACM Comput. Surv."},{"key":"10_CR17","unstructured":"Minaee, S., et al.: Large language models: a survey (2024). arXiv preprint arXiv:2402.06196"},{"key":"10_CR18","unstructured":"Team, Q.: Qwen1.5-moe: matching 7b model performance with 1\/3 activated parameters (2024). https:\/\/qwenlm.github.io\/blog\/qwen- moe\/"},{"key":"10_CR19","unstructured":"Vaswani, A.: attention is all you need. Adv. Neural Inf. Process. Syst. (2017)"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2025: Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-99854-6_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,10]],"date-time":"2025-09-10T00:32:58Z","timestamp":1757464378000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-99854-6_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,27]]},"ISBN":["9783031998539","9783031998546"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-99854-6_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025,8,27]]},"assertion":[{"value":"27 August 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dresden","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2025.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}