{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,11]],"date-time":"2025-06-11T04:11:27Z","timestamp":1749615087808,"version":"3.41.0"},"publisher-location":"Cham","reference-count":19,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031901997","type":"print"},{"value":"9783031902000","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-90200-0_14","type":"book-chapter","created":{"date-parts":[[2025,6,10]],"date-time":"2025-06-10T16:43:04Z","timestamp":1749573784000},"page":"163-172","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Using Partitioned GPU Virtual Functions for\u00a0Mixture of\u00a0Experts"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-5669-2702","authenticated-orcid":false,"given":"Vignesh","family":"Chander","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5110-4161","authenticated-orcid":false,"given":"Tony","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4343-911X","authenticated-orcid":false,"given":"Jerry","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3369-2217","authenticated-orcid":false,"given":"Vamsi","family":"Alla","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,11]]},"reference":[{"key":"14_CR1","unstructured":"Advanced Micro Devices, Inc. AMD CDNA\u2122\u00a03 Architecture. https:\/\/www.amd.com\/content\/dam\/amd\/en\/documents\/instinct-tech-docs\/white-papers\/amdcdna-3-white-paper.pdf"},{"key":"14_CR2","doi-asserted-by":"publisher","unstructured":"Ali-Eldin, A., Tordsson, J., Elmroth, E.: An adaptive hybrid elasticity controller for cloud infrastructures. In: 2012 IEEE Network Operations and Management Symposium, pp. 204\u2013212 (2012). https:\/\/doi.org\/10.1109\/NOMS.2012.6211900.","DOI":"10.1109\/NOMS.2012.6211900."},{"key":"14_CR3","unstructured":"Anthony, Q., Biderman, S., Schoelkopf, H.: Transformer Math 101 (2023). https:\/\/blog.eleuther.ai\/transformer-math\/"},{"key":"14_CR4","unstructured":"Chiang, W.L., et al. Chatbot arena: an open platform for evaluating LLMs by human preference (2024). arXiv: 2403.04132 [cs.AI]"},{"key":"14_CR5","doi-asserted-by":"publisher","unstructured":"Devlin, J., et al.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Burstein, J., Doran, C., Solorio, T. (eds.) Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol. 1 (Long and Short Papers), pp. 4171\u20134186. Association for Computational Linguistics, Minneapolis (2019). https:\/\/doi.org\/10.18653\/v1\/N19-1423. https:\/\/aclanthology.org\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"14_CR6","unstructured":"Ding, D., et al.: Hybrid LLM: cost-efficient and quality-aware query routing (2024). arXiv: 2404.14618 [cs.LG]"},{"key":"14_CR7","doi-asserted-by":"crossref","unstructured":"Fang, C., et al.: LLM-ensemble: optimal large language model ensemble method for E-commerce product attribute value extraction (2024). arXiv: 2403.00863 [cs.IR]","DOI":"10.1145\/3626772.3661357"},{"key":"14_CR8","unstructured":"Fedus, W., Zoph, B., Shazeer, N.: Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. J. Mach. Learn. Res. 23(120), 1\u201339 (2022). http:\/\/jmlr.org\/papers\/v23\/21-0998.html"},{"key":"14_CR9","unstructured":"Grootendorst, M.: BERTopic: neural topic modeling with a class-based TF-IDF procedure. arXiv preprint arXiv:2203.05794 (2022)"},{"issue":"2","key":"14_CR10","doi-asserted-by":"publisher","first-page":"158","DOI":"10.1109\/LCA.2021.3117150","volume":"20","author":"S Gurumurthi","year":"2021","unstructured":"Gurumurthi, S., et al.: HBM3 RAS: enhancing resilience at scale. IEEE Comput. Arch. Lett. 20(2), 158\u2013161 (2021). https:\/\/doi.org\/10.1109\/LCA.2021.3117150","journal-title":"IEEE Comput. Arch. Lett."},{"key":"14_CR11","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models (2021). arXiv: 2106.09685 [cs.CL]"},{"key":"14_CR12","unstructured":"Jiang, A.Q., et al.: Mixtral of experts (2024). arXiv: 2401.04088 [cs.LG]"},{"key":"14_CR13","doi-asserted-by":"crossref","unstructured":"Jiang, D., Ren, X., Lin, B.Y.: LLM-blender: ensembling large language models with pairwise ranking and generative fusion (2023). arXiv: 2306.02561 [cs.CL]","DOI":"10.18653\/v1\/2023.acl-long.792"},{"key":"14_CR14","unstructured":"Jin, R., et al.: A comprehensive evaluation of quantization strategies for large language models (2024). arXiv: 2402.16775 [cs.CL]"},{"key":"14_CR15","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with PagedAttention (2023). arXiv: 2309.06180 [cs.LG]"},{"key":"14_CR16","unstructured":"Smith, A., et al.: Realizing the AMD exascale heterogeneous processor vision (2024). https:\/\/drive.google.com\/file\/d\/1J9tLeVbFRtarzIzkVrULiEBsRsEMNmEO\/view"},{"key":"14_CR17","unstructured":"Ray Team. Ray v2 Architecture (2022). https:\/\/docs.google.com\/document\/d\/1tBw9A4j62ruI5omIJbMxly-la5w4q_TjyJgJL_jN2fI\/preview"},{"key":"14_CR18","unstructured":"Touvron, H., et al.: LLaMA: open and efficient foundation language models (2023). arXiv: 2302.13971 [cs.CL]"},{"key":"14_CR19","unstructured":"Zhou, Y., et al.: Mixture-of-experts with expert choice routing. In: Koyejo, S., et al. Advances in Neural Information Processing Systems, vol. 35, pp. 7103-7114. Curran Associates, Inc. (2022). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/2f00ecd787b432c1d36f3de9800728eb-Paper-Conference.pdf"}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2024: Parallel Processing Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-90200-0_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,10]],"date-time":"2025-06-10T16:43:09Z","timestamp":1749573789000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-90200-0_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031901997","9783031902000"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-90200-0_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"11 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Our future work will focus on implementing, testing and improving our architecture, focusing on our inference and load balancing steps.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Future Work"}},{"value":"Euro-Par","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Madrid","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Spain","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 August 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 August 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"europar2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2024.euro-par.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}