{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,31]],"date-time":"2025-12-31T10:53:02Z","timestamp":1767178382934,"version":"3.48.0"},"publisher-location":"New York, NY, USA","reference-count":11,"publisher":"ACM","funder":[{"name":"JSPS KAKENHI","award":["23H03384"],"award-info":[{"award-number":["23H03384"]}]},{"name":"MIC Forward","award":["JPMI240720005"],"award-info":[{"award-number":["JPMI240720005"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,1,6]]},"DOI":"10.1145\/3737611.3776613","type":"proceedings-article","created":{"date-parts":[[2025,12,31]],"date-time":"2025-12-31T10:35:20Z","timestamp":1767177320000},"page":"72-77","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Transformer-Based Resource and Stage-Aware Scheduling for Model-Parallel LLM Inference"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4329-1935","authenticated-orcid":false,"given":"Rami","family":"Naeem","sequence":"first","affiliation":[{"name":"Graduate School of Information Science and Technology, The University of Osaka, Suita, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9560-2091","authenticated-orcid":false,"given":"Tengis","family":"Buyantogtokh","sequence":"additional","affiliation":[{"name":"Graduate School of Information Science and Technology, The University of Osaka, Suita, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8278-8801","authenticated-orcid":false,"given":"Hamada","family":"Rizk","sequence":"additional","affiliation":[{"name":"Graduate School of Information Science and Technology, The University of Osaka, Suita, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8011-247X","authenticated-orcid":false,"given":"Tatsuya","family":"Amano","sequence":"additional","affiliation":[{"name":"Graduate School of Information Science and Technology, The University of Osaka, Suita, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2273-4876","authenticated-orcid":false,"given":"Hirozumi","family":"Yamaguchi","sequence":"additional","affiliation":[{"name":"Graduate School of Information Science and Technology, RIKEN Center for Computational Science, The University of Osaka, Suita, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,1,4]]},"reference":[{"key":"e_1_3_3_1_2_2","first-page":"117","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Agrawal Amey","year":"2024","unstructured":"Amey Agrawal, Nitin Kedia, Ashish Panwar, Jayashree Mohan, Nipun Kwatra, Bhargav Gulavani, Alexey Tumanov, and Ramachandran Ramjee. 2024. Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). USENIX Association, Santa Clara, CA, 117\u2013134. https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/agrawal"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","unstructured":"Alexander Borzunov Dmitry Baranchuk Tim Dettmers Max Ryabinin Younes Belkada Artem Chumachenko Pavel Samygin and Colin Raffel. 2023. Petals: Collaborative Inference and Fine-tuning of Large Models. 10.48550\/arXiv.2209.01188arXiv:https:\/\/arXiv.org\/abs\/2209.01188 [cs].","DOI":"10.48550\/arXiv.2209.01188"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","unstructured":"Alexander Borzunov Max Ryabinin Artem Chumachenko Dmitry Baranchuk Tim Dettmers Younes Belkada Pavel Samygin and Colin Raffel. 2023. Distributed Inference and Fine-tuning of Large Language Models Over The Internet. 10.48550\/arXiv.2312.08361arXiv:https:\/\/arXiv.org\/abs\/2312.08361 [cs].","DOI":"10.48550\/arXiv.2312.08361"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","unstructured":"Weiwei Jiang Haoyu Han Yang Zhang Ji\u2019an Wang Miao He Weixi Gu Jianbin Mu and Xirong Cheng. 2024. Graph Neural Networks for Routing Optimization: Challenges and Opportunities. Sustainability 16 21 (Oct. 2024) 9239. 10.3390\/su16219239Number: 21 Publisher: Multidisciplinary Digital Publishing Institute.","DOI":"10.3390\/su16219239"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","unstructured":"Woosuk Kwon Zhuohan Li Siyuan Zhuang Ying Sheng Lianmin Zheng Cody\u00a0Hao Yu Joseph\u00a0E. Gonzalez Hao Zhang and Ion Stoica. 2023. Efficient Memory Management for Large Language Model Serving with PagedAttention. 10.48550\/arXiv.2309.06180arXiv:https:\/\/arXiv.org\/abs\/2309.06180 [cs].","DOI":"10.48550\/arXiv.2309.06180"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","unstructured":"Zeinab Nezami Maryam Hafeez Karim Djemame and Syed Ali\u00a0Raza Zaidi. 2024. Generative AI on the Edge: Architecture and Performance Evaluation. 10.48550\/arXiv.2411.17712arXiv:https:\/\/arXiv.org\/abs\/2411.17712 [cs].","DOI":"10.48550\/arXiv.2411.17712"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Minh N.\u00a0H. Nguyen Thien-Lac Ho and Eui-Nam Huh. 2022. Graph Neural Networks for Intelligent Modelling in Network Management and Orchestration: A Survey on Communications. Electronics 11 20 (Oct. 2022) 3371. 10.3390\/electronics11203371","DOI":"10.3390\/electronics11203371"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","unstructured":"Mohammad Shoeybi Mostofa Patwary Raul Puri Patrick LeGresley Jared Casper and Bryan Catanzaro. 2020. Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism. 10.48550\/arXiv.1909.08053arXiv:https:\/\/arXiv.org\/abs\/1909.08053 [cs].","DOI":"10.48550\/arXiv.1909.08053"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","unstructured":"Qidong Su Wei Zhao Xin Li Muralidhar Andoorveedu Chenhao Jiang Zhanda Zhu Kevin Song Christina Giannoula and Gennady Pekhimenko. 2025. Seesaw: High-throughput LLM Inference via Model Re-sharding. 10.48550\/arXiv.2503.06433arXiv:https:\/\/arXiv.org\/abs\/2503.06433 [cs].","DOI":"10.48550\/arXiv.2503.06433"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","unstructured":"Jos\u00e9 Su\u00e1rez-Varela Paul Almasan Miquel Ferriol-Galm\u00e9s Krzysztof Rusek Fabien Geyer Xiangle Cheng Xiang Shi Shihan Xiao Franco Scarselli Albert Cabellos-Aparicio and Pere Barlet-Ros. 2023. Graph Neural Networks for Communication Networks: Context Use Cases and Opportunities. IEEE Network 37 3 (May 2023) 146\u2013153. 10.1109\/MNET.123.2100773","DOI":"10.1109\/MNET.123.2100773"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","unstructured":"Yue Zhang Yuansheng Chen Xuan Mo Alex Xi Jialun Li and WeiGang Wu. 2025. A Predictive and Synergistic Two-Layer Scheduling Framework for LLM Serving. 10.48550\/arXiv.2509.23384arXiv:https:\/\/arXiv.org\/abs\/2509.23384 [cs].","DOI":"10.48550\/arXiv.2509.23384"}],"event":{"name":"ICDCN 2026: 27th International Conference on Distributed Computing and Networking","acronym":"ICDCN Companion 2026","location":"Nara Japan"},"container-title":["Companion Proceedings of the 27th International Conference on Distributed Computing and Networking"],"original-title":[],"deposited":{"date-parts":[[2025,12,31]],"date-time":"2025-12-31T10:48:56Z","timestamp":1767178136000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3737611.3776613"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,4]]},"references-count":11,"alternative-id":["10.1145\/3737611.3776613","10.1145\/3737611"],"URL":"https:\/\/doi.org\/10.1145\/3737611.3776613","relation":{},"subject":[],"published":{"date-parts":[[2026,1,4]]},"assertion":[{"value":"2026-01-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}