{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T23:15:39Z","timestamp":1780355739667,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":19,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819510207","type":"print"},{"value":"9789819510214","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T00:00:00Z","timestamp":1762214400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T00:00:00Z","timestamp":1762214400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-1021-4_19","type":"book-chapter","created":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T10:28:50Z","timestamp":1762165730000},"page":"257-266","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["TokenSim: Enabling Hardware and\u00a0Software Exploration for\u00a0Large Language Model Inference Systems"],"prefix":"10.1007","author":[{"given":"Feiyang","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhuohang","family":"Bian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guoyang","family":"Duan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianle","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junchi","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Teng","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongqiang","family":"Yao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruihao","family":"Gong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Youwei","family":"Zhuo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,11,4]]},"reference":[{"key":"19_CR1","unstructured":"Agrawal, A., et al.: VIDUR: a large-scale simulation framework for LLM inference. In: Gibbons, P., Pekhimenko, G., Sa, C.D. (eds.) Proceedings of Machine Learning and Systems, vol.\u00a06, pp. 351\u2013366 (2024)"},{"key":"19_CR2","unstructured":"Bambhaniya, A., et al.: Demystifying platform requirements for diverse LLM inference use cases (2024). https:\/\/arxiv.org\/abs\/2406.01698"},{"key":"19_CR3","unstructured":"Brown, T.B., et al.: Language models are few-shot learners (2020). https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"19_CR4","unstructured":"Chen, M., et al.: Evaluating large language models trained on code (2021). https:\/\/arxiv.org\/abs\/2107.03374"},{"key":"19_CR5","doi-asserted-by":"publisher","unstructured":"Cho, J., Kim, M., Choi, H., Heo, G., Park, J.: LLMservingsim: a HW\/SW co-simulation infrastructure for LLM inference serving at scale. In: 2024 IEEE International Symposium on Workload Characterization (IISWC), pp. 15\u201329 (2024). https:\/\/doi.org\/10.1109\/IISWC63097.2024.00012","DOI":"10.1109\/IISWC63097.2024.00012"},{"key":"19_CR6","unstructured":"Gao, B., He, Z., Sharma, P., Kang, Q., Jevdjic, D., Deng, J., Yang, X., Yu, Z., Zuo, P.: Cost-Efficient large language model serving for multi-turn conversations with CachedAttention. In: 2024 USENIX Annual Technical Conference (USENIX ATC 24), pp. 111\u2013126. USENIX Association, Santa Clara (2024). https:\/\/www.usenix.org\/conference\/atc24\/presentation\/gao-bin-cost"},{"key":"19_CR7","unstructured":"Google: Gemini - chat to supercharge your ideas (2023). https:\/\/gemini.google.com\/"},{"key":"19_CR8","unstructured":"Google: Google bard (2023). https:\/\/bard.google.com\/"},{"key":"19_CR9","unstructured":"Hu, C., et al.: MemServe: context caching for disaggregated LLM serving with elastic memory pool (2024). https:\/\/arxiv.org\/abs\/2406.17565"},{"key":"19_CR10","doi-asserted-by":"publisher","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with pagedattention. In: Proceedings of the 29th Symposium on Operating Systems Principles. p. 611\u2013626. SOSP \u201923, Association for Computing Machinery, New York, NY, USA (2023). https:\/\/doi.org\/10.1145\/3600006.3613165","DOI":"10.1145\/3600006.3613165"},{"key":"19_CR11","doi-asserted-by":"publisher","unstructured":"Kwon, Y., et al.: System architecture and software stack for gddr6-aim. In: 2022 IEEE Hot Chips 34 Symposium (HCS), pp. 1\u201325 (2022). https:\/\/doi.org\/10.1109\/HCS55958.2022.9895629","DOI":"10.1109\/HCS55958.2022.9895629"},{"key":"19_CR12","unstructured":"M\u00fcller, K.G., Vignaux, T., L\u00fcnsdorf, O., Scherfke, S.: SimPy: discrete event simulation for python (2002). https:\/\/simpy.readthedocs.io\/. version 4.1.1. Accessed 12 Nov 2023"},{"key":"19_CR13","unstructured":"OpenAI: ChatGPT: language model (2023). https:\/\/www.openai.com\/chatgpt"},{"key":"19_CR14","unstructured":"Pope, R., et al.: Efficiently scaling transformer inference (2022). https:\/\/arxiv.org\/abs\/2211.05102"},{"key":"19_CR15","unstructured":"UPMEM: UPMEM: processing-in-memory (PIM) solutions. https:\/\/www.upmem.com\/"},{"key":"19_CR16","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of the 31st International Conference on Neural Information Processing Systems, NIPS\u201917, pp. 6000\u20136010. Curran Associates Inc., Red Hook, NY, USA (2017)"},{"key":"19_CR17","unstructured":"Yu, G.I., Jeong, J.S., Kim, G.W., Kim, S., Chun, B.G.: ORCA: a distributed serving system for Transformer-Based generative models. In: 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pp. 521\u2013538. USENIX Association, Carlsbad, CA (2022). https:\/\/www.usenix.org\/conference\/osdi22\/presentation\/yu"},{"key":"19_CR18","unstructured":"Zhang, H., Ning, A., Prabhakar, R., Wentzlaff, D.: A hardware evaluation framework for large language model inference (2023). https:\/\/arxiv.org\/abs\/2312.03134"},{"key":"19_CR19","unstructured":"Zhong, Y., et al.: DistServe: disaggregating prefill and decoding for goodput-optimized large language model serving (2024). https:\/\/arxiv.org\/abs\/2401.09670"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-1021-4_19","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T10:28:56Z","timestamp":1762165736000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-1021-4_19"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,4]]},"ISBN":["9789819510207","9789819510214"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-1021-4_19","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,4]]},"assertion":[{"value":"4 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Athens","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Greece","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 July 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}