{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:41:27Z","timestamp":1777873287736,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":12,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737865","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T20:52:41Z","timestamp":1754254361000},"page":"6302-6303","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["KDD 2025 Workshop on Inference Optimization for Generative AI"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8180-7460","authenticated-orcid":false,"given":"Panpan","family":"Xu","sequence":"first","affiliation":[{"name":"Amazon Web Services, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0970-9214","authenticated-orcid":false,"given":"Youngsuk","family":"Park","sequence":"additional","affiliation":[{"name":"Amazon Web Services, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8935-6602","authenticated-orcid":false,"given":"Lin Lee","family":"Cheong","sequence":"additional","affiliation":[{"name":"Amazon Web Services, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8165-840X","authenticated-orcid":false,"given":"Yida","family":"Wang","sequence":"additional","affiliation":[{"name":"Amazon Web Services, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-6263-7802","authenticated-orcid":false,"given":"Yiying","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of California, San Diego, San Diego, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2753-1437","authenticated-orcid":false,"given":"George","family":"Karypis","sequence":"additional","affiliation":[{"name":"Amazon Web Services, Santa Clara, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4651-3263","authenticated-orcid":false,"given":"Sherry","family":"Marcus","sequence":"additional","affiliation":[{"name":"Amazon Web Services, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Infercept: Efficient intercept support for augmented large language model inference. arXiv preprint arXiv:2402.01869","author":"Abhyankar Reyna","year":"2024","unstructured":"Reyna Abhyankar, Zijian He, Vikranth Srivatsa, Hao Zhang, and Yiying Zhang. 2024. Infercept: Efficient intercept support for augmented large language model inference. arXiv preprint arXiv:2402.01869 (2024)."},{"key":"e_1_3_2_1_2_1","volume-title":"A Proximal Operator for Inducing 2: 4-Sparsity. arXiv preprint arXiv:2501.18015","author":"K\u00fcbler Jonas M","year":"2025","unstructured":"Jonas M K\u00fcbler, Yu-Xiang Wang, Shoham Sabach, Navid Ansari, Matth\u00e4us Kleindessner, Kailash Budhathoki, Volkan Cevher, and George Karypis. 2025. A Proximal Operator for Inducing 2: 4-Sparsity. arXiv preprint arXiv:2501.18015 (2025)."},{"key":"e_1_3_2_1_3_1","volume-title":"ProxSparse: Regularized Learning of Semi-Structured Sparsity Masks for Pretrained LLMs. arXiv preprint arXiv:2502.00258","author":"Liu Hongyi","year":"2025","unstructured":"Hongyi Liu, Rajarshi Saha, Zhen Jia, Youngsuk Park, Jiaji Huang, Shoham Sabach, Yu-Xiang Wang, and George Karypis. 2025. ProxSparse: Regularized Learning of Semi-Structured Sparsity Masks for Pretrained LLMs. arXiv preprint arXiv:2502.00258 (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Stochastic rounding for LLM training: Theory and practice. arXiv preprint arXiv:2502.20566","author":"Ozkara Kaan","year":"2025","unstructured":"Kaan Ozkara, Tao Yu, and Youngsuk Park. 2025. Stochastic rounding for LLM training: Theory and practice. arXiv preprint arXiv:2502.20566 (2025)."},{"key":"e_1_3_2_1_5_1","volume-title":"Marconi: Prefix caching for the era of hybrid llms. arXiv preprint arXiv:2411.19379","author":"Pan Rui","year":"2024","unstructured":"Rui Pan, Zhuang Wang, Zhen Jia, Can Karakus, Luca Zancato, Tri Dao, Yida Wang, and Ravi Netravali. 2024. Marconi: Prefix caching for the era of hybrid llms. arXiv preprint arXiv:2411.19379 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"FastTree: Optimizing Attention Kernel and Runtime for Tree-Structured LLM Inference. In Eighth Conference on Machine Learning and Systems.","author":"Pan Zaifeng","year":"2024","unstructured":"Zaifeng Pan, Yitong Ding, Yue Guan, Zheng Wang, Zhongkai Yu, Xulong Tang, Yida Wang, and Yufei Ding. 2024. FastTree: Optimizing Attention Kernel and Runtime for Tree-Structured LLM Inference. In Eighth Conference on Machine Learning and Systems."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671465"},{"key":"e_1_3_2_1_8_1","volume-title":"Preble: Efficient distributed prompt scheduling for llm serving. arXiv preprint arXiv:2407.00023","author":"Srivatsa Vikranth","year":"2024","unstructured":"Vikranth Srivatsa, Zijian He, Reyna Abhyankar, Dongming Li, and Yiying Zhang. 2024. Preble: Efficient distributed prompt scheduling for llm serving. arXiv preprint arXiv:2407.00023 (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"Training llms with mxfp4. arXiv preprint arXiv:2502.20586","author":"Tseng Albert","year":"2025","unstructured":"Albert Tseng, Tao Yu, and Youngsuk Park. 2025. Training llms with mxfp4. arXiv preprint arXiv:2502.20586 (2025)."},{"key":"e_1_3_2_1_10_1","volume-title":"VL-Cache: Sparsity and Modality-Aware KV Cache Compression for Vision-Language Model Inference Acceleration. arXiv preprint arXiv:2410.23317","author":"Tu Dezhan","year":"2024","unstructured":"Dezhan Tu, Danylo Vashchilenko, Yuzhe Lu, and Panpan Xu. 2024. VL-Cache: Sparsity and Modality-Aware KV Cache Compression for Vision-Language Model Inference Acceleration. arXiv preprint arXiv:2410.23317 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"Dongyeop Kang, Youngsuk Park, and Mingyi Hong.","author":"Yau Chung-Yiu","year":"2025","unstructured":"QuanWei, Chung-Yiu Yau, Hoi-ToWai, Yang Katie Zhao, Dongyeop Kang, Youngsuk Park, and Mingyi Hong. 2025. RoSTE: An Efficient Quantization-Aware Supervised Fine-Tuning Approach for Large Language Models. arXiv preprint arXiv:2502.09003 (2025)."},{"key":"e_1_3_2_1_12_1","volume-title":"ScaleFusion: Scalable Inference of Spatial-Temporal Diffusion Transformers for High-Resolution Long Video Generation. In Eighth Conference on Machine Learning and Systems.","author":"Yang Jiacheng","year":"2024","unstructured":"Jiacheng Yang, Jun Wu, Zhen Zhang, Xinwei Fu, Zhiying Xu, Zhen Jia, Yida Wang, and Gennady Pekhimenko. 2024. ScaleFusion: Scalable Inference of Spatial-Temporal Diffusion Transformers for High-Resolution Long Video Generation. In Eighth Conference on Machine Learning and Systems."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737865","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:53:22Z","timestamp":1777571602000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737865"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":12,"alternative-id":["10.1145\/3711896.3737865","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737865","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}