{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:45:53Z","timestamp":1784738753087,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":74,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,22]]},"DOI":"10.1145\/3779212.3790161","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T13:55:26Z","timestamp":1773150926000},"page":"732-748","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["FastTTS: Accelerating Test-Time Scaling for Edge LLM Reasoning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-0379-9679","authenticated-orcid":false,"given":"Hao Mark","family":"Chen","sequence":"first","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6435-263X","authenticated-orcid":false,"given":"Zhiwen","family":"Mo","sequence":"additional","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2848-8913","authenticated-orcid":false,"given":"Guanxi","family":"Lu","sequence":"additional","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3836-5343","authenticated-orcid":false,"given":"Shuang","family":"Liang","sequence":"additional","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9524-5476","authenticated-orcid":false,"given":"Lingxiao","family":"Ma","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6750-927X","authenticated-orcid":false,"given":"Wayne","family":"Luk","sequence":"additional","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2387-5611","authenticated-orcid":false,"given":"Hongxiang","family":"Fan","sequence":"additional","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,3,22]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2025. Artificial Analysis: LLM Leaderboard. https:\/\/artificialanalysis.ai\/leaderboards\/models'size_class=large&reasoning=reasoning. Accessed: 2025-08--16."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.678"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41404.2022.00051"},{"key":"e_1_3_2_1_4_1","unstructured":"Mislav Balunovi? Jasper Dekoninck Ivo Petrov Nikola Jovanovi? and Martin Vechev. 2025. MathArena: Evaluating LLMs on Uncontaminated Math Competitions. https:\/\/matharena.ai\/"},{"key":"e_1_3_2_1_5_1","unstructured":"Edward Beeching Lewis Tunstall and Sasha Rush. [n.d.]. Scaling test-time compute with open models. https:\/\/huggingface.co\/spaces\/HuggingFaceH4\/blogpost-scaling-test-time-compute"},{"key":"e_1_3_2_1_6_1","volume-title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads. In International Conference on Machine Learning (ICML).","author":"Cai Tianle","year":"2024","unstructured":"Tianle Cai, Yuhong Li, Zhengyang Geng, Hongwu Peng, Jason D. Lee, Deming Chen, and Tri Dao. 2024. Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads. In International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_7_1","volume-title":"Rethinking Optimal Verification Granularity for Compute-Efficient Test-Time Scaling. Advances in Neural Information Processing Systems","author":"Chen Hao Mark","year":"2025","unstructured":"Hao Mark Chen, Guanxi Lu, Yasuyuki Okoshi, Zhiwen Mo, Masato Motomura, and Hongxiang Fan. 2025. Rethinking Optimal Verification Granularity for Compute-Efficient Test-Time Scaling. Advances in Neural Information Processing Systems (2025)."},{"key":"e_1_3_2_1_8_1","volume-title":"Rui Li, Konstantin Mishchenko, Stylianos I Venieris, and Hongxiang Fan.","author":"Chen Hao Mark","year":"2024","unstructured":"Hao Mark Chen, Wayne Luk, Ka Fai Cedric Yiu, Rui Li, Konstantin Mishchenko, Stylianos I Venieris, and Hongxiang Fan. 2024. Hardwareaware parallel prompt decoding for memory-efficient acceleration of llm inference. arXiv preprint arXiv:2405.18628 (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"Toward adaptive reasoning in large language models with thought rollback. arXiv preprint arXiv:2412.19707","author":"Chen Sijia","year":"2024","unstructured":"Sijia Chen and Baochun Li. 2024. Toward adaptive reasoning in large language models with thought rollback. arXiv preprint arXiv:2412.19707 (2024)."},{"key":"e_1_3_2_1_10_1","unstructured":"Xinhao Cheng. [n.d.]. SpecInfer: Accelerating Generative Large Language Model Serving with Speculative Inference and Token Tree Verification. Ph.D. Dissertation. Carnegie Mellon University."},{"key":"e_1_3_2_1_11_1","unstructured":"Karl Cobbe Vineet Kosaraju Mohammad Bavarian Mark Chen Heewoo Jun Lukasz Kaiser Matthias Plappert Jerry Tworek Jacob Hilton Reiichiro Nakano et al. 2021. Training verifiers to solve math word problems. arXiv preprint arXiv:2110.14168 (2021)."},{"key":"e_1_3_2_1_12_1","volume-title":"Ying Wen, Weinan Zhang, and Jun Wang.","author":"Feng Xidong","year":"2023","unstructured":"Xidong Feng, Ziyu Wan, Muning Wen, Stephen Marcus McAleer, Ying Wen, Weinan Zhang, and Jun Wang. 2023. Alphazero-like tree-search can guide large language model decoding and training. arXiv preprint arXiv:2309.17179 (2023)."},{"key":"e_1_3_2_1_13_1","volume-title":"Break the Sequential Dependency of LLM Inference Using Lookahead Decoding. In International Conference on Machine Learning (ICML).","author":"Fu Yichao","year":"2024","unstructured":"Yichao Fu, Peter Bailis, Ion Stoica, and Hao Zhang. 2024. Break the Sequential Dependency of LLM Inference Using Lookahead Decoding. In International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_14_1","unstructured":"Yichao Fu Junda Chen Siqi Zhu Zheyu Fu Zhongdongming Dai Yonghao Zhuang Yian Ma Aurick Qiao Tajana Rosing Ion Stoica et al. 2024. Efficiently Scaling LLM Reasoning with Certaindex. arXiv preprint arXiv:2412.20993 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Scaling Speculative Decoding with Lookahead Reasoning. arXiv preprint arXiv:2506.19830","author":"Fu Yichao","year":"2025","unstructured":"Yichao Fu, Rui Ge, Zelei Shao, Zhijie Deng, and Hao Zhang. 2025. Scaling Speculative Decoding with Lookahead Reasoning. arXiv preprint arXiv:2506.19830 (2025)."},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24)","author":"Fu Yaoqi","year":"2024","unstructured":"Yaoqi Fu, Yanqi Zhang, et al. 2024. ServerlessLLM: Low-Latency Serverless Inference for Large Language Models. In Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24). https:\/\/www.usenix.org\/system\/files\/osdi24-fu.pdf"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3658617.3697616"},{"key":"e_1_3_2_1_18_1","volume-title":"Does Thinking More always Help? Understanding Test-Time Scaling in Reasoning Models. arXiv preprint arXiv:2506.04210","author":"Ghosal Soumya Suvra","year":"2025","unstructured":"Soumya Suvra Ghosal, Souradip Chakraborty, Avinash Reddy, Yifu Lu, Mengdi Wang, Dinesh Manocha, Furong Huang, Mohammad Ghavamzadeh, and Amrit Singh Bedi. 2025. Does Thinking More always Help? Understanding Test-Time Scaling in Reasoning Models. arXiv preprint arXiv:2506.04210 (2025)."},{"key":"e_1_3_2_1_19_1","unstructured":"Google DeepMind. 2025. AlphaEvolve: A Gemini-Powered Coding Agent for Designing Advanced Algorithms. https:\/\/deepmind.google\/discover\/blog\/alphaevolve-a-gemini-poweredcoding- agent-for-designing-advanced-algorithms\/. Accessed: 2025-07-08."},{"key":"e_1_3_2_1_20_1","unstructured":"Daya Guo Qihao Zhu Dejian Yang Zhenda Xie Kai Dong Wentao Zhang Guanting Chen Xiao Bi Yu Wu YK Li et al. 2024. DeepSeek-Coder: When the Large Language Model Meets Programming--The Rise of Code Intelligence. arXiv preprint arXiv:2401.14196 (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"Liang Zeng, XiaokunWang, Boyang Wang, Yongcong Li, Fuxiang Zhang, Jiacheng Xu, Bo An, Yang Liu, and Yahui Zhou.","author":"He Jujie","year":"2024","unstructured":"Jujie He, Tianwen Wei, Rui Yan, Jiacai Liu, Chaojie Wang, Yimeng Gan, Shiwen Tu, Chris Yuhao Liu, Liang Zeng, XiaokunWang, Boyang Wang, Yongcong Li, Fuxiang Zhang, Jiacheng Xu, Bo An, Yang Liu, and Yahui Zhou. 2024. Skywork-o1 Open Series. https:\/\/huggingface.co\/Skywork. https:\/\/huggingface.co\/Skywork"},{"key":"e_1_3_2_1_22_1","volume-title":"Fastdecode: High-throughput gpuefficient llm serving using heterogeneous pipelines. arXiv preprint arXiv:2403.11421","author":"He Jiaao","year":"2024","unstructured":"Jiaao He and Jidong Zhai. 2024. Fastdecode: High-throughput gpuefficient llm serving using heterogeneous pipelines. arXiv preprint arXiv:2403.11421 (2024)."},{"key":"e_1_3_2_1_23_1","volume-title":"Ets: Efficient tree search for inference-time scaling. arXiv preprint arXiv:2502.13575","author":"Hooper Coleman","year":"2025","unstructured":"Coleman Hooper, Sehoon Kim, Suhong Moon, Kerem Dilmen, Monishwaran Maheswaran, Nicholas Lee, Michael W Mahoney, Sophia Shao, Kurt Keutzer, and Amir Gholami. 2025. Ets: Efficient tree search for inference-time scaling. arXiv preprint arXiv:2502.13575 (2025)."},{"key":"e_1_3_2_1_24_1","volume-title":"Speculative decoding and beyond: An in-depth survey of techniques. arXiv preprint arXiv:2502.19732","author":"Hu Yunhai","year":"2025","unstructured":"Yunhai Hu, Zining Liu, Zhenyuan Dong, Tianfan Peng, Bradley McDanel, and Sai Qian Zhang. 2025. Speculative decoding and beyond: An in-depth survey of techniques. arXiv preprint arXiv:2502.19732 (2025)."},{"key":"e_1_3_2_1_25_1","volume-title":"HedraRAG: Coordinating LLM Generation and Database Retrieval in Heterogeneous RAG Serving. arXiv preprint arXiv:2507.09138","author":"Hu Zhengding","year":"2025","unstructured":"Zhengding Hu, Vibha Murthy, Zaifeng Pan, Wanlu Li, Xiaoyi Fang, Yufei Ding, and Yuke Wang. 2025. HedraRAG: Coordinating LLM Generation and Database Retrieval in Heterogeneous RAG Serving. arXiv preprint arXiv:2507.09138 (2025)."},{"key":"e_1_3_2_1_26_1","volume-title":"Neo: Saving gpu memory crisis with cpu offloading for online llm inference. arXiv preprint arXiv:2411.01142","author":"Jiang Xuanlin","year":"2024","unstructured":"Xuanlin Jiang, Yang Zhou, Shiyi Cao, Ion Stoica, and Minlan Yu. 2024. Neo: Saving gpu memory crisis with cpu offloading for online llm inference. arXiv preprint arXiv:2411.01142 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Ragcache: Efficient knowledge caching for retrieval augmented generation. arXiv preprint arXiv:2404.12457","author":"Jin Chao","year":"2024","unstructured":"Chao Jin, Zili Zhang, Xuanlin Jiang, Fangyue Liu, Xin Liu, Xuanzhe Liu, and Xin Jin. 2024. Ragcache: Efficient knowledge caching for retrieval augmented generation. arXiv preprint arXiv:2404.12457 (2024)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731092"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731019"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731073"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24)","author":"Lee Wonbeom","year":"2024","unstructured":"Wonbeom Lee, Jungi Lee, Junghwan Seo, and Jaewoong Sim. 2024. InfiniGen: Efficient Generative Inference of Large Language Models with Dynamic KV Cache Management. In Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24). https:\/\/www.usenix.org\/system\/files\/osdi24-lee.pdf"},{"key":"e_1_3_2_1_33_1","volume-title":"Eagle-2: Faster inference of language models with dynamic draft trees. arXiv preprint arXiv:2406.16858","author":"Li Yuhui","year":"2024","unstructured":"Yuhui Li, Fangyun Wei, Chao Zhang, and Hongyang Zhang. 2024. Eagle-2: Faster inference of language models with dynamic draft trees. arXiv preprint arXiv:2406.16858 (2024)."},{"key":"e_1_3_2_1_34_1","volume-title":"Eagle-3: Scaling up inference acceleration of large language models via training-time test. arXiv preprint arXiv:2503.01840","author":"Li Yuhui","year":"2025","unstructured":"Yuhui Li, Fangyun Wei, Chao Zhang, and Hongyang Zhang. 2025. Eagle-3: Scaling up inference acceleration of large language models via training-time test. arXiv preprint arXiv:2503.01840 (2025)."},{"key":"e_1_3_2_1_35_1","volume-title":"Rewardguided speculative decoding for efficient llm reasoning. arXiv preprint arXiv:2501.19324","author":"Liao Baohao","year":"2025","unstructured":"Baohao Liao, Yuhui Xu, Hanze Dong, Junnan Li, Christof Monz, Silvio Savarese, Doyen Sahoo, and Caiming Xiong. 2025. Rewardguided speculative decoding for efficient llm reasoning. arXiv preprint arXiv:2501.19324 (2025)."},{"key":"e_1_3_2_1_36_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Lightman Hunter","year":"2023","unstructured":"Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe. 2023. Let's verify step by step. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24)","author":"Lin Chaofan","year":"2024","unstructured":"Chaofan Lin, Zhenhua Han, Chengruidong Zhang, Yuqing Yang, and Fan Yang. 2024. Parrot: Efficient Serving of LLM-based Applications with Semantic Variable. In Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24). https:\/\/www.usenix.org\/system\/files\/osdi24-lin-chaofan.pdf"},{"key":"e_1_3_2_1_38_1","volume-title":"Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling. arXiv preprint arXiv:2502.06703","author":"Liu Runze","year":"2025","unstructured":"Runze Liu, Junqi Gao, Jian Zhao, Kaiyan Zhang, Xiu Li, Biqing Qi, Wanli Ouyang, and Bowen Zhou. 2025. Can 1B LLM Surpass 405B LLM? Rethinking Compute-Optimal Test-Time Scaling. arXiv preprint arXiv:2502.06703 (2025)."},{"key":"e_1_3_2_1_39_1","unstructured":"Saumya Malik Valentina Pyatkin Sander Land Jacob Morrison Noah A. Smith Hannaneh Hajishirzi and Nathan Lambert. 2025. RewardBench 2: Advancing Reward Model Evaluation."},{"key":"e_1_3_2_1_40_1","volume-title":"American Invitational Mathematics Examination (AIME). https:\/\/maa.org\/mathcompetitions\/american-invitational-mathematics-examinationaime. Accessed","author":"Mathematical Association of America. 2024.","year":"2024","unstructured":"Mathematical Association of America. 2024. American Invitational Mathematics Examination (AIME). https:\/\/maa.org\/mathcompetitions\/american-invitational-mathematics-examinationaime. Accessed February 2024."},{"key":"e_1_3_2_1_41_1","volume-title":"Benedict Emoekabu, Aswanth Krishnan, Tanya Gupta, Mara Schilling-Wilhelmi, Macjonathan Okereke, Anagha Aneesh, et al.","author":"Mirza Adrian","year":"2025","unstructured":"Adrian Mirza, Nawaf Alampara, Sreekanth Kunchapu, Marti\u00f1o R\u00edos- Garc\u00eda, Benedict Emoekabu, Aswanth Krishnan, Tanya Gupta, Mara Schilling-Wilhelmi, Macjonathan Okereke, Anagha Aneesh, et al. 2025. A framework for evaluating the chemical knowledge and reasoning abilities of large language models against the expertise of chemists. Nature Chemistry (2025), 1--8."},{"key":"e_1_3_2_1_42_1","volume-title":"Stephan Jonas, Oliver Aalami, and Paul Schmiedmayer.","author":"Nissen Leon","year":"2025","unstructured":"Leon Nissen, Philipp Zagar, Vishnu Ravi,Aydin Zahedivash, Lara Marie Reimer, Stephan Jonas, Oliver Aalami, and Paul Schmiedmayer. 2025. Medicine on the edge: Comparative performance analysis of on-device LLMs for clinical reasoning. arXiv preprint arXiv:2502.08954 (2025)."},{"key":"e_1_3_2_1_43_1","unstructured":"NVIDIA Corporation. 2025. NVIDIA Nsight Systems. https:\/\/developer.nvidia.com\/nsight-systems Accessed: 2025-04--15."},{"key":"e_1_3_2_1_44_1","volume-title":"Specreason: Fast and accurate inference-time compute via speculative reasoning. arXiv preprint arXiv:2504.07891","author":"Pan Rui","year":"2025","unstructured":"Rui Pan, Yinwei Dai, Zhihao Zhang, Gabriele Oliaro, Zhihao Jia, and Ravi Netravali. 2025. Specreason: Fast and accurate inference-time compute via speculative reasoning. arXiv preprint arXiv:2504.07891 (2025)."},{"key":"e_1_3_2_1_45_1","volume-title":"FastTree: Optimizing Attention Kernel and Runtime for Tree-Structured LLM Inference. Eighth Conference on Machine Learning and Systems.","author":"Pan Zaifeng","year":"2025","unstructured":"Zaifeng Pan, Yitong Ding, Yue Guan, Zheng Wang, Zhongkai Yu, Xulong Tang, Yida Wang, and Yufei Ding. 2025. FastTree: Optimizing Attention Kernel and Runtime for Tree-Structured LLM Inference. Eighth Conference on Machine Learning and Systems."},{"key":"e_1_3_2_1_46_1","volume-title":"KVFlow: Efficient Prefix Caching for Accelerating LLM-Based Multi-Agent Workflows. arXiv preprint arXiv:2507.07400","author":"Pan Zaifeng","year":"2025","unstructured":"Zaifeng Pan, Ajjkumar Patel, Zhengding Hu, Yipeng Shen, Yue Guan, Wan-Lu Li, Lianhui Qin, YidaWang, and Yufei Ding. 2025. KVFlow: Efficient Prefix Caching for Accelerating LLM-Based Multi-Agent Workflows. arXiv preprint arXiv:2507.07400 (2025)."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i23.34690"},{"key":"e_1_3_2_1_48_1","volume-title":"J Griffin, Herumb Shandilya, Adrian Gamarra Lafuente, Medhya Goel, Rebecca Joseph, Shlok Natarajan, Etash Kumar Guha, et al.","author":"Saad-Falcon Jon","year":"2025","unstructured":"Jon Saad-Falcon, Avanika Narayan, Hakki Orhun Akengin, J Griffin, Herumb Shandilya, Adrian Gamarra Lafuente, Medhya Goel, Rebecca Joseph, Shlok Natarajan, Etash Kumar Guha, et al. 2025. Intelligence per watt: Measuring intelligence efficiency of local ai. arXiv preprint arXiv:2511.07885 (2025)."},{"key":"e_1_3_2_1_49_1","volume-title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300","author":"Shao Zhihong","year":"2024","unstructured":"Zhihong Shao, PeiyiWang, Qihao Zhu, Runxin Xu, Junxiao Song, Xiao Bi, Haowei Zhang, Mingchuan Zhang, YK Li, Yang Wu, et al. 2024. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"International Conference on Machine Learning. PMLR, 31094--31116","author":"Sheng Ying","year":"2023","unstructured":"Ying Sheng, Lianmin Zheng, Binhang Yuan, Zhuohan Li, Max Ryabinin, Beidi Chen, Percy Liang, Christopher R\u00e9, Ion Stoica, and Ce Zhang. 2023. Flexgen: High-throughput generative inference of large language models with a single gpu. In International Conference on Machine Learning. PMLR, 31094--31116."},{"key":"e_1_3_2_1_51_1","volume-title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters. arXiv preprint arXiv:2408.03314","author":"Snell Charlie","year":"2024","unstructured":"Charlie Snell, Jaehoon Lee, Kelvin Xu, and Aviral Kumar. 2024. Scaling llm test-time compute optimally can be more effective than scaling model parameters. arXiv preprint arXiv:2408.03314 (2024)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695964"},{"key":"e_1_3_2_1_53_1","volume-title":"Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24)","author":"Sun Biao","year":"2024","unstructured":"Biao Sun, Shuxin Zhang, et al. 2024. Llumnix: Dynamic Scheduling for Large Language Model Serving. In Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24). https:\/\/www.usenix.org\/system\/files\/osdi24-sun-biao.pdf"},{"key":"e_1_3_2_1_54_1","unstructured":"DeepSeek Team et al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. https:\/\/arxiv.org\/abs\/2501.12948. arXiv:2501.12948 [cs.CL]."},{"key":"e_1_3_2_1_55_1","volume-title":"Solving math word problems with process-and outcome-based feedback. arXiv preprint arXiv:2211.14275","author":"Uesato Jonathan","year":"2022","unstructured":"Jonathan Uesato, Nate Kushman, Ramana Kumar, Francis Song, Noah Siegel, LisaWang, Antonia Creswell, Geoffrey Irving, and Irina Higgins. 2022. Solving math word problems with process-and outcome-based feedback. arXiv preprint arXiv:2211.14275 (2022)."},{"key":"e_1_3_2_1_56_1","volume-title":"Math-shepherd: Verify and reinforce llms step-by-step without human annotations. arXiv preprint arXiv:2312.08935","author":"Wang Peiyi","year":"2023","unstructured":"Peiyi Wang, Lei Li, Zhihong Shao, RX Xu, Damai Dai, Yifei Li, Deli Chen, Yu Wu, and Zhifang Sui. 2023. Math-shepherd: Verify and reinforce llms step-by-step without human annotations. arXiv preprint arXiv:2312.08935 (2023)."},{"key":"e_1_3_2_1_57_1","volume-title":"Towards Hierarchical Multi-Step Reward Models for Enhanced Reasoning in Large Language Models. arXiv preprint arXiv:2503.13551","author":"Wang Teng","year":"2025","unstructured":"Teng Wang, Zhangyi Jiang, Zhenqi He, Wenhan Yang, Yanan Zheng, Zeyu Li, Zifan He, Shenyang Tong, and Hailei Gong. 2025. Towards Hierarchical Multi-Step Reward Models for Enhanced Reasoning in Large Language Models. arXiv preprint arXiv:2503.13551 (2025)."},{"key":"e_1_3_2_1_58_1","volume-title":"Aakanksha Chowdhery, and Denny Zhou.","author":"Wang Xuezhi","year":"2022","unstructured":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc Le, Ed Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou. 2022. Self consistency improves chain of thought reasoning in language models. arXiv preprint arXiv:2203.11171 (2022)."},{"key":"e_1_3_2_1_59_1","volume-title":"Faster and Better LLMs via Latency-Aware Test-Time Scaling. arXiv preprint arXiv:2505.19634","author":"Wang Zili","year":"2025","unstructured":"Zili Wang, Tianyu Zhang, Haoli Bai, Lu Hou, Xianzhi Yu, Wulong Liu, Shiming Xiang, and Lei Zhu. 2025. Faster and Better LLMs via Latency-Aware Test-Time Scaling. arXiv preprint arXiv:2505.19634 (2025)."},{"key":"e_1_3_2_1_60_1","volume-title":"xpu: Efficient Scheduling of Agentic LLM Workloads on Heterogeneous SoC. arXiv preprint arXiv:2506.24045","author":"Wei Xinming","year":"2025","unstructured":"Xinming Wei, Jiahao Zhang, Haoran Li, Jiayu Chen, Rui Qu, Maoliang Li, Xiang Chen, and Guojie Luo. 2025. Agent. xpu: Efficient Scheduling of Agentic LLM Workloads on Heterogeneous SoC. arXiv preprint arXiv:2506.24045 (2025)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695948"},{"key":"e_1_3_2_1_62_1","volume-title":"Inference scaling laws: An empirical analysis of compute optimal inference for problem-solving with language models. arXiv preprint arXiv:2408.00724","author":"Wu Yangzhen","year":"2024","unstructured":"Yangzhen Wu, Zhiqing Sun, Shanda Li, Sean Welleck, and Yiming Yang. 2024. Inference scaling laws: An empirical analysis of compute optimal inference for problem-solving with language models. arXiv preprint arXiv:2408.00724 (2024)."},{"key":"e_1_3_2_1_63_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et al. 2025. Qwen3 technical report. arXiv preprint arXiv:2505.09388 (2025)."},{"key":"e_1_3_2_1_64_1","unstructured":"An Yang Beichen Zhang Binyuan Hui Bofei Gao Bowen Yu Chengpeng Li Dayiheng Liu Jianhong Tu Jingren Zhou Junyang Lin et al. 2024. Qwen2. 5-math technical report: Toward mathematical expert model via self-improvement. arXiv preprint arXiv:2409.12122 (2024)."},{"key":"e_1_3_2_1_65_1","volume-title":"Reasonflux: Hierarchical llm reasoning via scaling thought templates. arXiv preprint arXiv:2502.06772","author":"Yang Ling","year":"2025","unstructured":"Ling Yang, Zhaochen Yu, Bin Cui, and Mengdi Wang. 2025. Reasonflux: Hierarchical llm reasoning via scaling thought templates. arXiv preprint arXiv:2502.06772 (2025)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3649329.3658473"},{"key":"e_1_3_2_1_67_1","volume-title":"Multi-Task, Multi-Dialogue Settings. arXiv preprint arXiv:2502.11007","author":"Yuan Liangqi","year":"2025","unstructured":"Liangqi Yuan, Dong-Jun Han, ShiqiangWang, and Christopher G Brinton. 2025. Local-Cloud Inference Offloading for LLMs in Multi-Modal, Multi-Task, Multi-Dialogue Settings. arXiv preprint arXiv:2502.11007 (2025)."},{"key":"e_1_3_2_1_68_1","unstructured":"Qiyuan Zhang Fuyuan Lyu Zexu Sun Lei Wang Weixu Zhang Wenyue Hua Haolun Wu Zhihan Guo Yufei Wang Niklas Muennighoff et al. 2025. A Survey on Test-Time Scaling in Large Language Models: What How Where and How Well? arXiv preprint arXiv:2503.24235 (2025)."},{"key":"e_1_3_2_1_69_1","volume-title":"Phitchaya Mangpo Phothilimthana, and Zhihao Jia","author":"Zhang Zhihao","year":"2024","unstructured":"Zhihao Zhang, Alan Zhu, Lijie Yang, Yihua Xu, Lanting Li, Phitchaya Mangpo Phothilimthana, and Zhihao Jia. 2024. Accelerating retrieval-augmented language model serving with speculation. arXiv preprint arXiv:2401.14021 (2024)."},{"key":"e_1_3_2_1_70_1","volume-title":"Genprm: Scaling test-time compute of process reward models via generative reasoning. arXiv preprint arXiv:2504.00891","author":"Zhao Jian","year":"2025","unstructured":"Jian Zhao, Runze Liu, Kaiyan Zhang, Zhimu Zhou, Junqi Gao, Dong Li, Jiafei Lyu, Zhouyi Qian, Biqing Qi, Xiu Li, et al. 2025. Genprm: Scaling test-time compute of process reward models via generative reasoning. arXiv preprint arXiv:2504.00891 (2025)."},{"key":"e_1_3_2_1_71_1","volume-title":"Jeff Huang, Cody Hao Yu, Shiyi Cao, Christos Kozyrakis, Ion Stoica, Joseph E Gonzalez, et al.","author":"Zheng Lianmin","year":"2024","unstructured":"Lianmin Zheng, Liangsheng Yin, Zhiqiang Xie, Chuyue Livia Sun, Jeff Huang, Cody Hao Yu, Shiyi Cao, Christos Kozyrakis, Ion Stoica, Joseph E Gonzalez, et al. 2024. Sglang: Efficient execution of structured language model programs. Advances in neural information processing systems 37 (2024), 62557--62583."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3719664"},{"key":"e_1_3_2_1_73_1","volume-title":"Batchllm: Optimizing large batched llm inference with global prefix sharing and throughput-oriented token batching. arXiv preprint arXiv:2412.03594","author":"Zheng Zhen","year":"2024","unstructured":"Zhen Zheng, Xin Ji, Taosong Fang, Fanghao Zhou, Chuanjie Liu, and Gang Peng. 2024. Batchllm: Optimizing large batched llm inference with global prefix sharing and throughput-oriented token batching. arXiv preprint arXiv:2412.03594 (2024)."},{"key":"e_1_3_2_1_74_1","volume-title":"Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24)","author":"Zhong Yinmin","year":"2024","unstructured":"Yinmin Zhong, Shengyu Liu, Junda Chen, Jianbo Hu, Yibo Zhu, Xuanzhe Liu, Xin Jin, and Hao Zhang. 2024. DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving. In Proceedings of the 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI '24). https:\/\/www.usenix.org\/system\/files\/osdi24-zhong-yinmin.pdf"}],"event":{"name":"ASPLOS '26: 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems","location":"Pittsburgh PA USA","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","SIGPLAN ACM Special Interest Group on Programming Languages","SIGARCH ACM Special Interest Group on Computer Architecture","SIGBED ACM Special Interest Group on Embedded Systems"]},"container-title":["Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T14:08:28Z","timestamp":1773583708000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779212.3790161"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,22]]},"references-count":74,"alternative-id":["10.1145\/3779212.3790161","10.1145\/3779212"],"URL":"https:\/\/doi.org\/10.1145\/3779212.3790161","relation":{},"subject":[],"published":{"date-parts":[[2026,3,22]]},"assertion":[{"value":"2026-03-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}