{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:30:29Z","timestamp":1787495429789,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_8","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:34Z","timestamp":1787492794000},"page":"112-125","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Window-Diffusion: Accelerating Diffusion Language Model Inference with\u00a0Windowed Token Pruning and\u00a0Caching"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-3468-9280","authenticated-orcid":false,"given":"Fengrui","family":"Zuo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7636-2446","authenticated-orcid":false,"given":"Zhiwei","family":"Ke","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2079-8359","authenticated-orcid":false,"given":"Yiming","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0882-5047","authenticated-orcid":false,"given":"Cheng","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2240-6672","authenticated-orcid":false,"given":"Wenqi","family":"Lou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7281-1203","authenticated-orcid":false,"given":"Teng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8391-5526","authenticated-orcid":false,"given":"Lei","family":"Gong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9403-5575","authenticated-orcid":false,"given":"Chao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8360-3143","authenticated-orcid":false,"given":"Xuehai","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"8_CR1","unstructured":"Arriola, M., et al.: Block diffusion: interpolating between autoregressive and diffusion language models. In: International Conference on Learning Representations, vol. 2025, pp. 50726\u201350753 (2025)"},{"key":"8_CR2","unstructured":"Austin, J., Johnson, D.D., Ho, J., Tarlow, D., Van Den, Berg, R.: Structured denoising diffusion models in discrete state-spaces. In: Advances in Neural Information Processing Systems, pp. 17981\u201317993 (2021)"},{"key":"8_CR3","unstructured":"Austin, J., et al.: Program synthesis with large language models. arXiv preprint arXiv:2108.07732 (2021)"},{"key":"8_CR4","unstructured":"Chen, M., et al.: Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)"},{"key":"8_CR5","unstructured":"Cobbe, K., et al.: Training verifiers to solve math word problems. arXiv preprint arXiv:2110.14168 (2021)"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Dao, T., Fu, D., Ermon, S., Rudra, A., R\u2019e, C.: Flashattention: fast and memory-efficient exact attention with io-awareness. In: Advances in Neural Information Processing Systems, vol. 35, pp. 16344\u201316359 (2022)","DOI":"10.52202\/068431-1189"},{"key":"8_CR7","doi-asserted-by":"crossref","unstructured":"Dong, J., et al.: Moe-sched: enabling efficient FPGA deployment of mixture-of-experts vision transformers via coordinated scheduling. IEEE Trans. Very Large Scale Integr. VLSI Syst. 34, 104\u2013117 (2026)","DOI":"10.1109\/TVLSI.2025.3604705"},{"key":"8_CR8","unstructured":"Feng, G.: Theoretical benefit and limitation of diffusion language model. arXiv abs\/2502.09622 (2025)"},{"key":"8_CR9","unstructured":"Hendrycks, D., et al.: Measuring mathematical problem solving with the math dataset. arXiv preprint arXiv:2103.03874 (2021)"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with pagedattention. In: Proceedings of the 29th Symposium on Operating Systems Principles, pp. 611\u2013626 (2023)","DOI":"10.1145\/3600006.3613165"},{"key":"8_CR11","unstructured":"Leviathan, Y., Kalman, M., Matias, Y.: Fast inference from transformers via speculative decoding. In: Proceedings of the 40th International Conference on Machine Learning. ICML 2023. JMLR.org (2023)"},{"key":"8_CR12","unstructured":"Li, T., Chen, M., Guo, B., Shen, Z.: A survey on diffusion language models. arXiv preprint arXiv:2508.10875 (2025)"},{"key":"8_CR13","unstructured":"Lin, H., et al.: Quantization meets dLLMs: a systematic study of post-training quantization for diffusion LLMs. arXiv preprint arXiv:2508.14896 (2025)"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"Liu, F., et al.: Earth: an efficient MOE accelerator with entropy-aware speculative prefetch and result reuse. In: Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 2, pp. 633\u2013646 (2026)","DOI":"10.1145\/3779212.3790155"},{"issue":"10","key":"8_CR15","first-page":"1","volume":"58","author":"J Liu","year":"2026","unstructured":"Liu, J., et al.: A survey on inference optimization techniques for mixture of experts models. ACM Comput. Surv. 58(10), 1\u201337 (2026)","journal-title":"ACM Comput. Surv."},{"key":"8_CR16","unstructured":"Liu, Y., et al.: Sequential diffusion language models. arXiv preprint arXiv:2509.24007 (2025)"},{"key":"8_CR17","unstructured":"Liu, Z., et al.: dLLM-cache: accelerating diffusion large language models with adaptive caching. arXiv preprint arXiv:2506.06295 (2025)"},{"key":"8_CR18","unstructured":"Lou, A., Meng, C., Ermon, S.: Discrete diffusion modeling by estimating the ratios of the data distribution. In: Proceedings of the 41st International Conference on Machine Learning, pp. 32819\u201332848 (2024)"},{"key":"8_CR19","first-page":"149009","volume":"38","author":"X Ma","year":"2026","unstructured":"Ma, X., Yu, R., Fang, G., Wang, X.: dKV-cache: the cache for diffusion language models. Adv. Neural. Inf. Process. Syst. 38, 149009\u2013149033 (2026)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"8_CR20","unstructured":"Nie, S., et al.: Scaling up masked diffusion models on text. In: International Conference on Learning Representations, vol. 2025, pp. 82974\u201382997 (2025)"},{"key":"8_CR21","unstructured":"Nie, S.: Large language diffusion models. arXiv abs\/2502.09992 (2025)"},{"key":"8_CR22","unstructured":"Peng, H., et al.: How efficient are diffusion language models? A critical examination of efficiency evaluation practices. arXiv preprint arXiv:2510.18480 (2025)"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Sahoo, S.S., et al.: Simple and effective masked diffusion language models. In: Advances in Neural Information Processing Systems, pp. 130136\u2013130184 (2024)","DOI":"10.52202\/079017-4135"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Song, Y., Mi, Z., Xie, H., Chen, H.: Powerinfer: fast large language model serving with a consumer-grade GPU. In: Proceedings of the ACM SIGOPS 30th Symposium on Operating Systems Principles, pp. 590\u2013606 (2024)","DOI":"10.1145\/3694715.3695964"},{"key":"8_CR25","unstructured":"Wang, X., et al.: Diffusion LLMs can do faster-than-AR inference via discrete diffusion forcing. arXiv abs\/2508.09192 (2025)"},{"issue":"5","key":"8_CR26","doi-asserted-by":"publisher","first-page":"337","DOI":"10.1109\/LES.2025.3600563","volume":"17","author":"H Wen","year":"2025","unstructured":"Wen, H., et al.: QLlama: an FPGA-based microscaling quantization accelerator for energy-efficient llama2 inference. IEEE Embed. Syst. Lett. 17(5), 337\u2013340 (2025)","journal-title":"IEEE Embed. Syst. Lett."},{"key":"8_CR27","unstructured":"Wu, C., et al.: Fast-dLLM: training-free acceleration of diffusion LLM by enabling KV cache and parallel decoding. arXiv preprint arXiv:2505.22618 (2025)"},{"key":"8_CR28","unstructured":"Xu, C., Yang, D.: Dllmquant: quantizing diffusion-based large language models. arXiv preprint arXiv:2508.14090 (2025)"},{"key":"8_CR29","unstructured":"Ye, J., et al.: Dream 7b: diffusion large language models. arXiv preprint arXiv:2508.15487 (2025)"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Yu, Z., et al.: Cambricon-LLM: a chiplet-based hybrid architecture for on-device inference of 70b LLM. In: 57th IEEE\/ACM International Symposium on Microarchitecture (MICRO), pp. 1474\u20131488. IEEE (2024)","DOI":"10.1109\/MICRO61859.2024.00108"},{"key":"8_CR31","doi-asserted-by":"crossref","unstructured":"Zeng, S., et al.: Flightllm: efficient large language model inference with a complete mapping flow on FPGAs. In: Proceedings of the 2024 ACM\/SIGDA International Symposium on Field Programmable Gate Arrays, pp. 223\u2013234 (2024)","DOI":"10.1145\/3626202.3637562"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Zhang, S., Zhao, Y., Geng, L., Cohan, A., Tuan, L.A., Zhao, C.: Diffusion vs. autoregressive language models: a text embedding perspective. In: Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, pp. 4273\u20134303 (2025)","DOI":"10.18653\/v1\/2025.emnlp-main.213"},{"key":"8_CR33","unstructured":"Zhang, T., Li, Z., Yan, X., Qin, H., Guo, Y., Zhang, Y.: Quant-dLLM: post-training extreme low-bit quantization for diffusion large language models. arXiv preprint arXiv:2510.03274 (2025)"},{"key":"8_CR34","doi-asserted-by":"crossref","unstructured":"Zheng, Z., et al.: Lora: a latency-oriented recurrent architecture for large language model on multi-FPGA platform with communication optimization. IEEE Trans. Comput.-Aided Des. Integr. Circ. Syst. (2025)","DOI":"10.1109\/TCAD.2025.3629537"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Zhu, F., et al.: LLADA 1.5: variance-reduced preference optimization for large language diffusion models. arXiv preprint arXiv:2505.19223 (2025)","DOI":"10.18653\/v1\/2026.acl-long.524"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:40Z","timestamp":1787492800000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The technical solution and experimental design presented in this paper were independently completed by the authors. AI tools were used solely for language polishing and formatting optimization, and did not participate in the development of the research ideas or core content.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}