{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:30:12Z","timestamp":1787495412761,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":41,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_7","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:12Z","timestamp":1787492772000},"page":"97-111","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MOCAP: Wafer-Scale-Chip-Oriented Memory-Orchestrated Chunked Pipelining Framework for\u00a0Prefill-Only LLM Inference"],"prefix":"10.1007","author":[{"given":"Zichuan","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huizheng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuheng","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haonan","family":"Zuo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Taiquan","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinyi","family":"Deng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shouyi","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"7_CR1","doi-asserted-by":"crossref","unstructured":"Du, K., Wang, B., Zhang, C., Cheng, Y., Lan, Q., Sang, H.: Prefillonly: an inference engine for prefill-only workloads in large language model applications. In: Proceedings of the ACM SIGOPS 31st Symposium on Operating Systems Principles, pp. 399\u2013414 (2025)","DOI":"10.1145\/3731569.3764834"},{"key":"7_CR2","unstructured":"Fang, J., Wang, H., Yang, Q., Kong, D., Dai, X., Deng, J.: PALM: a efficient performance simulator for tiled accelerators with large-scale model training. arXiv preprint arXiv:2406.03868 (2024)"},{"key":"7_CR3","unstructured":"He, C., Huang, Y., Mu, P., Miao, Z., Xue, J., Ma, L.: WaferLLM: large language model inference at wafer scale. arXiv preprint arXiv:2502.04563 (2025)"},{"issue":"1","key":"7_CR4","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1109\/MCAS.2024.3349669","volume":"24","author":"Y Hu","year":"2024","unstructured":"Hu, Y., Lin, X., Wang, H., He, Z., Yu, X., Zhang, J., et al.: Wafer-scale computing: advancements, challenges, and future perspectives. IEEE Circuits Syst. Mag. 24(1), 52\u201381 (2024)","journal-title":"IEEE Circuits Syst. Mag."},{"key":"7_CR5","unstructured":"Huang, Y., et al.: Gpipe: efficient training of giant neural networks using pipeline parallelism. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"7_CR6","unstructured":"Kim, T., Kim, H., Yu, G.I., Chun, B.G.: Bpipe: memory-balanced pipeline parallelism for training large language models. In: International Conference on Machine Learning, pp. 16639\u201316653. PMLR (2023)"},{"key":"7_CR7","doi-asserted-by":"crossref","unstructured":"Lee, S., Kim, B., Park, J., Jeon, D.: Clat: a clustering-based attention transformer accelerator for low-latency text generation in LLMs. IEEE Trans. Circ. Syst. I Reg. Pap. (2025)","DOI":"10.1109\/TCSI.2025.3576232"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Li, C., et al.: Rethermal: co-design of thermal-aware static and dynamic scheduling for LLM training on liquid-cooled wafer-scale chips. In: 2026 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 1\u201315. IEEE (2026)","DOI":"10.1109\/HPCA68181.2026.11408476"},{"key":"7_CR9","unstructured":"Li, Z., Zhuang, S., Guo, S., Zhuo, D., Zhang, H., Song, D.: Terapipe: token-level pipeline parallelism for training large-scale language models. In: International Conference on Machine Learning, pp. 6543\u20136552. PMLR (2021)"},{"key":"7_CR10","doi-asserted-by":"crossref","unstructured":"Lie, S.: Cerebras architecture deep dive: first look inside the HW\/SW co-design for deep learning: cerebras systems. In: 2022 IEEE Hot Chips 34 Symposium (HCS), pp. 1\u201334. IEEE Computer Society (2022)","DOI":"10.1109\/HCS55958.2022.9895479"},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: Ouroboros: wafer-scale sram cim with token-grained pipelining for large language model inference. In: Proceedings of the 31st ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 2, pp. 1349\u20131365 (2026)","DOI":"10.1145\/3779212.3790197"},{"key":"7_CR12","doi-asserted-by":"crossref","unstructured":"Lu, L., et al.: Sanger: a co-design framework for enabling sparse attention using reconfigurable architecture. In: MICRO-54: 54th Annual IEEE\/ACM International Symposium on Microarchitecture, pp. 977\u2013991 (2021)","DOI":"10.1145\/3466752.3480125"},{"key":"7_CR13","unstructured":"Meta: Meta-Llama-3-70B (2024). https:\/\/huggingface.co\/meta-llama\/Meta-Llama-3-70B"},{"key":"7_CR14","unstructured":"Meta AI: Llama-3.1-405B (2024). https:\/\/huggingface.co\/meta-llama\/Llama-3.1-405B"},{"key":"7_CR15","unstructured":"Mistral AI: Mistral-Large-Instruct-2407 (2024). https:\/\/huggingface.co\/mistralai\/Mistral-Large-Instruct-2407"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Moon, S., Li, M., Chen, G.K., Knag, P.C., Krishnamurthy, R.K., Seok, M.: T-REX: a 68-to-567$$\\mu $$s\/token 0.41-to-3.95 $$\\mu $$J\/token transformer accelerator with reduced external memory access and enhanced hardware utilization in 16nm FinFET. In: 2025 IEEE International Solid-State Circuits Conference (ISSCC). vol.\u00a068, pp. 406\u2013408. IEEE (2025)","DOI":"10.1109\/ISSCC49661.2025.10904793"},{"key":"7_CR17","unstructured":"NVIDIA: NVIDIA DGX Vera Rubin NVL72 (2026). https:\/\/www.nvidia.com\/en-us\/data-center\/dgx-vera-rubin-nvl72\/"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Pal, S., Liu, J., Alam, I., Cebry, N., Suhail, H., Bu, S.: Designing a 2048-chiplet, 14336-core waferscale processor. In: 2021 58th ACM\/IEEE Design Automation Conference (DAC), pp. 1183\u20131188. IEEE (2021)","DOI":"10.1109\/DAC18074.2021.9586194"},{"key":"7_CR19","unstructured":"Qwen: Qwen3-235B-A22B (2025). https:\/\/huggingface.co\/Qwen\/Qwen3-235B-A22B"},{"key":"7_CR20","unstructured":"Rashidi, S., Won, W., Srinivasan, S., Gupta, P., Krishna, T.: FRED: flexible reduction-distribution interconnect and communication implementation for wafer-scale distributed training of DNN models. arXiv preprint arXiv:2406.19580 (2024)"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Shih, P.C., Su, A.J., Tam, K.H., Huang, T.C., Chuang, K., Yeh, J.: SoW-X: a novel system-on-wafer technology for next generation AI server application. In: 2025 IEEE 75th Electronic Components and Technology Conference (ECTC), pp. 1\u20136. IEEE (2025)","DOI":"10.1109\/ECTC51687.2025.00005"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Talpes, E., Williams, D., Sarma, D.D.: Dojo: the microarchitecture of Tesla\u2019s exa-scale computer. In: 2022 IEEE Hot Chips 34 Symposium (HCS), pp. 1\u201328. IEEE Computer Society (2022)","DOI":"10.1109\/HCS55958.2022.9895534"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Tang, X., Hou, J., Jiang, D., Wei, T., Liu, J., Deng, J.: Moentwine: unleashing the potential of wafer-scale chips for large-scale expert parallel inference. In: 2026 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 1\u201315. IEEE (2026)","DOI":"10.1109\/HPCA68181.2026.11408594"},{"key":"7_CR24","doi-asserted-by":"crossref","unstructured":"Wang, H., Fang, J., Tang, X., Yue, Z., Li, J., Qin, Y.: SOFA: a compute-memory optimized sparsity accelerator via cross-stage coordinated tiling. In: 2024 57th IEEE\/ACM International Symposium on Microarchitecture (MICRO), pp. 1247\u20131263. IEEE (2024)","DOI":"10.1109\/MICRO61859.2024.00093"},{"key":"7_CR25","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: Pade: a predictor-free sparse attention accelerator via unified execution and stage fusion. In: 2026 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 1\u201319. IEEE (2026)","DOI":"10.1109\/HPCA68181.2026.11408448"},{"key":"7_CR26","doi-asserted-by":"crossref","unstructured":"Wang, H., Wang, H., Wei, S., Hu, Y., Yin, S.: Bitstopper: an efficient transformer attention accelerator via stage-fusion and early termination. In: 2026 31st Asia and South Pacific Design Automation Conference (ASP-DAC), pp. 162\u2013169. IEEE (2026)","DOI":"10.1109\/ASP-DAC66049.2026.11420780"},{"key":"7_CR27","doi-asserted-by":"crossref","unstructured":"Wang, H., Wang, H., Wei, S., Hu, Y., Yin, S.: Lapa: log-domain prediction-driven dynamic sparsity accelerator for transformer model. In: 2026 31st Asia and South Pacific Design Automation Conference (ASP-DAC), pp. 154\u2013161. IEEE (2026)","DOI":"10.1109\/ASP-DAC66049.2026.11420725"},{"key":"7_CR28","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: BETA: a bit-grained transformer attention accelerator with efficient early termination. IEEE Trans. Circ. Syst. II Exp. Briefs (2025)","DOI":"10.1109\/TCSII.2025.3596228"},{"key":"7_CR29","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: Watos: efficient LLM training strategies and architecture co-exploration for wafer-scale chip. In: 2026 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 1\u201319. IEEE (2026)","DOI":"10.1109\/HPCA68181.2026.11408457"},{"key":"7_CR30","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: MCBP: a memory-compute efficient LLM inference accelerator leveraging bit-slice-enabled sparsity and repetitiveness. In: Proceedings of the 58th IEEE\/ACM International Symposium on Microarchitecture\u00ae, pp. 1592\u20131608 (2025)","DOI":"10.1145\/3725843.3756037"},{"key":"7_CR31","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: Designing spatial architectures for sparse attention: star accelerator via cross-stage tiling. IEEE Trans. Comput. (2025)","DOI":"10.1109\/TC.2025.3648055"},{"key":"7_CR32","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: Temp: a memory efficient physical-aware tensor partition-mapping framework on wafer-scale chips. In: 2026 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 1\u201318. IEEE (2026)","DOI":"10.1109\/HPCA68181.2026.11408568"},{"key":"7_CR33","doi-asserted-by":"crossref","unstructured":"Wang, H., Yang, Q., Wei, T., Yu, X., Li, C., Fang, J.: TMAC: training-targeted mapping and architecture co-exploration for wafer-scale chips. Integr. Circ. Syst. (2024)","DOI":"10.23919\/ICS.2024.3515003"},{"key":"7_CR34","doi-asserted-by":"crossref","unstructured":"Wei, T., Wang, H., Wang, Z., Yin, S., Hu, Y.: Spatial-aware orchestration of LLM attention on waferscale chips. In: International Symposium on Advanced Parallel Processing Technologies, pp. 386\u2013391. Springer (2025)","DOI":"10.1007\/978-981-95-1021-4_29"},{"key":"7_CR35","doi-asserted-by":"crossref","unstructured":"Won, W., Heo, T., Rashidi, S., Sridharan, S., Srinivasan, S., Krishna, T.: Astra-sim2.0: modeling hierarchical networks and disaggregated systems for large-model training at scale. In: 2023 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS), pp. 283\u2013294. IEEE (2023)","DOI":"10.1109\/ISPASS57527.2023.00035"},{"key":"7_CR36","doi-asserted-by":"crossref","unstructured":"Xu, Z., Kong, D., Liu, J., Jiang, D., Dai, X., Deng, J.: Face: fully overlapped pd scheduling and multi-level architecture co-exploration on wafer. In: 2026 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp. 1\u201316. IEEE (2026)","DOI":"10.1109\/HPCA68181.2026.11408518"},{"key":"7_CR37","doi-asserted-by":"crossref","unstructured":"Xu, Z., Kong, D., Liu, J., Li, J., Hou, J., Dai, X.: WSC-LLM: efficient LLM service and architecture co-exploration for wafer-scale chips. In: Proceedings of the 52nd Annual International Symposium on Computer Architecture, pp. 1\u201317 (2025)","DOI":"10.1145\/3695053.3731101"},{"key":"7_CR38","doi-asserted-by":"crossref","unstructured":"Yang, Q., Wei, T., Guan, S., Li, C., Shang, H., Deng, J.: PD constraint-aware physical\/logical topology co-design for network on wafer. In: Proceedings of the 52nd Annual International Symposium on Computer Architecture, pp. 49\u201364 (2025)","DOI":"10.1145\/3695053.3731045"},{"key":"7_CR39","doi-asserted-by":"crossref","unstructured":"Yue, Z., Wang, H., Fang, J., Deng, J., Lu, G., Tu, F.: Exploiting similarity opportunities of emerging vision AI models on hybrid bonding architecture. In: 2024 ACM\/IEEE 51st Annual International Symposium on Computer Architecture (ISCA), pp. 396\u2013409. IEEE (2024)","DOI":"10.1109\/ISCA59077.2024.00037"},{"key":"7_CR40","unstructured":"Zhai, J., Liao, L., Liu, X., Wang, Y., Li, R., Cao, X.: Actions speak louder than words: trillion-parameter sequential transducers for generative recommendations. arXiv preprint arXiv:2402.17152 (2024)"},{"key":"7_CR41","doi-asserted-by":"crossref","unstructured":"Zhang, C., You, Y., Wang, N., Park, J., Zhang, L.: Generative AI through CAS lens: an integrated overview of algorithmic optimizations, architectural advances, and automated designs. IEEE J. Emerg. Sel. Top. Circ. Syst. (2025)","DOI":"10.1109\/JETCAS.2025.3575272"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:18Z","timestamp":1787492778000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Artificial intelligence tools, if used, were used only for language organization and figure\/table polishing, and did not participate in the generation of the core ideas or technical content.","order":1,"name":"Ethics","label":"Declaration on the Use of Artificial Intelligence","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}