{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T07:54:25Z","timestamp":1775634865801,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":14,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819584017","type":"print"},{"value":"9789819584024","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-8402-4_22","type":"book-chapter","created":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T07:17:43Z","timestamp":1775632663000},"page":"413-432","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["FPAMM: Fine-Grained Pipeline Architecture Accelerator for\u00a0the\u00a0Novel Transformer Architecture - Monarch Mixer"],"prefix":"10.1007","author":[{"given":"Hanyuan","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingche","family":"Lai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xingyun","family":"Qi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Puguang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenqi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yihang","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,9]]},"reference":[{"key":"22_CR1","doi-asserted-by":"publisher","unstructured":"Bai, Y., et al.: LTrans-OPU: a low-latency FPGA-based overlay processor for transformer networks. In: 2023 33rd International Conference on Field-Programmable Logic and Applications (FPL), pp. 283\u2013287. IEEE, Gothenburg, Sweden (2023). https:\/\/doi.org\/10.1109\/FPL60245.2023.00048. https:\/\/ieeexplore.ieee.org\/document\/10296349\/","DOI":"10.1109\/FPL60245.2023.00048"},{"key":"22_CR2","doi-asserted-by":"publisher","unstructured":"Chen, H., et al.: Understanding the potential of FPGA-based spatial acceleration for large language model inference. ACM Trans. Reconfigurable Technol. Syst. 18(1), 1\u201329 (2024). https:\/\/doi.org\/10.1145\/3656177. https:\/\/dl.acm.org\/doi\/10.1145\/3656177","DOI":"10.1145\/3656177"},{"key":"22_CR3","first-page":"4479","volume":"33","author":"L Chi","year":"2020","unstructured":"Chi, L., Jiang, B., Mu, Y.: Fast fourier convolution. Adv. Neural. Inf. Process. Syst. 33, 4479\u20134488 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"22_CR4","doi-asserted-by":"crossref","unstructured":"Chollet, F.: Xception: deep learning with depthwise separable convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1251\u20131258 (2017)","DOI":"10.1109\/CVPR.2017.195"},{"key":"22_CR5","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv abs\/1810.04805 (2019)"},{"key":"22_CR6","first-page":"77546","volume":"36","author":"D Fu","year":"2023","unstructured":"Fu, D., et al.: Monarch mixer: a simple sub-quadratic GEMM-based architecture. Adv. Neural. Inf. Process. Syst. 36, 77546\u201377603 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"22_CR7","unstructured":"Fu, D.Y., Kumbong, H., Nguyen, E., Re, C.: Flashfftconv: efficient convolutions for long sequences with tensor cores. In: The Twelfth International Conference on Learning Representations"},{"key":"22_CR8","doi-asserted-by":"publisher","unstructured":"Liu, Z., Li, G., Cheng, J.: Hardware acceleration of fully quantized BERT for efficient natural language processing. In: 2021 Design, Automation & Test in Europe Conference & Exhibition (DATE), pp. 513\u2013516. IEEE, Grenoble, France (2021). https:\/\/doi.org\/10.23919\/DATE51398.2021.9474043. https:\/\/ieeexplore.ieee.org\/document\/9474043\/","DOI":"10.23919\/DATE51398.2021.9474043"},{"key":"22_CR9","unstructured":"Minaee, S., et al.: Large language models: a survey. arxiv (2024). arXiv preprint arXiv:2402.06196"},{"key":"22_CR10","doi-asserted-by":"publisher","unstructured":"Plagwitz, P., Hannig, F., Teich, J.: TRAC: compilation-based design of transformer accelerators for FPGAs. In: 2022 32nd International Conference on Field-Programmable Logic and Applications (FPL), pp. 17\u201323. IEEE, Belfast, United Kingdom (2022). https:\/\/doi.org\/10.1109\/FPL57034.2022.00015. https:\/\/ieeexplore.ieee.org\/document\/10035242\/","DOI":"10.1109\/FPL57034.2022.00015"},{"key":"22_CR11","doi-asserted-by":"publisher","unstructured":"Qi, P., Song, Y., Peng, H., Huang, S., Zhuge, Q., Sha, E.H.M.: Accommodating transformer onto FPGA: coupling the balanced model compression and FPGA-implementation optimization. In: Proceedings of the 2021 Great Lakes Symposium on VLSI, pp. 163\u2013168. ACM, Virtual Event USA (2021). https:\/\/doi.org\/10.1145\/3453688.3461739. https:\/\/dl.acm.org\/doi\/10.1145\/3453688.3461739","DOI":"10.1145\/3453688.3461739"},{"key":"22_CR12","doi-asserted-by":"crossref","unstructured":"Sarkar, R., Liang, H., Fan, Z., Wang, Z., Hao, C.: Edge-MoE: memory-efficient multi-task vision transformer architecture with task-level sparsity via mixture-of-experts. In: 2023 IEEE\/ACM International Conference on Computer Aided Design (ICCAD), pp. 1\u20139. IEEE (2023)","DOI":"10.1109\/ICCAD57390.2023.10323651"},{"key":"22_CR13","doi-asserted-by":"crossref","unstructured":"Vaidya, M., Sukumaran-Rajam, A., Rountev, A., Sadayappan, P.: Comprehensive accelerator-dataflow co-design optimization for convolutional neural networks. In: 2022 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO), pp. 325\u2013335. IEEE (2022)","DOI":"10.1109\/CGO53902.2022.9741281"},{"key":"22_CR14","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"}],"container-title":["Lecture Notes in Computer Science","Algorithms and Architectures for Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-8402-4_22","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T07:17:49Z","timestamp":1775632669000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-8402-4_22"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819584017","9789819584024"],"references-count":14,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-8402-4_22","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"9 April 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICA3PP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Algorithms and Architectures for Parallel Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Zhengzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 October 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ica3pp2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ieee-cybermatics.org\/2025\/ica3pp\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}