{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:31:13Z","timestamp":1787495473009,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":24,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_23","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:42Z","timestamp":1787492802000},"page":"343-358","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Helios: Scalable Multi-accelerator FPGA Architecture for\u00a0Efficient Inference of\u00a0Transformer-Based Models"],"prefix":"10.1007","author":[{"given":"Yanwei","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qun","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tianle","family":"Mai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Keqiu","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"23_CR1","doi-asserted-by":"crossref","unstructured":"Bjerregaard, T., Mahadevan, S.: A survey of research and practices of network-on-chip. ACM Computing Surveys (CSUR) 38(1), 1\u2013es (2006)","DOI":"10.1145\/1132952.1132953"},{"issue":"4","key":"23_CR2","doi-asserted-by":"publisher","first-page":"822","DOI":"10.3390\/electronics12040822","volume":"12","author":"Y Chen","year":"2023","unstructured":"Chen, Y., Li, T., Chen, X., et al.: High-frequency systolic array-based transformer accelerator on field programmable gate arrays. Electronics 12(4), 822 (2023)","journal-title":"Electronics"},{"key":"23_CR3","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations (ICLR) (2021), https:\/\/arxiv.org\/abs\/2010.11929"},{"key":"23_CR4","unstructured":"Firestone, D., Putnam, A., Mundkur, S., Chiou, D., et\u00a0al.: Azure accelerated networking:$$\\{$$SmartNICs$$\\}$$ in the public cloud. In: 15th USENIX Symposium on Networked Systems Design and Implementation (NSDI 18). pp. 51\u201366 (2018)"},{"key":"23_CR5","unstructured":"Gouk, D., Lee, S., Kwon, M., Jung, M.: Direct access, high-performance memory disaggregation with directcxl. In: USENIX Annual Technical Conference, pp. 287\u2013294. (2022)"},{"issue":"3","key":"23_CR6","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1007\/s11554-024-01442-8","volume":"21","author":"H Hong","year":"2024","unstructured":"Hong, H., Choi, D., Kim, N., Lee, H., et al.: Survey of convolutional neural network accelerators on field-programmable gate array platforms: architectures and optimization techniques. J. Real-Time Image Proc. 21(3), 64 (2024)","journal-title":"J. Real-Time Image Proc."},{"key":"23_CR7","doi-asserted-by":"publisher","unstructured":"Lai, Y.H., Rong, H., Zheng, S., et\u00a0al.: Susy: a programming model for productive construction of high-performance systolic arrays on fpgas. In: Proceedings of the 39th International Conference on Computer-Aided Design. Association for Computing Machinery, New York, NY, USA (2020). https:\/\/doi.org\/10.1145\/3400302.3415644","DOI":"10.1145\/3400302.3415644"},{"key":"23_CR8","doi-asserted-by":"publisher","DOI":"10.7717\/peerj-cs.1166","volume":"8","author":"S Li","year":"2022","unstructured":"Li, S., Tao, Y., Tang, E., Xie, T., Chen, R.: A survey of field programmable gate array (fpga)-based graph convolutional neural network accelerators: challenges and opportunities. PeerJ Computer Science 8, e1166 (2022)","journal-title":"PeerJ Computer Science"},{"issue":"5","key":"23_CR9","doi-asserted-by":"publisher","DOI":"10.1049\/ell2.12751","volume":"59","author":"T Li","year":"2023","unstructured":"Li, T., Zhang, F., Xie, G., Fan, X., et al.: A high speed reconfigurable architecture for softmax and gelu in vision transformer. Electron. Lett. 59(5), e12751 (2023)","journal-title":"Electron. Lett."},{"key":"23_CR10","doi-asserted-by":"publisher","unstructured":"Li, Z., Sun, M., Lu, A., et\u00a0al.: Auto-vit-acc: An fpga-aware automatic acceleration framework for vision transformer with mixed-scheme quantization. In: 2022 32nd International Conference on Field-Programmable Logic and Applications (FPL). pp. 109\u2013116 (2022). https:\/\/doi.org\/10.1109\/FPL57034.2022.00027","DOI":"10.1109\/FPL57034.2022.00027"},{"key":"23_CR11","unstructured":"Li\u00a0Tianyang, Zhang\u00a0Fan, W.S.C.W.C.L.: Fpga-based unified accelerator for convolutional neural network and vision transformer. Journal of Electronics & Information Technology 46(6), 2663\u20132672 (2024)"},{"key":"23_CR12","doi-asserted-by":"publisher","unstructured":"Lu, S., Wang, M., Liang, S., Lin, J., Wang, Z.: Hardware accelerator for multi-head attention and position-wise feed-forward in the transformer. In: 2020 IEEE 33rd International System-on-Chip Conference (SOCC). pp. 84\u201389 (2020). https:\/\/doi.org\/10.1109\/SOCC49529.2020.9524802","DOI":"10.1109\/SOCC49529.2020.9524802"},{"key":"23_CR13","doi-asserted-by":"publisher","unstructured":"Ma, W., Yang, X., Zeng, S., Liu, T., et\u00a0al.: Cd-llm: A heterogeneous multi-fpga system for batched decoding of 70b+ llms using a compute-dedicated architecture. ACM Trans. Reconfigurable Technol. Syst. 19(1) (Mar 2026). https:\/\/doi.org\/10.1145\/3771288","DOI":"10.1145\/3771288"},{"key":"23_CR14","doi-asserted-by":"publisher","unstructured":"Marino, K., Zhang, P., Prasanna, V.K.: ME-ViT: A Single-Load Memory-Efficient FPGA Accelerator for Vision Transformers . In: 30th International Conference on High Performance Computing, Data, and Analytics (HiPC). pp. 213\u2013223. IEEE Computer Society, Los Alamitos, CA, USA (2023). https:\/\/doi.org\/10.1109\/HiPC58850.2023.00039","DOI":"10.1109\/HiPC58850.2023.00039"},{"key":"23_CR15","doi-asserted-by":"publisher","unstructured":"Nag, S., Datta, G., Kundu, S., Chandrachoodan, N., Beerel, P.A.: Vita: A vision transformer inference accelerator for edge applications. In: 2023 IEEE International Symposium on Circuits and Systems (ISCAS). pp.\u00a01\u20135 (2023). https:\/\/doi.org\/10.1109\/ISCAS46773.2023.10181988","DOI":"10.1109\/ISCAS46773.2023.10181988"},{"key":"23_CR16","unstructured":"Sun, M., Ma, H., Kang, G., Jiang, Y., Chen, T., Ma, X., Wang, Z., Wang, Y.: Vaqf: Fully automatic software-hardware co-design framework for low-bit vision transformer (2022), https:\/\/arxiv.org\/abs\/2201.06618"},{"issue":"1","key":"23_CR17","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1007\/s11390-017-1747-6","volume":"33","author":"X Tan","year":"2018","unstructured":"Tan, X., Shen, X.W., et al.: A non-stop double buffering mechanism for dataflow architecture. J. Comput. Sci. Technol. 33(1), 145\u2013157 (2018)","journal-title":"J. Comput. Sci. Technol."},{"key":"23_CR18","doi-asserted-by":"publisher","first-page":"168162","DOI":"10.1109\/ACCESS.2021.3135690","volume":"9","author":"AV Trusov","year":"2021","unstructured":"Trusov, A.V., Limonova, E.E., Nikolaev, D.P., Arlazarov, V.V.: p-im2col: Simple yet efficient convolution algorithm with flexibly controlled memory overhead. IEEE Access 9, 168162\u2013168184 (2021)","journal-title":"IEEE Access"},{"key":"23_CR19","doi-asserted-by":"crossref","unstructured":"Vasudevan, A., Anderson, A., Gregg, D.: Parallel multi channel convolution using general matrix multiplication. In: 28th international conference on application-specific systems, architectures and processors (ASAP), pp. 19\u201324. IEEE (2017)","DOI":"10.1109\/ASAP.2017.7995254"},{"key":"23_CR20","unstructured":"Vaswani, A., Shazeer, N., Parmar, N.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30, pp. 5998\u20136008. Curran Associates, Inc (2017)"},{"key":"23_CR21","doi-asserted-by":"crossref","unstructured":"Wang, J., Guo, L., Cong, J.: Autosa: A polyhedral compiler for high-performance systolic arrays on fpga. In: The 2021 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays, pp. 93\u2013104. (2021)","DOI":"10.1145\/3431920.3439292"},{"issue":"6","key":"23_CR22","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3549937","volume":"22","author":"W Ye","year":"2023","unstructured":"Ye, W., Zhou, X., Zhou, J., Chen, C., Li, K.: Accelerating attention mechanism on fpgas based on efficient reconfigurable systolic array. ACM Transactions on Embedded Computing Systems 22(6), 1\u201322 (2023)","journal-title":"ACM Transactions on Embedded Computing Systems"},{"key":"23_CR23","doi-asserted-by":"crossref","unstructured":"Zhang, W., Peng, X., Wu, H., et\u00a0al.: Design guidelines of rram based neural-processing-unit: A joint device-circuit-algorithm analysis. In: Proceedings of the 56th Annual Design Automation Conference 2019. pp.\u00a01\u20136 (2019)","DOI":"10.1145\/3316781.3317797"},{"key":"23_CR24","doi-asserted-by":"crossref","unstructured":"Zheng, J., Chen, G.: Looplynx: A scalable dataflow architecture for efficient llm inference (2025), https:\/\/arxiv.org\/abs\/2504.09561","DOI":"10.23919\/DATE64628.2025.10993078"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_23","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:46:44Z","timestamp":1787492804000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_23"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_23","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}