{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:29:22Z","timestamp":1787495362309,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":31,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_21","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:45:54Z","timestamp":1787492754000},"page":"313-329","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["BitFly: A Low-Bit Mixed-Precision Acceleration Framework for\u00a0Edge RISC-V Vector Processors"],"prefix":"10.1007","author":[{"given":"Zixuan","family":"Zeng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chen","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhe","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"21_CR1","doi-asserted-by":"publisher","first-page":"96892","DOI":"10.1109\/ACCESS.2023.3294111","volume":"11","author":"Y Abadade","year":"2023","unstructured":"Abadade, Y., Temouden, A., Bamoumen, H., Benamar, N., Chtouki, Y., Hafid, A.S.: A comprehensive survey on tinyml. IEEE Access 11, 96892\u201396922 (2023). https:\/\/doi.org\/10.1109\/ACCESS.2023.3294111","journal-title":"IEEE Access"},{"key":"21_CR2","unstructured":"Andrej Karpathy: Karpathy\/llama2.c: Inference Llama 2 in one file of pure C. GitHub repository (2025). https:\/\/github.com\/karpathy\/llama2.c"},{"key":"21_CR3","doi-asserted-by":"crossref","unstructured":"Armeniakos, G., Maras, A., Xydis, S., Soudris, D.: Mixed-precision Neural Networks on RISC-V Cores: ISA extensions for Multi-Pumped Soft SIMD Operations 10(1145\/3676536), 3676840 (2024)","DOI":"10.1145\/3676536.3676840"},{"issue":"8","key":"21_CR4","doi-asserted-by":"publisher","first-page":"1253","DOI":"10.1109\/TC.2021.3066883","volume":"70","author":"A Burrello","year":"2021","unstructured":"Burrello, A., Garofalo, A., Bruschi, N., Tagliavini, G., Rossi, D., Conti, F.: Dory: automatic end-to-end deployment of real-world DNNs on low-cost IoT MCUs. IEEE Trans. Comput. 70(8), 1253\u20131268 (2021). https:\/\/doi.org\/10.1109\/TC.2021.3066883","journal-title":"IEEE Trans. Comput."},{"key":"21_CR5","doi-asserted-by":"publisher","unstructured":"Cavalcante, M., Schuiki, F., Zaruba, F., Schaffner, M., Benini, L.: Ara: a 1-GHz+ scalable and energy-efficient RISC-V vector processor with multiprecision floating-point support in 22-nm FD-SOI. IEEE Trans. Very Large Scale Integration (VLSI) Syst. 28(2), 530\u2013543 (2020). https:\/\/doi.org\/10.1109\/TVLSI.2019.2950087","DOI":"10.1109\/TVLSI.2019.2950087"},{"key":"21_CR6","doi-asserted-by":"publisher","unstructured":"Cavalcante, M., W\u00fcthrich, D., Perotti, M., Riedel, S., Benini, L.: Spatz: a compact vector processing unit for high-performance and energy-efficient shared-L1 clusters. In: Proceedings of the 41st IEEE\/ACM International Conference on Computer-Aided Design, pp. 1\u20139 (2022). https:\/\/doi.org\/10.1145\/3508352.3549367","DOI":"10.1145\/3508352.3549367"},{"key":"21_CR7","doi-asserted-by":"publisher","unstructured":"Conti, F., et al.: Marsellus: a heterogeneous RISC-v AI-Iot end-node soc with 2\u20138 b DNN acceleration and 30%-boost adaptive body biasing. IEEE J. Solid-State Circuits 59(1), 128\u2013142 (2024). https:\/\/doi.org\/10.1109\/JSSC.2023.3318301","DOI":"10.1109\/JSSC.2023.3318301"},{"key":"21_CR8","doi-asserted-by":"publisher","unstructured":"Davide\u00a0Schiavone, P., et al.: Slow and steady wins the race? a comparison of ultra-low-power RISC-v cores for internet-of-things applications. In: 2017 27th International Symposium on Power and Timing Modeling, Optimization and Simulation (PATMOS), pp.\u00a01\u20138 (2017). https:\/\/doi.org\/10.1109\/PATMOS.2017.8106976","DOI":"10.1109\/PATMOS.2017.8106976"},{"key":"21_CR9","doi-asserted-by":"publisher","unstructured":"Fornt, J., et al.: Mix-GEMM: extending RISC-V CPUs for energy-efficient mixed-precision DNN inference using binary segmentation. IEEE Trans. Comput. 74(2), 582\u2013596 (2025). https:\/\/doi.org\/10.1109\/TC.2024.3500369","DOI":"10.1109\/TC.2024.3500369"},{"issue":"3","key":"21_CR10","doi-asserted-by":"publisher","first-page":"1489","DOI":"10.1109\/TETC.2021.3072337","volume":"9","author":"A Garofalo","year":"2021","unstructured":"Garofalo, A., Tagliavini, G., Conti, F., Benini, L., Rossi, D.: XPULPNN: enabling energy efficient and flexible inference of quantized neural networks on RISC-v based IoT end nodes. IEEE Trans. Emerg. Top. Comput. 9(3), 1489\u20131505 (2021). https:\/\/doi.org\/10.1109\/TETC.2021.3072337","journal-title":"IEEE Trans. Emerg. Top. Comput."},{"key":"21_CR11","doi-asserted-by":"publisher","unstructured":"Garofalo, A., et al.: DARKSIDE: a heterogeneous RISC-V compute cluster for extreme-edge on-chip DNN inference and training. IEEE Open J. Solid-State Circuits Soc. 2, 231\u2013243 (2022). https:\/\/doi.org\/10.1109\/OJSSCS.2022.3210082","DOI":"10.1109\/OJSSCS.2022.3210082"},{"key":"21_CR12","doi-asserted-by":"publisher","unstructured":"Genc, H., et al.: Gemmini: enabling systematic deep-learning architecture evaluation via full-stack integration (2021). https:\/\/doi.org\/10.48550\/arXiv.1911.09925","DOI":"10.48550\/arXiv.1911.09925"},{"key":"21_CR13","unstructured":"ggml-org: ggml-org\/llama.cpp. GitHub repository (2025). https:\/\/github.com\/ggml-org\/llama.cpp"},{"key":"21_CR14","doi-asserted-by":"crossref","unstructured":"Gholami, A., Kim, S., Dong, Z., Yao, Z., Mahoney, M.W., Keutzer, K.: A survey of quantization methods for efficient neural network inference (2021). https:\/\/arxiv.org\/abs\/2103.13630","DOI":"10.1201\/9781003162810-13"},{"key":"21_CR15","doi-asserted-by":"publisher","unstructured":"Gong, R., et al.: A survey of low-bit large language models: basics, systems, and algorithms (2024). https:\/\/doi.org\/10.48550\/arXiv.2409.16694","DOI":"10.48550\/arXiv.2409.16694"},{"key":"21_CR16","doi-asserted-by":"publisher","unstructured":"Ma, L., Sun, M., Shen, Z.: FBI-LLM: scaling up fully binarized LLMs from scratch via autoregressive distillation (2024). https:\/\/doi.org\/10.48550\/arXiv.2407.07093","DOI":"10.48550\/arXiv.2407.07093"},{"key":"21_CR17","unstructured":"Minaee, S., et al.: Large language models: a survey (2025).https:\/\/arxiv.org\/abs\/2402.06196"},{"key":"21_CR18","doi-asserted-by":"publisher","unstructured":"Ottavi, G., et al.: Dustin: a 16-cores parallel ultra-low-power cluster with 2b-to-32b fully flexible bit-precision and vector lockstep execution mode. IEEE Trans. Circuits Syst. I Regul. Pap. 70(6), 2450\u20132463 (2023). https:\/\/doi.org\/10.1109\/TCSI.2023.3254810","DOI":"10.1109\/TCSI.2023.3254810"},{"key":"21_CR19","doi-asserted-by":"publisher","unstructured":"Perotti, M., Cavalcante, M., Andri, R., Cavigelli, L., Benini, L.: Ara2: exploring single- and multi-core vector processing with an efficient RVV 1.0 compliant open-source processor. IEEE Trans. Comput. 73(7), 1822\u20131836 (2024). https:\/\/doi.org\/10.1109\/TC.2024.3388896","DOI":"10.1109\/TC.2024.3388896"},{"issue":"10","key":"21_CR20","doi-asserted-by":"publisher","first-page":"3732","DOI":"10.1109\/TCSII.2023.3292579","volume":"70","author":"M Perotti","year":"2023","unstructured":"Perotti, M., Cavalcante, M., Ottaviano, A., Liu, J., Benini, L.: Yun: an open-source, 64-Bit RISC-V-based vector processor with multi-precision integer and floating-point support in 65-nm CMOS. IEEE Trans. Circuits Syst. II Express Briefs 70(10), 3732\u20133736 (2023). https:\/\/doi.org\/10.1109\/TCSII.2023.3292579","journal-title":"IEEE Trans. Circuits Syst. II Express Briefs"},{"key":"21_CR21","doi-asserted-by":"publisher","unstructured":"Perotti, M., Cavalcante, M., Wistoff, N., Andri, R., Cavigelli, L., Benini, L.: A new ARA for vector computing: an open source highly efficient RISC-V v 1.0 vector processor design. In: 2022 IEEE 33rd International Conference on Application-specific Systems, Architectures and Processors (ASAP), pp. 43\u201351 (2022). https:\/\/doi.org\/10.1109\/ASAP54787.2022.00017","DOI":"10.1109\/ASAP54787.2022.00017"},{"key":"21_CR22","doi-asserted-by":"publisher","unstructured":"Rossi, D., et al.: Vega: a ten-core soc for IoT endnodes with DNN acceleration and cognitive wake-up from MRAM-based state-retentive sleep mode. IEEE J. Solid-State Circuits 57(1), 127\u2013139 (2022). https:\/\/doi.org\/10.1109\/JSSC.2021.3114881","DOI":"10.1109\/JSSC.2021.3114881"},{"issue":"4","key":"21_CR23","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1109\/MCE.2024.3518761","volume":"14","author":"S Sai","year":"2025","unstructured":"Sai, S., Prasad, M., Dashore, G., Chamola, V., Sikdar, B.: On-device generative AI: the need, architectures, and challenges. IEEE Consumer Electron. Magazine 14(4), 21\u201332 (2025). https:\/\/doi.org\/10.1109\/MCE.2024.3518761","journal-title":"IEEE Consumer Electron. Magazine"},{"key":"21_CR24","unstructured":"Shang, Y., Yuan, Z., Wu, Q., Dong, Z.: Pb-LLM: partially binarized large language models (2023). https:\/\/arxiv.org\/abs\/2310.00034"},{"key":"21_CR25","doi-asserted-by":"publisher","unstructured":"Wang, C., Fang, C., Wu, X., Wang, Z., Lin, J.: SPEED: a scalable RISC-V vector processor enabling efficient multi-precision DNN inference. IEEE Trans. Very Large Scale Integration (VLSI) Syst. 33(1), 207\u2013220 (2025). https:\/\/doi.org\/10.1109\/TVLSI.2024.3466224","DOI":"10.1109\/TVLSI.2024.3466224"},{"key":"21_CR26","unstructured":"Wang, H., et al.: Bitnet: scaling 1-bit transformers for large language models (2023). https:\/\/arxiv.org\/abs\/2310.11453"},{"key":"21_CR27","doi-asserted-by":"publisher","unstructured":"Wang, H., Ma, S., Wei, F.: BitNet v2: native 4-bit activations with Hadamard transformation for 1-bit LLMs (2025). https:\/\/doi.org\/10.48550\/arXiv.2504.18415","DOI":"10.48550\/arXiv.2504.18415"},{"key":"21_CR28","doi-asserted-by":"publisher","unstructured":"Wang, J., et al.: Bitnet.cpp: Efficient Edge Inference for Ternary LLMs (2025). https:\/\/doi.org\/10.48550\/arXiv.2502.11880","DOI":"10.48550\/arXiv.2502.11880"},{"key":"21_CR29","doi-asserted-by":"publisher","unstructured":"Xu, Y., et al.: OneBit: towards extremely low-bit large language models (2024). https:\/\/doi.org\/10.48550\/arXiv.2402.11295","DOI":"10.48550\/arXiv.2402.11295"},{"issue":"11","key":"21_CR30","doi-asserted-by":"publisher","first-page":"1845","DOI":"10.1109\/TC.2020.3027900","volume":"70","author":"F Zaruba","year":"2021","unstructured":"Zaruba, F., Schuiki, F., Hoefler, T., Benini, L.: Snitch: a tiny pseudo dual-issue processor for area and energy efficient execution of floating-point intensive workloads. IEEE Trans. Comput. 70(11), 1845\u20131860 (2021). https:\/\/doi.org\/10.1109\/TC.2020.3027900","journal-title":"IEEE Trans. Comput."},{"key":"21_CR31","unstructured":"Zheng, Y., Chen, Y., Qian, B., Shi, X., Shu, Y., Chen, J.: A review on edge large language models: design, execution, and applications (2025). https:\/\/arxiv.org\/abs\/2410.11845"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:45:56Z","timestamp":1787492756000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}