{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T00:54:17Z","timestamp":1767315257353,"version":"3.48.0"},"publisher-location":"Cham","reference-count":40,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032101846","type":"print"},{"value":"9783032101853","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-10185-3_39","type":"book-chapter","created":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T00:49:26Z","timestamp":1767314966000},"page":"494-506","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Efficient Attention on\u00a0Microcontrollers via\u00a0Linearized Transformer Blocks"],"prefix":"10.1007","author":[{"given":"Alberto","family":"Ancilotto","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edoardo","family":"Castagnini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Elisabetta","family":"Farella","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"39_CR1","doi-asserted-by":"crossref","unstructured":"Ancilotto, A., Paissan, F., Farella, E.: XiNet: efficient neural networks for tinyML. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 16968\u201316977 (2023)","DOI":"10.1109\/ICCV51070.2023.01556"},{"key":"39_CR2","first-page":"26831","volume":"34","author":"Y Bai","year":"2021","unstructured":"Bai, Y., Mei, J., Yuille, A.L., Xie, C.: Are transformers more robust than CNNs? Adv. Neural. Inf. Process. Syst. 34, 26831\u201326843 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"39_CR3","unstructured":"Cai, H., Gan, C., Zhu, L., Han, S.: Tiny transfer learning: towards memory-efficient on-device learning. arXiv preprint arXiv:2007.11622 (2020)"},{"key":"39_CR4","doi-asserted-by":"crossref","unstructured":"Cai, H., Li, J., Hu, M., Gan, C., Han, S.: Efficientvit: lightweight multi-scale attention for high-resolution dense prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17302\u201317313 (2023)","DOI":"10.1109\/ICCV51070.2023.01587"},{"key":"39_CR5","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"39_CR6","doi-asserted-by":"crossref","unstructured":"Chollet, F.: Xception: Deep learning with depthwise separable convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1251\u20131258 (2017)","DOI":"10.1109\/CVPR.2017.195"},{"key":"39_CR7","unstructured":"Dosovitskiy, A., et\u00a0al, L.B.: An image is worth 16x16 words: transformers for image recognition at scale. In: International Conference on Learning Representations (2021)"},{"key":"39_CR8","unstructured":"Google: tflite-micro. https:\/\/github.com\/tensorflow\/tflite-micro"},{"key":"39_CR9","doi-asserted-by":"crossref","unstructured":"Gu, P., Zhang, Y., Wang, C., Chen, D.: Convformer: combining CNN and transformer for medical image segmentation. In: 2023 IEEE 20th International Symposium on Biomedical Imaging (ISBI), pp.\u00a01\u20135 (2022)","DOI":"10.1109\/ISBI53787.2023.10230838"},{"key":"39_CR10","unstructured":"Hendrycks, D., Dietterich, T.: Benchmarking neural network robustness to common corruptions and perturbations. In: International Conference on Learning Representations (2019). https:\/\/openreview.net\/forum?id=HJz6tiCqYm"},{"key":"39_CR11","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., Song, D.: Natural adversarial examples. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15262\u201315271 (2021)","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"39_CR12","doi-asserted-by":"crossref","unstructured":"Howard, A., Sandler, M., et\u00a0al.: Searching for mobilenetv3. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2019)","DOI":"10.1109\/ICCV.2019.00140"},{"key":"39_CR13","unstructured":"Howard, A.G.E.A.: Mobilenets: efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861 (2017)"},{"key":"39_CR14","unstructured":"Jocher, G., Qiu, J., Chaurasia, A.: Ultralytics YOLO (2023). https:\/\/github.com\/ultralytics\/ultralytics"},{"key":"39_CR15","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N., Fleuret, F.: Transformers are RNNS: fast autoregressive transformers with linear attention. In: International Conference on Machine Learning, pp. 5156\u20135165. PMLR (2020)"},{"key":"39_CR16","unstructured":"kendryte: nncase. https:\/\/github.com\/kendryte\/nncase"},{"key":"39_CR17","unstructured":"Khanam, R., Hussain, M.: Yolov11: an overview of the key architectural enhancements (2024). https:\/\/api.semanticscholar.org\/CorpusID:273532028"},{"key":"39_CR18","unstructured":"Lai, L., Suda, N., Chandra, V.: CMSIS-NN: efficient neural network kernels for arm cortex-m CPUS. arXiv preprint arXiv:1801.06601 (2018)"},{"key":"39_CR19","unstructured":"Li, Y., et al.: Efficientformer: vision transformers at mobilenet speed. In: Proceedings of the 2022 Conference on Neural Information Processing Systems arXiv:2206.01191 (2022). https:\/\/api.semanticscholar.org\/CorpusID:249282517"},{"key":"39_CR20","unstructured":"Lin, J., Chen, W.M., Cai, H., Gan, C., Han, S.: Mcunetv2: memory-efficient patch-based inference for tiny deep learning. In: Neural Information Processing Systems (2021)"},{"key":"39_CR21","unstructured":"Lin, J., Chen, W.M., Lin, Y., Cohn, J., Gan, C., Han, S.: Mcunet: tiny deep learning on IoT devices. arXiv preprint arXiv:2007.10319 (2020)"},{"key":"39_CR22","first-page":"22941","volume":"35","author":"J Lin","year":"2022","unstructured":"Lin, J., Zhu, L., Chen, W.M., Wang, W.C., Gan, C., Han, S.: On-device training under 256kb memory. Adv. Neural. Inf. Process. Syst. 35, 22941\u201322954 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"39_CR23","doi-asserted-by":"crossref","unstructured":"Liu, W., et al.: SSD: single shot multibox detector. arXiv preprint arXiv:1512.02325 (2016)","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"39_CR24","doi-asserted-by":"crossref","unstructured":"Liu, X., Peng, H., Zheng, N., Yang, Y., Hu, H., Yuan, Y.: Efficientvit: memory efficient vision transformer with cascaded group attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14420\u201314430 (2023)","DOI":"10.1109\/CVPR52729.2023.01386"},{"key":"39_CR25","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"39_CR26","unstructured":"Lv, W., Zhao, Y., Chang, Q., Huang, K., Wang, G., Liu, Y.: Rt-detrv2: improved baseline with bag-of-freebies for real-time detection transformer. arXiv preprint arXiv:2407.17140 (2024)"},{"key":"39_CR27","doi-asserted-by":"crossref","unstructured":"Ma, N., Zhang, X., Zheng, H.T., Sun, J.: Shufflenet v2: practical guidelines for efficient CNN architecture design. In: Proceedings of the European Conference on Computer Vision (ECCV) (2018)","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"39_CR28","unstructured":"Mehta, S., Rastegari, M.: Separable self-attention for mobile vision transformers. Trans. Mach. Learn. Res. (2023). https:\/\/openreview.net\/forum?id=tBl4yBEjKi"},{"key":"39_CR29","unstructured":"ST Microelectronics: X-cube-ai. https:\/\/www.st.com\/en\/embedded-software\/x-cube-ai.html"},{"key":"39_CR30","doi-asserted-by":"publisher","unstructured":"Paissan, F., Ancilotto, A., Farella, E.: Phinets: a scalable backbone for low-power AI at the edge. ACM Trans. Embed. Comput. Syst. (2022). https:\/\/doi.org\/10.1145\/3510832","DOI":"10.1145\/3510832"},{"key":"39_CR31","doi-asserted-by":"crossref","unstructured":"Paissan, F., et al.: Structured sparse back-propagation for lightweight on-device continual learning on microcontroller units. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops (2024)","DOI":"10.1109\/CVPRW63382.2024.00222"},{"key":"39_CR32","doi-asserted-by":"crossref","unstructured":"Qin, D., et al.: Mobilenetv4 - universal models for the mobile ecosystem. In: European Conference on Computer Vision (ECCV) arXiv:2404.10518 (2024)","DOI":"10.1007\/978-3-031-73661-2_5"},{"key":"39_CR33","unstructured":"Radford, A., Kim, J.W., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning (2021)"},{"key":"39_CR34","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.C.: Mobilenetv2: inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"39_CR35","unstructured":"Tan, M., Le, Q.: Efficientnet: rethinking model scaling for convolutional neural networks. In: International Conference on Machine Learning, pp. 6105\u20136114. PMLR (2019)"},{"key":"39_CR36","doi-asserted-by":"crossref","unstructured":"Tan, M., Pang, R., Le, Q.V.: Efficientdet: scalable and efficient object detection. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"39_CR37","unstructured":"Technologies, G.: Automated design intelligence with gapflow. https:\/\/greenwaves-technologies.com\/automated-design-intelligence-with-gapflow-overview-and-benchmarks-on-gap8\/"},{"key":"39_CR38","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: International Conference on Machine Learning, pp. 10347\u201310357. PMLR (2021)"},{"key":"39_CR39","unstructured":"Wang, S., Li, B.Z., Khabsa, M., Fang, H., Ma, H.: Linformer: self-attention with linear complexity. arXiv preprint arXiv:2006.04768 (2020)"},{"key":"39_CR40","doi-asserted-by":"crossref","unstructured":"Zhao, Y., et al.: DETRS beat YOLOS on real-time object detection (2023)","DOI":"10.1109\/CVPR52733.2024.01605"}],"container-title":["Lecture Notes in Computer Science","Image Analysis and Processing \u2013 ICIAP 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-10185-3_39","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T00:49:29Z","timestamp":1767314969000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-10185-3_39"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032101846","9783032101853"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-10185-3_39","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ICIAP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Image Analysis and Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Rome","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iciap2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.iciap.org\/home","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}