{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T15:48:24Z","timestamp":1784821704776,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,8]],"date-time":"2024-10-08T00:00:00Z","timestamp":1728345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,8]]},"DOI":"10.1145\/3703412.3703415","type":"proceedings-article","created":{"date-parts":[[2025,3,5]],"date-time":"2025-03-05T11:49:51Z","timestamp":1741175391000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["TinySV: Speaker Verification in TinyML with On-device Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5964-5685","authenticated-orcid":false,"given":"Massimo","family":"Pavan","sequence":"first","affiliation":[{"name":"DEIB, Politecnico di Milano, Milan, IT"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0344-3741","authenticated-orcid":false,"given":"Gioele","family":"Mombelli","sequence":"additional","affiliation":[{"name":"DEIB, Politecnico di Milano, Milan, IT"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4592-0416","authenticated-orcid":false,"given":"Francesco","family":"Sinacori","sequence":"additional","affiliation":[{"name":"Infineon Technologies Italia s.r.l, Milan, IT"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7828-7687","authenticated-orcid":false,"given":"Manuel","family":"Roveri","sequence":"additional","affiliation":[{"name":"DEIB, Politecnico di Milano, Milan, IT"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,3,5]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"[n. d.]. Infineon Modus Toolbox. https:\/\/www.infineon.com\/cms\/en\/design-support\/tools\/sdk\/modustoolbox-software\/. Accessed: 2023-10-17."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPSN.2018.00049"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","unstructured":"Cesare Alippi and Manuel Roveri. 2017. The (Not) Far-Away Path to Smart Cyber-Physical Systems: An Information-Centric Framework. Computer 50 4 (April 2017) 38\u201347. 10.1109\/MC.2017.111Conference Name: Computer.","DOI":"10.1109\/MC.2017.111"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","unstructured":"Mattia Antonini Miguel Pincheira Massimo Vecchio and Fabio Antonelli. 2023. An Adaptable and Unsupervised TinyML Anomaly Detection System for Extreme Industrial Environments. Sensors 23 4 (2023). 10.3390\/s23042344","DOI":"10.3390\/s23042344"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC58517.2023.10317337"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Yu-hsin Chen Ignacio\u00a0Lopez Moreno Tara Sainath Mirk\u00f3 Visontai Raziel Alvarez and Carolina Parada. 2015. Locally-connected and convolutional neural networks for small footprint speaker recognition. (2015).","DOI":"10.21437\/Interspeech.2015-297"},{"key":"e_1_3_3_2_8_2","unstructured":"Aakanksha Chowdhery Pete Warden Jonathon Shlens Andrew Howard and Rocky Rhodes. 2019. Visual Wake Words Dataset. arXiv:https:\/\/arXiv.org\/abs\/1906.05721 [cs eess] (June 2019). http:\/\/arxiv.org\/abs\/1906.05721 arXiv:https:\/\/arXiv.org\/abs\/1906.05721."},{"key":"e_1_3_3_2_9_2","unstructured":"Robert David Jared Duke Advait Jain Vijay\u00a0Janapa Reddi Nat Jeffries Jian Li Nick Kreeger Ian Nappier Meghna Natraj Shlomi Regev Rocky Rhodes Tiezhen Wang and Pete Warden. 2021. TensorFlow Lite Micro: Embedded Machine Learning on TinyML Systems. Proceedings of the 4 th MLSys Conference San Jose CA USA (2021) 12."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","unstructured":"Najim Dehak Patrick Kenny R. Dehak Pierre Dumouchel and Pierre Ouellet. 2011. Front-End Factor Analysis for Speaker Verification. Audio Speech and Language Processing IEEE Transactions on 19 (06 2011) 788 \u2013 798. 10.1109\/TASL.2010.2064307","DOI":"10.1109\/TASL.2010.2064307"},{"key":"e_1_3_3_2_11_2","unstructured":"Simone Disabato and Roveri. 2021. Tiny Machine Learning for Concept Drift. arXiv:https:\/\/arXiv.org\/abs\/2107.14759 (jul 2021). http:\/\/arxiv.org\/abs\/2107.14759 arXiv:https:\/\/arXiv.org\/abs\/2107.14759."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2018.8489276"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"crossref","unstructured":"Simone Disabato and Manuel Roveri. 2020. Incremental On-Device Tiny Machine Learning. (2020) 7.","DOI":"10.1145\/3417313.3429378"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472652"},{"key":"e_1_3_3_2_15_2","unstructured":"Andrew\u00a0G. Howard Menglong Zhu Bo Chen Dmitry Kalenichenko Weijun Wang Tobias Weyand Marco Andreetto and Hartwig Adam. 2017. MobileNets: Efficient Convolutional Neural Networks for Mobile Vision Applications. http:\/\/arxiv.org\/abs\/1704.04861 Number: arXiv:https:\/\/arXiv.org\/abs\/1704.04861 arXiv:1704.04861 [cs]."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"crossref","unstructured":"Amna Irum and Ahmad Salman. 2019. Speaker verification using deep neural networks: A. International Journal of Machine Learning and Computing 9 1 (2019).","DOI":"10.18178\/ijmlc.2019.9.1.760"},{"key":"e_1_3_3_2_17_2","unstructured":"Benoit Jacob Skirmantas Kligys Bo Chen Menglong Zhu Matthew Tang Andrew Howard Hartwig Adam and Dmitry Kalenichenko. 2017. Quantization and Training of Neural Networks for Efficient Integer-Arithmetic-Only Inference. arXiv:https:\/\/arXiv.org\/abs\/1712.05877 [cs stat] (Dec. 2017). http:\/\/arxiv.org\/abs\/1712.05877 arXiv:https:\/\/arXiv.org\/abs\/1712.05877 version: 1."},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Anthony Larcher Kong\u00a0Aik Lee Bin Ma and Haizhou Li. 2014. Text-dependent speaker verification: Classifiers databases and RSR2015. Speech Communication 60 (2014) 56\u201377.","DOI":"10.1016\/j.specom.2014.03.001"},{"key":"e_1_3_3_2_19_2","unstructured":"Ji Lin Ligeng Zhu Wei-Ming Chen Wei-Chen Wang Chuang Gan and Song Han. 2022. On-Device Training Under 256KB Memory. http:\/\/arxiv.org\/abs\/2206.15472 arXiv:https:\/\/arXiv.org\/abs\/2206.15472 [cs]."},{"key":"e_1_3_3_2_20_2","unstructured":"Jiayi Liu Samarth Tripathi Unmesh Kurup and Mohak Shah. 2020. Pruning Algorithms to Accelerate Convolutional Neural Networks for Edge Applications: A Survey. arXiv:https:\/\/arXiv.org\/abs\/2005.04275 [cs stat] (May 2020). http:\/\/arxiv.org\/abs\/2005.04275 arXiv:https:\/\/arXiv.org\/abs\/2005.04275."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN55064.2022.9892925"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","unstructured":"Massimo Pavan Eugeniu Ostrovan Armando Caltabiano and Manuel Roveri. 2023. TyBox: an automatic design and code-generation toolbox for TinyML incremental on-device learning. ACM Transactions on Embedded Computing Systems (June 2023) 3604566. 10.1145\/3604566","DOI":"10.1145\/3604566"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054423"},{"key":"e_1_3_3_2_25_2","volume-title":"Intelligence at the Extreme Edge: A Survey on Reformable TinyML","author":"Rajapakse Visal","year":"2022","unstructured":"Visal Rajapakse, Ishan Karunanayake, and Nadeem Ahmed. 2022. Intelligence at the Extreme Edge: A Survey on Reformable TinyML. Technical Report arXiv:https:\/\/arXiv.org\/abs\/2204.00827.arXiv. http:\/\/arxiv.org\/abs\/2204.00827 arXiv:https:\/\/arXiv.org\/abs\/2204.00827 [cs, eess] type: article."},{"key":"e_1_3_3_2_26_2","unstructured":"Vikram Ramanathan. [n. d.]. Online On-device MCU Transfer Learning. ([n. d.]) 7."},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","unstructured":"Leonardo Ravaglia Manuele Rusci Davide Nadalini Alessandro Capotondi Francesco Conti Luca Benini and Luca Benini. 2021. A TinyML Platform for On-Device Continual Learning with Quantized Latent Replays. IEEE Journal on Emerging and Selected Topics in Circuits and Systems (2021) 1\u20131. 10.1109\/JETCAS.2021.3121554","DOI":"10.1109\/JETCAS.2021.3121554"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","unstructured":"Partha\u00a0Pratim Ray. 2021. A review on TinyML: State-of-the-art and prospects. Journal of King Saud University - Computer and Information Sciences (Nov. 2021). 10.1016\/j.jksuci.2021.11.019","DOI":"10.1016\/j.jksuci.2021.11.019"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"crossref","unstructured":"Haoyu Ren Darko Anicic and Thomas Runkler. 2021. TinyOL: TinyML with Online-Learning on Microcontrollers. arXiv:https:\/\/arXiv.org\/abs\/2103.08295 [cs eess] (April 2021). http:\/\/arxiv.org\/abs\/2103.08295 arXiv:https:\/\/arXiv.org\/abs\/2103.08295.","DOI":"10.1109\/IJCNN52387.2021.9533927"},{"key":"e_1_3_3_2_30_2","first-page":"23","volume-title":"Computational Intelligence and Data Analytics: Proceedings of ICCIDA 2022","author":"Roveri Manuel","year":"2022","unstructured":"Manuel Roveri. 2022. Is tiny deep learning the new deep learning? In Computational Intelligence and Data Analytics: Proceedings of ICCIDA 2022. Springer, 23\u201339."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"Manuele Rusci and Tinne Tuytelaars. 2023. Few-Shot Open-Set Learning for On-Device Customization of KeyWord Spotting Systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.02161 (2023).","DOI":"10.21437\/Interspeech.2023-1904"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"crossref","unstructured":"Tara Sainath and Carolina Parada. 2015. Convolutional neural networks for small-footprint keyword spotting. (2015).","DOI":"10.21437\/Interspeech.2015-352"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","unstructured":"R. Sanchez-Iborra and A.\u00a0F. Skarmeta. 2020. TinyML-Enabled Frugal Smart Objects: Challenges and Opportunities. IEEE Circuits and Systems Magazine 20 3 (2020) 4\u201318. 10.1109\/MCAS.2020.3005467Conference Name: IEEE Circuits and Systems Magazine.","DOI":"10.1109\/MCAS.2020.3005467"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/SWC50871.2021.00023"},{"key":"e_1_3_3_2_35_2","unstructured":"Mingxing Tan and Quoc\u00a0V. Le. 2019. EfficientNet: Rethinking Model Scaling for Convolutional Neural Networks. CoRR abs\/1905.11946 (2019). arXiv:https:\/\/arXiv.org\/abs\/1905.11946http:\/\/arxiv.org\/abs\/1905.11946"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","unstructured":"Youzhi Tu Weiwei Lin and Man-Wai Mak. 2022. A Survey on Text-Dependent and Text-Independent Speaker Verification. IEEE Access PP (01 2022) 1\u20131. 10.1109\/ACCESS.2022.3206541","DOI":"10.1109\/ACCESS.2022.3206541"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","DOI":"10.1109\/icassp.2014.6854363"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"e_1_3_3_2_39_2","unstructured":"Pete Warden. 2018. Speech Commands: A Dataset for Limited-Vocabulary Speech Recognition. arXiv:https:\/\/arXiv.org\/abs\/1804.03209 [cs] (April 2018). http:\/\/arxiv.org\/abs\/1804.03209 arXiv:https:\/\/arXiv.org\/abs\/1804.03209."},{"key":"e_1_3_3_2_40_2","unstructured":"Pete Warden. 2020. Why isn\u2019t there more training on the edge? Online. https:\/\/petewarden.com\/2022\/09\/06\/why-isnt-there-more-training-on-the-edge\/"},{"key":"e_1_3_3_2_41_2","volume-title":"TinyML: machine learning with TensorFlow Lite on Arduino and ultra-low-power microcontrollers (first edition ed.)","author":"Warden Pete","year":"2020","unstructured":"Pete Warden and Daniel Situnayake. 2020. TinyML: machine learning with TensorFlow Lite on Arduino and ultra-low-power microcontrollers (first edition ed.). O\u2019Reilly, Bejing Boston Farnham Sebastopol Tokyo."},{"key":"e_1_3_3_2_42_2","unstructured":"YistLin. 2023. Generalized End-to-End Loss for Speaker Verification. https:\/\/github.com\/yistLin\/dvector."},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","unstructured":"Xinyu Yuan Guanyu Li Jiao Han Di Wang and Zhi Tiankai. 2021. Overview of the development of speaker recognition. Journal of Physics: Conference Series 1827 (03 2021) 012125. 10.1088\/1742-6596\/1827\/1\/012125","DOI":"10.1088\/1742-6596\/1827\/1\/012125"},{"key":"e_1_3_3_2_44_2","unstructured":"Yundong Zhang Naveen Suda Liangzhen Lai and Vikas Chandra. 2017. Hello edge: Keyword spotting on microcontrollers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1711.07128 (2017)."}],"event":{"name":"AIMLSystems 2024: The 4th International Conference on AI-ML Systems","location":"Baton Rouge Louisiana USA","acronym":"AIMLSystems 2024"},"container-title":["Proceedings of the 4th International Conference on AI-ML Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3703412.3703415","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3703412.3703415","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:07Z","timestamp":1750295887000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3703412.3703415"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,8]]},"references-count":43,"alternative-id":["10.1145\/3703412.3703415","10.1145\/3703412"],"URL":"https:\/\/doi.org\/10.1145\/3703412.3703415","relation":{},"subject":[],"published":{"date-parts":[[2024,10,8]]},"assertion":[{"value":"2025-03-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}