{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:05Z","timestamp":1779228365073,"version":"3.51.4"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,3,29]],"date-time":"2026-03-29T00:00:00Z","timestamp":1774742400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100006245","name":"Ministry of Science and Technology, Israel","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100006245","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000002","name":"National Institutes of Health","doi-asserted-by":"publisher","award":["MH134369"],"award-info":[{"award-number":["MH134369"]}],"id":[{"id":"10.13039\/100000002","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101986","type":"journal-article","created":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T07:29:25Z","timestamp":1774682965000},"page":"101986","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Open-vocabulary keyword spotting with hyper-matched filters for small footprint devices"],"prefix":"10.1016","volume":"100","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2513-0752","authenticated-orcid":false,"given":"Yael","family":"Segal-Feldman","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ann R.","family":"Bradlow","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Matthew","family":"Goldrick","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2332-5783","authenticated-orcid":false,"given":"Joseph","family":"Keshet","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101986_b1","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Small-footprint slimmable networks for keyword spotting","author":"Akhtar","year":"2023"},{"key":"10.1016\/j.csl.2026.101986_b2","series-title":"18th Annual Conference of the International-Speech-Communication-Association (INTERSPEECH 2017) Conference Location Stockholm, SWEDEN","first-page":"1606","article-title":"Convolutional recurrent neural networks for small-footprint keyword spotting","author":"Arik","year":"2017"},{"key":"10.1016\/j.csl.2026.101986_b3","doi-asserted-by":"crossref","unstructured":"Bluche, T., Gisselbrecht, T., 2020. Predicting detection filters for small footprint open-vocabulary keyword spotting. In: Proc. Interspeech. pp. 2552\u20132556.","DOI":"10.21437\/Interspeech.2020-1186"},{"key":"10.1016\/j.csl.2026.101986_b4","series-title":"Small-footprint open-vocabulary keyword spotting with quantized lstm networks","author":"Bluche","year":"2020"},{"key":"10.1016\/j.csl.2026.101986_b5","series-title":"2014 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"4087","article-title":"Small-footprint keyword spotting using deep neural networks","author":"Chen","year":"2014"},{"key":"10.1016\/j.csl.2026.101986_b6","series-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5236","article-title":"Query-by-example keyword spotting using long short-term memory networks","author":"Chen","year":"2015"},{"key":"10.1016\/j.csl.2026.101986_b7","series-title":"2022 IEEE Spoken Language Technology Workshop","first-page":"798","article-title":"Fleurs: Few-shot learning evaluation of universal representations of speech","author":"Conneau","year":"2023"},{"key":"10.1016\/j.csl.2026.101986_b8","series-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6853","article-title":"CNN-based spoken term detection and localization without dynamic programming","author":"Fuchs","year":"2021"},{"key":"10.1016\/j.csl.2026.101986_b9","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J., 2006. Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference on Machine Learning. pp. 369\u2013376.","DOI":"10.1145\/1143844.1143891"},{"key":"10.1016\/j.csl.2026.101986_b10","series-title":"Hypernetworks","author":"Ha","year":"2016"},{"key":"10.1016\/j.csl.2026.101986_b11","series-title":"2017 IEEE Automatic Speech Recognition and Understanding Workshop","first-page":"474","article-title":"Streaming small-footprint keyword spotting using sequence-to-sequence models","author":"He","year":"2017"},{"key":"10.1016\/j.csl.2026.101986_b12","series-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6858","article-title":"Query-by-example keyword spotting system using multi-head attention and soft-triple loss","author":"Huang","year":"2021"},{"key":"10.1016\/j.csl.2026.101986_b13","series-title":"International Conference on Machine Learning","first-page":"4651","article-title":"Perceiver: General perception with iterative attention","author":"Jaegle","year":"2021"},{"issue":"4","key":"10.1016\/j.csl.2026.101986_b14","doi-asserted-by":"crossref","first-page":"317","DOI":"10.1016\/j.specom.2008.10.002","article-title":"Discriminative keyword spotting","volume":"51","author":"Keshet","year":"2009","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101986_b15","unstructured":"Kingma, D.P., Ba, J., 2015. Adam: A Method for Stochastic Optimization. In: 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings."},{"key":"10.1016\/j.csl.2026.101986_b16","doi-asserted-by":"crossref","unstructured":"Klein, B., Wolf, L., Afek, Y., 2015. A dynamic convolutional layer for short range weather prediction. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 4840\u20134848.","DOI":"10.1109\/CVPR.2015.7299117"},{"key":"10.1016\/j.csl.2026.101986_b17","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"10816","article-title":"Hot-fixing wake word recognition for end-to-end ASR via neural model reprogramming","author":"Ku","year":"2024"},{"key":"10.1016\/j.csl.2026.101986_b18","series-title":"Proc. INTERSPEECH 2023","first-page":"3964","article-title":"PhonMatchNet: Phoneme-Guided Zero-Shot Keyword Spotting for User-Defined Keywords","author":"Lee","year":"2023"},{"key":"10.1016\/j.csl.2026.101986_b19","doi-asserted-by":"crossref","unstructured":"Li, X., Wei, X., Qin, X., 2020. Small-Footprint Keyword Spotting with Multi-Scale Temporal Convolution. In: 21th Annual Conference of the International-Speech-Communication-Association (INTERSPEECH 2020) Conference Location Shanghai, China. pp. 1987\u2014-1991.","DOI":"10.21437\/Interspeech.2020-3177"},{"key":"10.1016\/j.csl.2026.101986_b20","doi-asserted-by":"crossref","first-page":"4169","DOI":"10.1109\/ACCESS.2021.3139508","article-title":"Deep spoken keyword spotting: An overview","volume":"10","author":"L\u00f3pez-Espejo","year":"2021","journal-title":"IEEE Access"},{"key":"10.1016\/j.csl.2026.101986_b21","article-title":"DONUT: CTC-based query-by-example keyword spotting","author":"Lugosch","year":"2018","journal-title":"NeurIPS Work. Interpret. Robust. Audio Speech Lang."},{"key":"10.1016\/j.csl.2026.101986_b22","series-title":"2017 IEEE Automatic Speech Recognition and Understanding Workshop","first-page":"272","article-title":"Keyword spotting for google assistant using contextual speech recognition","author":"Michaely","year":"2017"},{"key":"10.1016\/j.csl.2026.101986_b23","series-title":"ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7454","article-title":"Small-footprint keyword spotting on raw audio data with sinc-convolutions","author":"Mittermaier","year":"2020"},{"key":"10.1016\/j.csl.2026.101986_b24","series-title":"Learning the pareto front with hypernetworks","author":"Navon","year":"2020"},{"key":"10.1016\/j.csl.2026.101986_b25","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"11656","article-title":"Open-vocabulary keyword-spotting with adaptive instance normalization","author":"Navon","year":"2024"},{"key":"10.1016\/j.csl.2026.101986_b26","series-title":"Flexible keyword spotting based on homogeneous audio-text embedding","author":"Nishu","year":"2023"},{"key":"10.1016\/j.csl.2026.101986_b27","series-title":"Matching latent encoding for audio-text based keyword spotting","author":"Nishu","year":"2023"},{"key":"10.1016\/j.csl.2026.101986_b28","series-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5206","article-title":"Librispeech: an asr corpus based on public domain audio books","author":"Panayotov","year":"2015"},{"key":"10.1016\/j.csl.2026.101986_b29","doi-asserted-by":"crossref","DOI":"10.1109\/ACCESS.2024.3421605","article-title":"Improved small-footprint ASR-based solution for open vocabulary keyword spotting","author":"Pudo","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.csl.2026.101986_b30","series-title":"Proc. Interspeech 2022","first-page":"126","article-title":"Generalized Keyword Spotting using ASR embeddings","author":"R","year":"2022"},{"key":"10.1016\/j.csl.2026.101986_b31","series-title":"International Conference on Machine Learning","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2023"},{"key":"10.1016\/j.csl.2026.101986_b32","series-title":"Proc. Interspeech 2021","first-page":"4204","article-title":"Personalized Keyphrase Detection Using Speaker and Environment Information","author":"Rikhye","year":"2021"},{"key":"10.1016\/j.csl.2026.101986_b33","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Wei, W., Hou, T., Pritch, Y., Wadhwa, N., Rubinstein, M., Aberman, K., 2024. Hyperdreambooth: Hypernetworks for fast personalization of text-to-image models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 6527\u20136536.","DOI":"10.1109\/CVPR52733.2024.00624"},{"key":"10.1016\/j.csl.2026.101986_b34","doi-asserted-by":"crossref","unstructured":"Sacchi, N., Nanchen, A., Jaggi, M., Cernak, M., 2019. Open-vocabulary keyword spotting with audio and text embeddings. In: INTERSPEECH 2019-IEEE International Conference on Acoustics, Speech, and Signal Processing.","DOI":"10.21437\/Interspeech.2019-1846"},{"key":"10.1016\/j.csl.2026.101986_b35","series-title":"Interspeech","first-page":"1478","article-title":"Convolutional neural networks for small-footprint keyword spotting.","author":"Sainath","year":"2015"},{"key":"10.1016\/j.csl.2026.101986_b36","doi-asserted-by":"crossref","unstructured":"Sendera, M., Przewie\u017alikowski, M., Karanowski, K., Zieba, M., Tabor, J., Spurek, P., 2023. Hypershot: Few-shot learning by kernel hypernetworks. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 2469\u20132478.","DOI":"10.1109\/WACV56688.2023.00250"},{"key":"10.1016\/j.csl.2026.101986_b37","series-title":"Attention-based end-to-end models for small-footprint keyword spotting","author":"Shan","year":"2018"},{"key":"10.1016\/j.csl.2026.101986_b38","series-title":"Proc. Interspeech 2022","first-page":"1871","article-title":"Learning Audio-Text Agreement for Open-vocabulary Keyword Spotting","author":"Shin","year":"2022"},{"key":"10.1016\/j.csl.2026.101986_b39","series-title":"Musan: A music, speech, and noise corpus","author":"Snyder","year":"2015"},{"key":"10.1016\/j.csl.2026.101986_b40","article-title":"Language modeling with recurrent highway hypernetworks","volume":"30","author":"Suarez","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.101986_b41","series-title":"Compressed time delay neural network for small-footprint keyword spotting","author":"Sun","year":"2017"},{"key":"10.1016\/j.csl.2026.101986_b42","series-title":"2018 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5484","article-title":"Deep residual learning for small-footprint keyword spotting","author":"Tang","year":"2018"},{"key":"10.1016\/j.csl.2026.101986_b43","series-title":"Discrete Random Signals and Statistical Signal Processing","author":"Therrien","year":"1992"},{"key":"10.1016\/j.csl.2026.101986_b44","series-title":"Interspeech","first-page":"3597","article-title":"The kaldi openkws system: Improving low resource keyword search.","author":"Trmal","year":"2017"},{"issue":"4","key":"10.1016\/j.csl.2026.101986_b45","doi-asserted-by":"crossref","first-page":"510","DOI":"10.1177\/0023830910372495","article-title":"The wildcat corpus of native-and foreign-accented english: Communicative efficiency across conversational dyads with varying language alignment profiles","volume":"53","author":"Van Engen","year":"2010","journal-title":"Lang. Speech"},{"key":"10.1016\/j.csl.2026.101986_b46","article-title":"Attention is all you need","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.101986_b47","series-title":"Open vocabulary keyword spotting through transfer learning from speech synthesis","author":"Vuppala","year":"2024"},{"key":"10.1016\/j.csl.2026.101986_b48","series-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)","first-page":"993","article-title":"VoxPopuli: A large-scale multilingual speech corpus for representation learning, semi-supervised learning and interpretation","author":"Wang","year":"2021"},{"key":"10.1016\/j.csl.2026.101986_b49","series-title":"Speech commands: A dataset for limited-vocabulary speech recognition","author":"Warden","year":"2018"},{"key":"10.1016\/j.csl.2026.101986_b50","series-title":"Interspeech","first-page":"361","article-title":"End-to-end transformer-based open-vocabulary keyword spotting with location-guided local attention.","author":"Wei","year":"2021"},{"key":"10.1016\/j.csl.2026.101986_b51","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhang, X.-L., 2020. Depthwise Separable Convolutional ResNet with Squeeze-and-Excitation Blocks for Small-footprint Keyword Spotting. In: 21th Annual Conference of the International-Speech-Communication-Association (INTERSPEECH 2020) Conference Location Shanghai, China. pp. 3372\u2014-3376.","DOI":"10.21437\/Interspeech.2020-1045"},{"key":"10.1016\/j.csl.2026.101986_b52","series-title":"2019 IEEE 2nd International Conference on Information Communication and Signal Processing","first-page":"400","article-title":"Robust small-footprint keyword spotting using sequence-to-sequence model with connectionist temporal classifier","author":"Xuan","year":"2019"},{"key":"10.1016\/j.csl.2026.101986_b53","series-title":"4th Workshop on Meta-Learning At NeurIPS 2020 (MetaLearn 2020)","article-title":"Meta-learning via hypernetworks","author":"Zhao","year":"2020"},{"key":"10.1016\/j.csl.2026.101986_b54","series-title":"Interspeech","first-page":"938","article-title":"Unrestricted vocabulary keyword spotting using LSTM-ctc.","author":"Zhuang","year":"2016"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000495?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000495?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:11:07Z","timestamp":1779225067000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000495"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":54,"alternative-id":["S0885230826000495"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101986","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Open-vocabulary keyword spotting with hyper-matched filters for small footprint devices","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101986","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"101986"}}