{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T19:40:27Z","timestamp":1774467627578,"version":"3.50.1"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030878016","type":"print"},{"value":"9783030878023","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-87802-3_69","type":"book-chapter","created":{"date-parts":[[2021,9,21]],"date-time":"2021-09-21T23:36:52Z","timestamp":1632267412000},"page":"773-785","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":36,"title":["Learning Efficient Representations for Keyword Spotting with Triplet Loss"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3684-7356","authenticated-orcid":false,"given":"Roman","family":"Vygon","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5660-0601","authenticated-orcid":false,"given":"Nikolay","family":"Mikhaylovskiy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,9,22]]},"reference":[{"key":"69_CR1","doi-asserted-by":"crossref","unstructured":"Tang, R., Lin, J.: Deep residual learning for small-footprint keyword spotting. In: International Conference on Acoustics, Speech and Signal Processing, pp.\u00a05484\u20135488 (2018)","DOI":"10.1109\/ICASSP.2018.8462688"},{"key":"69_CR2","unstructured":"Zhang, Y., Suda, N., Lai, L., Chandra, V.: Hello Edge: Keyword Spotting on Microcontrollers"},{"key":"69_CR3","unstructured":"de Andrade, D., Sabato, L., Viana, M., Bernkopf, C.: A neural attention model for speech command recognition"},{"issue":"3","key":"69_CR4","doi-asserted-by":"publisher","first-page":"127","DOI":"10.1109\/TAU.1967.1161911","volume":"15","author":"C Teacher","year":"1967","unstructured":"Teacher, C., Kellett, Y., Focht, L.: Experimental, limited vocabulary, speech recognizer. IEEE Trans. Audio Electroacoust. 15(3), 127\u2013130 (1967)","journal-title":"IEEE Trans. Audio Electroacoust."},{"key":"69_CR5","unstructured":"Rohlicek, J.R., Russell, W., Roukos, S., Gish, H.: Continuous hidden Markov modeling for speaker-independent word spotting. In: Acoustics, Speech, and Signal Processing, pp.\u00a0627\u2013630 (1989)"},{"key":"69_CR6","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1007\/11551874_39","volume-title":"Text, Speech and Dialogue","author":"I Sz\u00f6ke","year":"2005","unstructured":"Sz\u00f6ke, I., Schwarz, P., Mat\u011bjka, P., Burget, L., Karafi\u00e1t, M., \u010cernock\u00fd, J.: Phoneme based acoustics keyword spotting in informal continuous speech. In: Matou\u0161ek, V., Mautner, P., Pavelka, T. (eds.) TSD 2005. LNCS (LNAI), vol. 3658, pp. 302\u2013309. Springer, Heidelberg (2005). https:\/\/doi.org\/10.1007\/11551874_39"},{"key":"69_CR7","doi-asserted-by":"crossref","unstructured":"Zhang, S., Shuang, Z., Shi, Q., Qin, Y.: Improved mandarin keyword spotting using confusion garbage model. In: 2010 20th International Conference on Pattern Recognition (ICPR), pp.\u00a03700\u20133703","DOI":"10.1109\/ICPR.2010.901"},{"key":"69_CR8","series-title":"Communications in Computer and Information Science","doi-asserted-by":"publisher","first-page":"186","DOI":"10.1007\/978-3-642-41947-8_17","volume-title":"Information and Software Technologies","author":"M Greibus","year":"2013","unstructured":"Greibus, M., Telksnys, L.: Speech keyword spotting with rule based segmentation. In: Skersys, T., Butleris, R., Butkiene, R. (eds.) ICIST 2013. CCIS, vol. 403, pp. 186\u2013197. Springer, Heidelberg (2013). https:\/\/doi.org\/10.1007\/978-3-642-41947-8_17"},{"issue":"13","key":"69_CR9","doi-asserted-by":"publisher","first-page":"5668","DOI":"10.1016\/j.eswa.2015.02.036","volume":"42","author":"SS Principi","year":"2015","unstructured":"Principi, S.S., Bonfigli, R., Ferroni, G., Piazza, F.: An integrated system for voice command recognition and emergency detection based on audio signals. Expert Syst. Appl. 42(13), 5668\u20135683 (2015). https:\/\/doi.org\/10.1016\/j.eswa.2015.02.036","journal-title":"Expert Syst. Appl."},{"key":"69_CR10","doi-asserted-by":"crossref","unstructured":"Chen, G., Parada, C., Heigold, G.: Small-footprint keyword spotting using deep neural networks. In: Acoustics, Speech and Signal Processing, International Conference on, p.\u00a04087\u20134091 (2014)","DOI":"10.1109\/ICASSP.2014.6854370"},{"key":"69_CR11","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Parada C.: Convolutional neural networks for small-footprint keyword spotting. In: Sixteenth Annual Conference of the International Speech Communication Association (2015)","DOI":"10.21437\/Interspeech.2015-352"},{"key":"69_CR12","doi-asserted-by":"crossref","unstructured":"Arik, S.O., et al.: Convolutional recurrent neural networks for small-footprint keyword spotting (2017)","DOI":"10.21437\/Interspeech.2017-1737"},{"key":"69_CR13","doi-asserted-by":"crossref","unstructured":"Sun, M., et al.: Max-pooling loss training of long short-term memory networks for small-footprint keyword spotting. In: Spoken Language Technology Workshop, pp.\u00a0474\u2013480 (2016)","DOI":"10.1109\/SLT.2016.7846306"},{"key":"69_CR14","doi-asserted-by":"crossref","unstructured":"He, Y., Prabhavalkar, R., Rao, K., Li, W., Bakhtin, A., McGraw, I.: Streaming small-footprint keyword spotting using sequence-to-sequence models. In: Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a0474\u2013481 (2017)","DOI":"10.1109\/ASRU.2017.8268974"},{"key":"69_CR15","doi-asserted-by":"crossref","unstructured":"Lei, J., et al.: Low-power audio keyword spotting using Tsetlin machines. J. Low Power Electron. Appl. 11(2): 18","DOI":"10.3390\/jlpea11020018"},{"key":"69_CR16","unstructured":"Warden, P.: Speech commands: a public dataset for single-word speech recognition"},{"key":"69_CR17","unstructured":"Jansson, P.: Single-word speech recognition with convolutional neural networks on raw waveforms. Degree Thesis, Information technology, ARCADA University, Finland"},{"key":"69_CR18","doi-asserted-by":"publisher","unstructured":"Majumdar, S., Ginsburg, B.: MatchboxNet: 1D time-channel separable convolutional neural network architecture for speech commands recognition. In: Proceedings of Interspeech, pp.\u00a03356\u20133360. https:\/\/doi.org\/10.21437\/Interspeech.2020-1058 (2020)","DOI":"10.21437\/Interspeech.2020-1058"},{"key":"69_CR19","doi-asserted-by":"crossref","unstructured":"Mordido, G., Van Keirsbilck, M., Keller, A.: Compressing 1D time-channel separable convolutions using sparse random ternary matrices (2021)","DOI":"10.21437\/Interspeech.2021-141"},{"key":"69_CR20","doi-asserted-by":"crossref","unstructured":"Rybakov O., Kononenko N., Subrahmanya N., Visontai M., Laurenzo S.: Streaming keyword spotting on mobile devices. In: Proceedings of Interspeech, pp.\u00a02277\u20132281 (2020)","DOI":"10.21437\/Interspeech.2020-1003"},{"key":"69_CR21","doi-asserted-by":"publisher","unstructured":"Wei, Y., Gong, Z., Yang, S., Ye, K., Wen, Y.: EdgeCRNN: an edge-computing oriented model of acoustic feature enhancement for keyword spotting. J. Ambient. Intell. Humaniz. Comput. 1\u201311 (2021). https:\/\/doi.org\/10.1007\/s12652-021-03022-1","DOI":"10.1007\/s12652-021-03022-1"},{"key":"69_CR22","doi-asserted-by":"crossref","unstructured":"Tang, R., et al.: Howl: a deployed, open-source wake word detection system. In: Proceedings of Second Workshop for NLP Open-Source Software (NLP-OSS), pp.\u00a061\u201365 (2020)","DOI":"10.18653\/v1\/2020.nlposs-1.9"},{"key":"69_CR23","unstructured":"Hermans, A., Beyer, L., Leibe, B.: In defense of the triplet loss for person re-identification"},{"key":"69_CR24","doi-asserted-by":"crossref","unstructured":"Wang, J., et al.: Learning fine-grained image similarity with deep ranking. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition, pp.\u00a01386\u20131393 (2014)","DOI":"10.1109\/CVPR.2014.180"},{"key":"69_CR25","first-page":"1109","volume":"11","author":"G Chechik","year":"2010","unstructured":"Chechik, G., Sharma, V., Shalit, U., Bengio, S.: Large scale online learning of image similarity through ranking. J. Mach. Learn. Res. 11, 1109\u20131135 (2010)","journal-title":"J. Mach. Learn. Res."},{"key":"69_CR26","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J.: FaceNet: a unified embedding for face recognition and clustering. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR),\u00a0pp.\u00a0815\u2013823 (2015)","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"69_CR27","doi-asserted-by":"crossref","unstructured":"Huang, J., Li, Y., Tao, J., Lian, Z.: Speech emotion recognition from variable-length inputs with triplet loss function. In: Proceedings of INTERSPEECH, pp.\u00a03673\u20133677 (2018)","DOI":"10.21437\/Interspeech.2018-1432"},{"issue":"3","key":"69_CR28","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1016\/j.visinf.2019.10.003","volume":"3","author":"M Ren","year":"2019","unstructured":"Ren, M., Nie, W., Liu, A., Su, Y.: Multi-modal correlated network for emotion recognition in speech. Vis. Informat. 3(3), 150\u2013155 (2019)","journal-title":"Vis. Informat."},{"key":"69_CR29","doi-asserted-by":"crossref","unstructured":"Kumar, P., Jain, S., Raman, B, Roy, P.P., Iwamura, M.: End-to-end triplet loss based emotion embedding system for speech emotion recognition. In: 2020 25th International Conference on Pattern Recognition (ICPR), pp.\u00a08766\u20138773 (2021)","DOI":"10.1109\/ICPR48806.2021.9413144"},{"key":"69_CR30","doi-asserted-by":"crossref","unstructured":"Harvill, J., AbdelWahab, M., Lotfian, R., Busso, C.: Retrieving speech samples with similar emotional content using a triplet loss function. In: International Conference on Acoustics, Speech and Signal Processing, Brighton, United Kingdom, pp.\u00a07400\u20137404 (2019)","DOI":"10.1109\/ICASSP.2019.8683273"},{"key":"69_CR31","doi-asserted-by":"crossref","unstructured":"Bredin, H.: Tristounet: triplet loss for speaker turns embedding. In: 2017 IEEE International Conference on Acoustics, speech and Signal Processing (ICASSP), pp.\u00a05430\u20135434 (2017)","DOI":"10.1109\/ICASSP.2017.7953194"},{"key":"69_CR32","doi-asserted-by":"crossref","unstructured":"Song H., Willi, M., Thiagarajan, J.J., Berisha, V., Spanias, A.: Triplet network with attention for speaker diarization. In: Proceedings of Interspeech, pp.\u00a03608\u20133612 (2018)","DOI":"10.21437\/Interspeech.2018-2305"},{"key":"69_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, C., Koishida, K.: End-to-end text-independent speaker verification with triplet loss on short utterances. In: Proceedings of Interspeech, pp.\u00a01487\u20131491 (2017)","DOI":"10.21437\/Interspeech.2017-1608"},{"key":"69_CR34","unstructured":"Li, C., et al.: Deep speaker: an end-to-end neural speaker embedding system"},{"key":"69_CR35","doi-asserted-by":"crossref","unstructured":"Turpault, N., Serizel, R., Vincent, E.: Semi-supervised triplet loss-based learning of ambient audio embeddings. ICASSP 2019. Brighton, United Kingdom (2019)","DOI":"10.1109\/ICASSP.2019.8683774"},{"key":"69_CR36","doi-asserted-by":"crossref","unstructured":"Sacchi, N., Nanchen, A., Jaggi, M., Cer\u0148ak, M.: Open-vocabulary keyword spotting with audio and text embeddings, pp.\u00a03362\u20133366","DOI":"10.21437\/Interspeech.2019-1846"},{"key":"69_CR37","doi-asserted-by":"crossref","unstructured":"Shor, J., et al.: Towards learning a universal non-semantic representation of speech. In: Proceedings of Interspeech, pp.\u00a0140\u2013144 (2020)","DOI":"10.21437\/Interspeech.2020-1242"},{"key":"69_CR38","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Lv, Z., Huang, S., Xie, L.: Verifying deep keyword spotting detection with acoustic word embeddings. In: 2019 IEEE Automatic Speech Recognition and Understanding Workshop, ASRU 2019 Proceedings, no. 61571363, pp.\u00a0613\u2013620 (2019)","DOI":"10.1109\/ASRU46091.2019.9003781"},{"key":"69_CR39","doi-asserted-by":"crossref","unstructured":"Huh, J., Lee, M., Heo, H., Mun, S., Chung, J.S.: Metric learning for keyword spotting, 2021 IEEE Spoken Language Technology Workshop (SLT). In: IEEE, pp.\u00a0133\u2013140 (2021)","DOI":"10.1109\/SLT48900.2021.9383571"},{"key":"69_CR40","doi-asserted-by":"crossref","unstructured":"Huang, J., Gharbieh, W., Shim, H.S., Kim, E.: Query-by-example keyword spotting system using multi-head attention and softtriple loss. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a06858\u20136862 (2021)","DOI":"10.1109\/ICASSP39728.2021.9414156"},{"key":"69_CR41","unstructured":"Tang, R. and Lin, J.: Honk: A PyTorch reimplementation of convolutional neural networks for keyword spotting 2017. http:\/\/arxiv.org\/abs\/1710.06554 (2021)"},{"key":"69_CR42","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: Librispeech: An ASR corpus based on public domain audio books. In: ICASSP, IEEE International Conference on Acoustics, Speech and Signal Processing, pp.\u00a05206\u20135210 (2015)","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"69_CR43","doi-asserted-by":"crossref","unstructured":"Lugosch, L., Ravanelli, M., Ignoto, P., Tomar, V.S., Bengio, Y.: Speech model pre-training for end-to-end spoken language understanding. In: Proceedings of the Annual Conference of the International Speech Communication Association, INTERSPEECH, pp.\u00a0814\u2013818 (2019)","DOI":"10.21437\/Interspeech.2019-2396"},{"key":"69_CR44","doi-asserted-by":"crossref","unstructured":"McAuliffe, M., Socolof, M., Mihuc, S., Wagner, M., Sonderegger, M.: Montreal forced aligner: Trainable text-speech alignment using kaldi. In: Proceedings of the Annual Conference of the International Speech Communication Association INTERSPEECH, pp.\u00a0498\u2013502 (2017)","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"69_CR45","unstructured":"https:\/\/zenodo.org\/record\/2619474. Accessed 2 Jan 2021"},{"key":"69_CR46","doi-asserted-by":"crossref","unstructured":"Ahmed, A.F., Sherif, M.A., Ngomo, A.C.N.: Do your resources sound similar?: On the impact of using phonetic similarity in link discovery, in K-CAP 2019. In: 10th International Conference on Knowledge Capture 8(19), 53\u201360 (2019)","DOI":"10.1145\/3360901.3364426"},{"key":"69_CR47","unstructured":"Ginsburg, B., et al.: Stochastic gradient methods with layer-wise adaptive moments for training of deep networks"},{"issue":"3","key":"69_CR48","doi-asserted-by":"publisher","first-page":"535","DOI":"10.1109\/TBDATA.2019.2921572","volume":"7","author":"J Johnson","year":"2021","unstructured":"Johnson, J., Douze, M., J\u00e9gou, H.: Billion-scale similarity search with GPUs. IEEE Trans. Big Data 7(3), 535\u2013547 (2021)","journal-title":"IEEE Trans. Big Data"}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-87802-3_69","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,8]],"date-time":"2024-09-08T17:39:59Z","timestamp":1725817199000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-87802-3_69"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030878016","9783030878023"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-87802-3_69","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"22 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SPECOM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Speech and Computer","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"St Petersburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Russia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 September 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"specom2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/specom.nw.ru\/2021\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"163","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"74","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"45% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.5","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held online due to the COVID-19 pandemic.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}