{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T21:22:36Z","timestamp":1781385756507,"version":"3.54.1"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030594299","type":"print"},{"value":"9783030594305","type":"electronic"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-59430-5_11","type":"book-chapter","created":{"date-parts":[[2020,9,25]],"date-time":"2020-09-25T08:04:59Z","timestamp":1601021099000},"page":"137-148","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["A Comparison of Metric Learning Loss Functions for End-To-End Speaker Verification"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5035-147X","authenticated-orcid":false,"given":"Juan M.","family":"Coria","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3739-925X","authenticated-orcid":false,"given":"Herv\u00e9","family":"Bredin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7531-2522","authenticated-orcid":false,"given":"Sahar","family":"Ghannay","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6865-4989","authenticated-orcid":false,"given":"Sophie","family":"Rosset","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2020,9,26]]},"reference":[{"key":"11_CR1","doi-asserted-by":"crossref","unstructured":"Bredin, H.: TristouNet: triplet loss for speaker turn embedding. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5430\u20135434. IEEE (2017)","DOI":"10.1109\/ICASSP.2017.7953194"},{"key":"11_CR2","doi-asserted-by":"crossref","unstructured":"Bredin, H., et al.: pyannote.audio: neural building blocks for speaker diarization. In: 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7124\u20137128. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9052974"},{"key":"11_CR3","doi-asserted-by":"crossref","unstructured":"Chung, J.S., Nagrani, A., Zisserman, A.: VoxCeleb2: deep speaker recognition. In: Interspeech, pp. 1086\u20131090 (2018)","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"11_CR4","doi-asserted-by":"crossref","unstructured":"Deng, J., Guo, J., Zafeiriou, S.: ArcFace: additive angular margin loss for deep face recognition. In: 2019 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4685\u20134694 (2019)","DOI":"10.1109\/CVPR.2019.00482"},{"key":"11_CR5","unstructured":"Garcia-Romero, D., McCree, A., Snyder, D., Sell, G.: JHU-HLTCOE System for VoxSRC 2019 (2019). http:\/\/www.robots.ox.ac.uk\/~vgg\/data\/voxceleb\/data_workshop\/JHU-HLTCOE_VoxSRC.pdf"},{"key":"11_CR6","doi-asserted-by":"crossref","unstructured":"Gelly, G., Gauvain, J.: Spoken language identification using LSTM-based angular proximity. In: Interspeech, pp. 2566\u20132570 (2017)","DOI":"10.21437\/Interspeech.2017-1334"},{"key":"11_CR7","doi-asserted-by":"publisher","unstructured":"Haasnoot, E., Khodabakhsh, A., Zeinstra, C., Spreeuwers, L., Veldhuis, R.: FEERCI: a package for fast non-parametric confidence intervals for equal error rates in amortized O(m log n). In: Bromme, A., Uhl, A., Busch, C., Rathgeb, C., Dantcheva, A. (eds.) 2018 International Conference of the Biometrics Special Interest Group, BIOSIG 2018. IEEE (2018). https:\/\/doi.org\/10.23919\/BIOSIG.2018.8553607","DOI":"10.23919\/BIOSIG.2018.8553607"},{"key":"11_CR8","doi-asserted-by":"crossref","unstructured":"Hadsell, R., Chopra, S., LeCun, Y.: Dimensionality reduction by learning an invariant mapping. In: 2006 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2006), vol. 2, pp. 1735\u20131742. IEEE (2006)","DOI":"10.1109\/CVPR.2006.100"},{"key":"11_CR9","unstructured":"Hermans, A., Beyer, L., Leibe, B.. In defense of the triplet loss for person re-identification. arXiv preprint arXiv:1703.07737 (2017)"},{"key":"11_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1007\/11744085_41","volume-title":"Computer Vision \u2013 ECCV 2006","author":"S Ioffe","year":"2006","unstructured":"Ioffe, S.: Probabilistic linear discriminant analysis. In: Leonardis, A., Bischof, H., Pinz, A. (eds.) ECCV 2006. LNCS, vol. 3954, pp. 531\u2013542. Springer, Heidelberg (2006). https:\/\/doi.org\/10.1007\/11744085_41"},{"key":"11_CR11","doi-asserted-by":"crossref","unstructured":"Li, Y., Gao, F., Ou, Z., Sun, J.: Angular softmax loss for end-to-end speaker verification. In: 2018 11th International Symposium on Chinese Spoken Language Processing (ISCSLP), pp. 190\u2013194. IEEE (2018)","DOI":"10.1109\/ISCSLP.2018.8706570"},{"key":"11_CR12","unstructured":"Liu, Y., Li, H., Wang, X.: Rethinking feature discrimination and polymerization for large-scale recognition. arXiv preprint arXiv:1710.00870 (2017)"},{"key":"11_CR13","doi-asserted-by":"crossref","unstructured":"Matejka, P., et al.: Analysis of Score Normalization in Multilingual Speaker Recognition. In: Interspeech, pp. 1567\u20131571 (2017)","DOI":"10.21437\/Interspeech.2017-803"},{"key":"11_CR14","doi-asserted-by":"crossref","unstructured":"Mclaren, M., Cast\u00e1n, D., Nandwana, M.K., Ferrer, L., Yilmaz, E.: How to train your speaker embeddings extractor. In: Proceedings of the Odyssey 2018 the Speaker and Language Recognition Workshop, pp. 327\u2013334 (2018)","DOI":"10.21437\/Odyssey.2018-46"},{"key":"11_CR15","doi-asserted-by":"publisher","first-page":"101078","DOI":"10.1016\/j.csl.2020.101078","volume":"63","author":"V Mingote","year":"2020","unstructured":"Mingote, V., Miguel, A., Ortega, A., Lleida, E.: Optimization of the area under the ROC curve using neural network supervectors for text-dependent speaker verification. Comput. Speech Lang. 63, 101078 (2020)","journal-title":"Comput. Speech Lang."},{"key":"11_CR16","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: VoxCeleb: a large-scale speaker identification dataset. In: Interspeech, pp. 2616\u20132620 (2017)","DOI":"10.21437\/Interspeech.2017-950"},{"key":"11_CR17","unstructured":"Nunes, J.A.C., Mac\u00eado, D., Zanchettin, C.: Additive margin SincNet for speaker recognition. In: 2019 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20135. IEEE (2019)"},{"key":"11_CR18","doi-asserted-by":"crossref","unstructured":"Pascual, S., Ravanelli, M., Serr\u00e0, J., Bonafonte, A., Bengio, Y.: Learning problem-agnostic speech representations from multiple self-supervised tasks. In: Interspeech, pp. 161\u2013165 (2019)","DOI":"10.21437\/Interspeech.2019-2605"},{"key":"11_CR19","unstructured":"Povey, D., et al.: The kaldi speech recognition toolkit. In: IEEE 2011 Workshop on Automatic Speech Recognition and Understanding. IEEE Signal Processing Society, December 2011, IEEE Catalog No.: CFP11SRW-USB"},{"key":"11_CR20","doi-asserted-by":"crossref","unstructured":"Ravanelli, M., Bengio, Y.: Speaker recognition from raw waveform with SincNet. In: 2018 IEEE Spoken Language Technology Workshop (SLT), pp. 1021\u20131028 (2018)","DOI":"10.1109\/SLT.2018.8639585"},{"key":"11_CR21","doi-asserted-by":"crossref","unstructured":"Ravanelli, M., et al.: Multi-task self-supervised learning for robust speech recognition. In: 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6989\u20136993. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053569"},{"key":"11_CR22","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Sentence-BERT: sentence embeddings using siamese BERT-networks. In: EMNLP\/IJCNLP (2019)","DOI":"10.18653\/v1\/D19-1410"},{"key":"11_CR23","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., Philbin, J.: FaceNet: a unified embedding for face recognition and clustering. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 815\u2013823 (2015)","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"11_CR24","unstructured":"Snyder, D., Chen, G., Povey, D.: MUSAN: a music, speech, and noise corpus. arXiv preprint arXiv:1510.08484 (2015)"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Snyder, D., Garcia-Romero, D., Sell, G., Povey, D., Khudanpur, S.: X-Vectors: robust DNN embeddings for speaker recognition. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5329\u20135333 (2018)","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"11_CR26","doi-asserted-by":"crossref","unstructured":"Srivastava, Y., Murali, V., Dubey, S.R.: A performance comparison of loss functions for deep face recognition. arXiv preprint arXiv:1901.05903 (2019)","DOI":"10.1007\/978-981-15-8697-2_30"},{"key":"11_CR27","doi-asserted-by":"crossref","unstructured":"Wan, L., Wang, Q., Papir, A., Moreno, I.L.: Generalized end-to-end loss for speaker verification. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4879\u20134883. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"11_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"499","DOI":"10.1007\/978-3-319-46478-7_31","volume-title":"Computer Vision \u2013 ECCV 2016","author":"Y Wen","year":"2016","unstructured":"Wen, Y., Zhang, K., Li, Z., Qiao, Y.: A discriminative feature learning approach for deep face recognition. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9911, pp. 499\u2013515. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46478-7_31"},{"key":"11_CR29","doi-asserted-by":"publisher","unstructured":"Zhang, C., Koishida, K.: End-to-end text-independent speaker verification with triplet loss on short utterances. In: Interspeech, pp. 1487\u20131491 (2017). https:\/\/doi.org\/10.21437\/Interspeech.2017-1608","DOI":"10.21437\/Interspeech.2017-1608"}],"container-title":["Lecture Notes in Computer Science","Statistical Language and Speech Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-59430-5_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,14]],"date-time":"2024-08-14T20:53:13Z","timestamp":1723668793000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-59430-5_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030594299","9783030594305"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-59430-5_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"26 September 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SLSP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Statistical Language and Speech Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Cardiff","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 October 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 October 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"slsp2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/irdta.eu\/slsp2020\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"25","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"13","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"52% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Due to COVID-19 pandemic the conference was held virtually.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}