{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T18:45:52Z","timestamp":1755801952887,"version":"3.44.0"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783032025470"},{"type":"electronic","value":"9783032025487"}],"license":[{"start":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:00:00Z","timestamp":1755820800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:00:00Z","timestamp":1755820800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-02548-7_11","type":"book-chapter","created":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T05:39:24Z","timestamp":1755754764000},"page":"121-132","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Speaker Group Encoding in\u00a0Self-supervised Speech Recognition Models"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9771-4994","authenticated-orcid":false,"given":"Felix","family":"Herron","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Solange","family":"Rossato","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexandre","family":"Allauzen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Benoit","family":"Favre","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fran\u00e7ois","family":"Portet","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,22]]},"reference":[{"issue":"1","key":"11_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2021.102770","volume":"59","author":"A Alsudais","year":"2022","unstructured":"Alsudais, A., Alotaibi, W., Alomary, F.: Similarities between Arabic dialects: investigating geographical proximity. Inf. Process. Manag. 59(1), 102770 (2022). https:\/\/doi.org\/10.1016\/j.ipm.2021.102770","journal-title":"Inf. Process. Manag."},{"key":"11_CR2","unstructured":"Baevski, A., Zhou, H., Mohamed, A., Auli, M.: Wav2vec 2.0: a framework for self-supervised learning of speech representations (2020)"},{"issue":"6","key":"11_CR3","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"S Chen","year":"2022","unstructured":"Chen, S., et al.: WavLM: large-scale self-supervised pre-training for full stack speech processing. IEEE J. Sel. Top. Signal Process. 16(6), 1505\u20131518 (2022). https:\/\/doi.org\/10.1109\/JSTSP.2022.3188113","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"11_CR4","doi-asserted-by":"crossref","unstructured":"Choi, K., Pasad, A., Nakamura, T., Fukayama, S., Livescu, K., Watanabe, S.: Self-supervised speech representations are more phonetic than semantic (2024)","DOI":"10.21437\/Interspeech.2024-1157"},{"key":"11_CR5","doi-asserted-by":"publisher","unstructured":"Chung, Y.A., Belinkov, Y., Glass, J.: Similarity analysis of self-supervised speech representations (2021). https:\/\/doi.org\/10.48550\/arXiv.2010.11481","DOI":"10.48550\/arXiv.2010.11481"},{"key":"11_CR6","unstructured":"Coulange, S., Rossato, S.: Proximit\u00e9 rythmique entre apprenants et natifs du fran\u00e7ais \u00c9valuation d\u2019une m\u00e9trique bas\u00e9e sur le CEFC. In: Benzitoun, C., Braud, C., Huber, L., Langlois, D., Ouni, S., Pogodalla, S., Schneider, S. (eds.) Actes de La 6e Conf\u00e9rence Conjointe Journ\u00e9es d\u2019\u00c9tudes Sur La Parole (JEP, 33e \u00c9dition), Traitement Automatique Des Langues Naturelles (TALN, 27e \u00c9dition), Rencontre Des \u00c9tudiants Chercheurs En Informatique Pour Le Traitement Automatique Des Langues (R\u00c9CITAL, 22e \u00c9dition). Volume 1 : Journ\u00e9es d\u2019\u00c9tudes Sur La Parole, pp. 118\u2013126. ATALA, Nancy, France (2020)"},{"key":"11_CR7","doi-asserted-by":"publisher","unstructured":"Dorn, R.: Dialect-specific models for automatic speech recognition of African American vernacular English. In: Kovatchev, V., Temnikova, I., \u0160andrih, B., Nikolova, I. (eds.) Proceedings of the Student Research Workshop Associated with RANLP 2019, pp. 16\u201320. INCOMA Ltd., Varna, Bulgaria (2019). https:\/\/doi.org\/10.26615\/issn.2603-2821.2019_003","DOI":"10.26615\/issn.2603-2821.2019_003"},{"key":"11_CR8","doi-asserted-by":"publisher","unstructured":"Evain, S., et al.: LeBenchmark: a reproducible framework for assessing self-supervised representation learning from speech. In: Interspeech 2021, pp. 1439\u20131443 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-556","DOI":"10.21437\/Interspeech.2021-556"},{"key":"11_CR9","doi-asserted-by":"publisher","unstructured":"Feng, C.L., Hsu, P.c., Lee, H.Y.: Silence is sweeter than speech: self-supervised model using silence to store speaker information (2022). https:\/\/doi.org\/10.48550\/arXiv.2205.03759","DOI":"10.48550\/arXiv.2205.03759"},{"key":"11_CR10","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2023.101567","volume":"84","author":"S Feng","year":"2024","unstructured":"Feng, S., Halpern, B.M., Kudina, O., Scharenborg, O.: Towards inclusive automatic speech recognition. Comput. Speech Lang. 84, 101567 (2024). https:\/\/doi.org\/10.1016\/j.csl.2023.101567","journal-title":"Comput. Speech Lang."},{"key":"11_CR11","doi-asserted-by":"crossref","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: HuBERT: self-supervised speech representation learning by masked prediction of hidden units (2021)","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"11_CR12","doi-asserted-by":"publisher","unstructured":"Li, J., et al.: Accent-robust automatic speech recognition using supervised and unsupervised Wav2vec embeddings (2021). https:\/\/doi.org\/10.48550\/arXiv.2110.03520","DOI":"10.48550\/arXiv.2110.03520"},{"key":"11_CR13","doi-asserted-by":"publisher","unstructured":"Lima, L., Furtado, V., Furtado, E., Almeida, V.: Empirical analysis of bias in voice-based personal assistants. In: Companion Proceedings of The 2019 World Wide Web Conference, pp. 533\u2013538. ACM, San Francisco (2019). https:\/\/doi.org\/10.1145\/3308560.3317597","DOI":"10.1145\/3308560.3317597"},{"key":"11_CR14","doi-asserted-by":"publisher","unstructured":"Liu, O., Tang, H., Goldwater, S.: Self-supervised predictive coding models encode speaker and phonetic information in orthogonal subspaces (2023). https:\/\/doi.org\/10.48550\/arXiv.2305.12464","DOI":"10.48550\/arXiv.2305.12464"},{"key":"11_CR15","doi-asserted-by":"publisher","unstructured":"Mohamed, M., Liu, O.D., Tang, H., Goldwater, S.: Orthogonality and isotropy of speaker and phonetic information in self-supervised speech representations (2024). https:\/\/doi.org\/10.48550\/arXiv.2406.09200","DOI":"10.48550\/arXiv.2406.09200"},{"key":"11_CR16","doi-asserted-by":"publisher","unstructured":"van Niekerk, B., Nortje, L., Baas, M., Kamper, H.: Analyzing speaker information in self-supervised models to improve zero-resource speech processing (2021). https:\/\/doi.org\/10.48550\/arXiv.2108.00917","DOI":"10.48550\/arXiv.2108.00917"},{"key":"11_CR17","doi-asserted-by":"publisher","unstructured":"Pasad, A., Chien, C.M., Settle, S., Livescu, K.: What do self-supervised speech models know about words? (2024). https:\/\/doi.org\/10.48550\/arXiv.2307.00162","DOI":"10.48550\/arXiv.2307.00162"},{"key":"11_CR18","doi-asserted-by":"crossref","unstructured":"Pasad, A., Chou, J.C., Livescu, K.: Layer-wise analysis of a self-supervised speech representation model (2022)","DOI":"10.1109\/ICASSP49357.2023.10096149"},{"key":"11_CR19","doi-asserted-by":"crossref","unstructured":"Pasad, A., Shi, B., Livescu, K.: Comparative layer-wise analysis of self-supervised speech models (2023)","DOI":"10.1109\/ICASSP49357.2023.10096149"},{"key":"11_CR20","doi-asserted-by":"publisher","unstructured":"Ravanelli, M., et al.: SpeechBrain: a general-purpose speech toolkit (2021). https:\/\/doi.org\/10.48550\/arXiv.2106.04624","DOI":"10.48550\/arXiv.2106.04624"},{"key":"11_CR21","doi-asserted-by":"publisher","unstructured":"Salehghaffari, H.: Speaker verification using convolutional neural networks (2018). https:\/\/doi.org\/10.48550\/arXiv.1803.05427","DOI":"10.48550\/arXiv.1803.05427"},{"key":"11_CR22","doi-asserted-by":"publisher","unstructured":"Sanabria, R., Tang, H., Goldwater, S.: Analyzing acoustic word embeddings from pre-trained self-supervised speech models (2023). https:\/\/doi.org\/10.48550\/arXiv.2210.16043","DOI":"10.48550\/arXiv.2210.16043"},{"key":"11_CR23","unstructured":"Sekkat, C., Leroy, F., Mdhaffar, S., Smith, B.P., Est\u00e8ve, Y., Dureau, J., Coucke, A.: Sonos voice control bias assessment dataset: a methodology for demographic bias assessment in voice assistants. In: Calzolari, N., Kan, M.Y., Hoste, V., Lenci, A., Sakti, S., Xue, N. (eds.) Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024), pp. 15056\u201315075. ELRA and ICCL, Torino (2024)"},{"key":"11_CR24","doi-asserted-by":"publisher","unstructured":"Shon, S., Ali, A., Samih, Y., Mubarak, H., Glass, J.: ADI17: a fine-grained arabic dialect identification dataset. In: ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8244\u20138248. IEEE, Barcelona (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9052982","DOI":"10.1109\/ICASSP40776.2020.9052982"},{"key":"11_CR25","doi-asserted-by":"publisher","unstructured":"Veliche, I.E., Huang, Z., Kochaniyan, V.A., Peng, F., Kalinli, O., Seltzer, M.L.: Towards measuring fairness in speech recognition: fair-speech dataset (2024). https:\/\/doi.org\/10.48550\/arXiv.2408.12734","DOI":"10.48550\/arXiv.2408.12734"},{"key":"11_CR26","doi-asserted-by":"publisher","unstructured":"Whetten, R., Parcollet, T., Dinarelli, M., Est\u00e8ve, Y.: Open implementation and study of BEST-RQ for speech processing (2024). https:\/\/doi.org\/10.48550\/arXiv.2405.04296","DOI":"10.48550\/arXiv.2405.04296"},{"key":"11_CR27","doi-asserted-by":"publisher","unstructured":"Wu, Y., et al.: See what i\u2019m saying? Comparing intelligent personal assistant use for native and non-native language speakers. In: 22nd International Conference on Human-Computer Interaction with Mobile Devices and Services, pp.\u00a01\u20139. MobileHCI \u201920, Association for Computing Machinery, New York, NY, USA (2020). https:\/\/doi.org\/10.1145\/3379503.3403563","DOI":"10.1145\/3379503.3403563"},{"key":"11_CR28","doi-asserted-by":"publisher","unstructured":"Yang, S.W., et al.: SUPERB: speech processing universal PERformance benchmark (2021). https:\/\/doi.org\/10.48550\/arXiv.2105.01051","DOI":"10.48550\/arXiv.2105.01051"},{"key":"11_CR29","doi-asserted-by":"publisher","unstructured":"Zaiem, S., Parcollet, T., Essid, S.: Less forgetting for better generalization: exploring continual-learning fine-tuning methods for speech self-supervised representations (2024). https:\/\/doi.org\/10.48550\/arXiv.2407.00756","DOI":"10.48550\/arXiv.2407.00756"},{"key":"11_CR30","doi-asserted-by":"publisher","unstructured":"Zhou, W., et al.: Enhancing and adversarial: improve ASR with speaker labels. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10096722","DOI":"10.1109\/ICASSP49357.2023.10096722"}],"container-title":["Lecture Notes in Computer Science","Text, Speech, and Dialogue"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-02548-7_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T05:39:33Z","timestamp":1755754773000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-02548-7_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,22]]},"ISBN":["9783032025470","9783032025487"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-02548-7_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025,8,22]]},"assertion":[{"value":"22 August 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors\u00a0have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"TSD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Text, Speech, and Dialogue","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Erlangen","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"tsd2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.kiv.zcu.cz\/tsd2025\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}