{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:43:28Z","timestamp":1777657408993,"version":"3.51.4"},"publisher-location":"Cham","reference-count":34,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031404979","type":"print"},{"value":"9783031404986","type":"electronic"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-40498-6_24","type":"book-chapter","created":{"date-parts":[[2023,8,22]],"date-time":"2023-08-22T23:02:34Z","timestamp":1692745354000},"page":"270-282","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Evaluation of\u00a0Speech Representations for\u00a0MOS Prediction"],"prefix":"10.1007","author":[{"given":"Frederico","family":"S. Oliveira","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edresson","family":"Casanova","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arnaldo Candido","family":"Junior","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lucas","family":"R. S. Gris","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anderson","family":"S. Soares","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arlindo","family":"R. Galv\u00e3o Filho","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,23]]},"reference":[{"key":"24_CR1","unstructured":"Babu, A., et al.: XLS-R: self-supervised cross-lingual speech representation learning at scale. CoRR abs\/2111.09296 (2021). https:\/\/arxiv.org\/abs\/2111.09296"},{"key":"24_CR2","unstructured":"Baevski, A., Hsu, W.N., Xu, Q., Babu, A., Gu, J., Auli, M.: Data2vec: a general framework for self-supervised learning in speech, vision and language. In: International Conference on Machine Learning, pp. 1298\u20131312. PMLR (2022)"},{"key":"24_CR3","unstructured":"Baevski, A., Zhou, H., Mohamed, A., Auli, M.: Wav2vec 2.0: a framework for self-supervised learning of speech representations. In: Proceedings of the 34th International Conference on Neural Information Processing Systems, NIPS 2020, Curran Associates Inc., Red Hook, NY, USA (2020)"},{"key":"24_CR4","doi-asserted-by":"publisher","first-page":"1505","DOI":"10.1109\/JSTSP.2022.3188113","volume":"16","author":"S Chen","year":"2021","unstructured":"Chen, S., et al.: WavLM: large-scale self-supervised pre-training for full stack speech processing. IEEE J. Select. Top. Signal Process. 16, 1505\u20131518 (2021)","journal-title":"IEEE J. Select. Top. Signal Process."},{"key":"24_CR5","doi-asserted-by":"crossref","unstructured":"Chung, Y.A., Hsu, W.N., Tang, H., Glass, J.: An unsupervised autoregressive model for speech representation learning. In: Proceedings of the Interspeech 2019, pp. 146\u2013150 (2019). https:\/\/doi.org\/10.21437\/Interspeech.2019-1473","DOI":"10.21437\/Interspeech.2019-1473"},{"key":"24_CR6","doi-asserted-by":"crossref","unstructured":"Conneau, A., Baevski, A., Collobert, R., Mohamed, A., Auli, M.: Unsupervised cross-lingual representation learning for speech recognition, pp. 2426\u20132430 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-329","DOI":"10.21437\/Interspeech.2021-329"},{"key":"24_CR7","doi-asserted-by":"crossref","unstructured":"Cooper, E., Huang, W.C., Toda, T., Yamagishi, J.: Generalization ability of MOS prediction networks. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8442\u20138446. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746395"},{"key":"24_CR8","doi-asserted-by":"publisher","unstructured":"Das, R., et al.: Predictions of subjective ratings and spoofing assessments of voice conversion challenge 2020 submissions, pp. 99\u2013120 (2020). https:\/\/doi.org\/10.21437\/VCC_BC.2020-15","DOI":"10.21437\/VCC_BC.2020-15"},{"key":"24_CR9","doi-asserted-by":"crossref","unstructured":"Fu, S.W., Tsao, Y., Hwang, H.T., Wang, H.M.: Quality-net: an end-to-end non-intrusive speech quality assessment model based on BLSTM (2018)","DOI":"10.21437\/Interspeech.2018-1802"},{"key":"24_CR10","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"24_CR11","unstructured":"Heo, H.S., Lee, B.J., Huh, J., Chung, J.S.: Clova baseline system for the voxceleb speaker recognition challenge 2020. arXiv preprint arXiv:2009.14153 (2020)"},{"key":"24_CR12","doi-asserted-by":"publisher","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: HuBERT: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio, Speech Lang. Proc. 29, 3451\u20133460 (2021). https:\/\/doi.org\/10.1109\/TASLP.2021.3122291","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"24_CR13","doi-asserted-by":"crossref","unstructured":"King, S., Karaiskos, V.: The blizzard challenge 2016 (2016)","DOI":"10.21437\/Blizzard.2016-1"},{"key":"24_CR14","doi-asserted-by":"publisher","unstructured":"Koluguri, N.R., Li, J., Lavrukhin, V., Ginsburg, B.: Speakernet: 1d depth-wise separable convolutional network for text-independent speaker recognition and verification (2020). https:\/\/doi.org\/10.48550\/ARXIV.2010.12653,https:\/\/arxiv.org\/abs\/2010.12653","DOI":"10.48550\/ARXIV.2010.12653,"},{"key":"24_CR15","doi-asserted-by":"crossref","unstructured":"Koluguri, N.R., Park, T., Ginsburg, B.: Titanet: neural model for speaker representation with 1d depth-wise separable convolutions and global context. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8102\u20138106. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746806"},{"key":"24_CR16","doi-asserted-by":"publisher","unstructured":"Kriman, S., et al.: Quartznet: deep automatic speech recognition with 1d time-channel separable convolutions. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6124\u20136128 (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9053889","DOI":"10.1109\/ICASSP40776.2020.9053889"},{"key":"24_CR17","doi-asserted-by":"publisher","first-page":"2351","DOI":"10.1109\/TASLP.2021.3095662","volume":"29","author":"AT Liu","year":"2020","unstructured":"Liu, A.T., Li, S.W., Lee, H.Y.: Tera: self-supervised learning of transformer encoder representation for speech. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 2351\u20132366 (2020)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"24_CR18","doi-asserted-by":"publisher","first-page":"2351","DOI":"10.1109\/TASLP.2021.3095662","volume":"29","author":"AT Liu","year":"2021","unstructured":"Liu, A.T., Li, S.W., Lee, H.Y.: Tera: self-supervised learning of transformer encoder representation for speech. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 2351\u20132366 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"24_CR19","doi-asserted-by":"publisher","unstructured":"Lo, C.C., et al.: MOSNet: deep learning-based objective assessment for voice conversion. In: Interspeech 2019. ISCA (2019). https:\/\/doi.org\/10.21437\/interspeech.2019-2003, https:\/\/doi.org\/10.21437%2Finterspeech.2019-2003","DOI":"10.21437\/interspeech.2019-2003"},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Lorenzo-Trueba, J., et al.: The voice conversion challenge 2018: promoting development of parallel and nonparallel methods (2018)","DOI":"10.21437\/Odyssey.2018-28"},{"key":"24_CR21","unstructured":"Van der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(11) (2008)"},{"key":"24_CR22","doi-asserted-by":"publisher","unstructured":"Oord, A.v.d., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding (2018). https:\/\/doi.org\/10.48550\/ARXIV.1807.03748, https:\/\/arxiv.org\/abs\/1807.03748","DOI":"10.48550\/ARXIV.1807.03748"},{"key":"24_CR23","unstructured":"Patton, B., Agiomyrgiannakis, Y., Terry, M., Wilson, K.W., Saurous, R.A., Sculley, D.: AutoMOS: learning a non-intrusive assessor of naturalness-of-speech. CoRR abs\/1611.09207 (2016). https:\/\/arxiv.org\/abs\/1611.09207"},{"key":"24_CR24","doi-asserted-by":"publisher","unstructured":"Radford, A., Kim, J.W., Xu, T., Brockman, G., McLeavey, C., Sutskever, I.: Robust speech recognition via large-scale weak supervision (2022). https:\/\/doi.org\/10.48550\/ARXIV.2212.04356,https:\/\/arxiv.org\/abs\/2212.04356","DOI":"10.48550\/ARXIV.2212.04356,"},{"key":"24_CR25","doi-asserted-by":"crossref","unstructured":"Ragano, A., et al.: A comparison of deep learning MOS predictors for speech synthesis quality (2022)","DOI":"10.1109\/ISSC59246.2023.10162088"},{"key":"24_CR26","doi-asserted-by":"publisher","unstructured":"Rix, A., Beerends, J., Hollier, M., Hekstra, A.: Perceptual evaluation of speech quality (PESQ) - a new method for speech quality assessment of telephone networks and codecs. In: Proceedings 2001 IEEE International Conference on Acoustics, Speech, and Signal Processing, vol. 2, pp. 749\u2013752. (Cat. No.01CH37221) (2001). https:\/\/doi.org\/10.1109\/ICASSP.2001.941023","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"24_CR27","doi-asserted-by":"crossref","unstructured":"Todisco, M., et al.: ASVspoof 2019: Future horizons in spoofed and fake audio detection. arXiv preprint arXiv:1904.05441 (2019)","DOI":"10.21437\/Interspeech.2019-2249"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"Tseng, W.C., Huang, C.Y., Kao, W.T., Lin, Y.Y., Lee, H.Y.: Utilizing self-supervised representations for MOS prediction. In: Interspeech (2021)","DOI":"10.21437\/Interspeech.2021-2013"},{"key":"24_CR29","doi-asserted-by":"crossref","unstructured":"Tseng, W.C., Kao, W.T., Lee, H.Y.: DDOS: a MOS prediction framework utilizing domain adaptive pre-training and distribution of opinion scores. In: Interspeech (2022)","DOI":"10.21437\/Interspeech.2022-11247"},{"key":"24_CR30","doi-asserted-by":"crossref","unstructured":"Wan, L., Wang, Q., Papir, A., Moreno, I.L.: Generalized end-to-end loss for speaker verification. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4879\u20134883. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8462665"},{"key":"24_CR31","doi-asserted-by":"crossref","unstructured":"Wang, S., Qian, Y., Yu, K.: What does the speaker embedding encode? In: Interspeech, pp. 1497\u20131501 (2017)","DOI":"10.21437\/Interspeech.2017-1125"},{"key":"24_CR32","doi-asserted-by":"crossref","unstructured":"Wu, Z., Xie, Z., King, S.: The blizzard challenge 2019 (2019)","DOI":"10.21437\/Blizzard.2019-1"},{"key":"24_CR33","doi-asserted-by":"crossref","unstructured":"Yang, Z., et al.: Fusion of self-supervised learned models for MOS prediction. In: Proceedings of the Interspeech 2022, pp. 5443\u20135447 (2022). https:\/\/doi.org\/10.21437\/Interspeech.2022-10262","DOI":"10.21437\/Interspeech.2022-10262"},{"key":"24_CR34","doi-asserted-by":"publisher","first-page":"54","DOI":"10.1109\/TASLP.2022.3205757","volume":"31","author":"RE Zezario","year":"2022","unstructured":"Zezario, R.E., Fu, S.W., Chen, F., Fuh, C.S., Wang, H.M., Tsao, Y.: Deep learning-based non-intrusive multi-objective speech assessment model with cross-domain features. IEEE\/ACM Trans. Audio Speech Lang. Process. 31, 54\u201370 (2022)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."}],"container-title":["Lecture Notes in Computer Science","Text, Speech, and Dialogue"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-40498-6_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T16:18:56Z","timestamp":1729959536000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-40498-6_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031404979","9783031404986"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-40498-6_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"23 August 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"TSD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Text, Speech, and Dialogue","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pilsen","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Czech Republic","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 September 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 September 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"tsd2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.kiv.zcu.cz\/tsd2023\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMS & back-office system","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"64","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"31","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"48% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.56","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}