{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T07:57:06Z","timestamp":1743148626162,"version":"3.40.3"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031533105"},{"type":"electronic","value":"9783031533112"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-53311-2_8","type":"book-chapter","created":{"date-parts":[[2024,1,27]],"date-time":"2024-01-27T21:37:36Z","timestamp":1706391456000},"page":"101-111","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["ASF-Conformer: Audio Scoring Conformer with\u00a0FFC for\u00a0Speaker Verification in\u00a0Noisy Environments"],"prefix":"10.1007","author":[{"given":"Xiran","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haiyan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Caixia","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haiyang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiwei","family":"Huo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,1,28]]},"reference":[{"key":"8_CR1","doi-asserted-by":"publisher","unstructured":"Cai, D., Cai, W., Li, M.: Within-sample variability-invariant loss for robust speaker recognition under noisy environments. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6469\u20136473. IEEE (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9053407","DOI":"10.1109\/ICASSP40776.2020.9053407"},{"key":"8_CR2","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-030-67832-6_1","volume-title":"MultiMedia Modeling","author":"L Chen","year":"2021","unstructured":"Chen, L., Liang, Y., Shi, X., Zhou, Y., Wu, C.: Crossed-time delay neural network for speaker recognition. In: Loko\u010d, J., et al. (eds.) MMM 2021. LNCS, vol. 12572, pp. 1\u201310. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-67832-6_1"},{"key":"8_CR3","doi-asserted-by":"publisher","unstructured":"Chen, S., et al.: Continuous speech separation with conformer. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5749\u20135753. IEEE (2021). https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9413423","DOI":"10.1109\/ICASSP39728.2021.9413423"},{"key":"8_CR4","unstructured":"Chung, J.S., et al.: In defence of metric learning for speaker recognition. arXiv preprint arXiv:2003.11982 (2020)"},{"key":"8_CR5","doi-asserted-by":"crossref","unstructured":"Dai, Z., Yang, Z., Yang, Y., Carbonell, J., Le, Q.V., Salakhutdinov, R.: Transformer-xl: Attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860 (2019)","DOI":"10.18653\/v1\/P19-1285"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Desplanques, B., Thienpondt, J., Demuynck, K.: Ecapa-tdnn: emphasized channel attention, propagation and aggregation in tdnn based speaker verification. arXiv preprint arXiv:2005.07143 (2020)","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"8_CR7","doi-asserted-by":"publisher","unstructured":"Hong, J., Kim, M., Choi, J., Ro, Y.M.: Watch or listen: robust audio-visual speech recognition with visual corruption modeling and reliability scoring. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18783\u201318794 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01801","DOI":"10.1109\/CVPR52729.2023.01801"},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Jin, M., Yoo, C.D.: Speaker verification and identification. In: Behavioral Biometrics for Human Identification: Intelligent Applications, pp. 264\u2013289. IGI Global (2010)","DOI":"10.4018\/978-1-60566-725-6.ch013"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Jung, J.w., Heo, H.S., Kim, J.h., Shim, H.J., Yu, H.J.: Rawnet: advanced end-to-end deep neural network using raw waveforms for text-independent speaker verification. arXiv preprint arXiv:1904.08104 (2019)","DOI":"10.21437\/Interspeech.2019-1982"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Kim, J.h., Heo, J., Shim, H.j., Yu, H.J.: Extended u-net for speaker verification in noisy environments. arXiv preprint arXiv:2206.13044 (2022)","DOI":"10.21437\/Interspeech.2022-155"},{"key":"8_CR11","doi-asserted-by":"publisher","unstructured":"Koizumi, Y., et al.: Df-conformer: integrated architecture of conv-tasnet and conformer using linear complexity self-attention for speech enhancement. In: 2021 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA), pp. 161\u2013165. IEEE (2021). https:\/\/doi.org\/10.1109\/WASPAA52581.2021.9632794","DOI":"10.1109\/WASPAA52581.2021.9632794"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Li, Y., Lin, X.: Dual-stream time-delay neural network with dynamic global filter for speaker verification. arXiv preprint arXiv:2303.11020 (2023)","DOI":"10.1109\/TASLP.2024.3402072"},{"key":"8_CR13","doi-asserted-by":"publisher","unstructured":"Liu, T., Das, R.K., Lee, K.A., Li, H.: Mfa: Tdnn with multi-scale frequency-channel attention for text-independent speaker verification with short utterances. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7517\u20137521. IEEE (2022). https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9747021","DOI":"10.1109\/ICASSP43922.2022.9747021"},{"key":"8_CR14","doi-asserted-by":"publisher","unstructured":"Matejka, P., Novotn\u1ef3, O., Plchot, O., Burget, L., S\u00e1nchez, M.D., Cernock\u1ef3, J.: Analysis of score normalization in multilingual speaker recognition. In: Interspeech, pp. 1567\u20131571 (2017). https:\/\/doi.org\/10.21437\/Interspeech. 2017\u2013803","DOI":"10.21437\/Interspeech"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Chung, J.S., Zisserman, A.: Voxceleb: a large-scale speaker identification dataset. arXiv preprint arXiv:1706.08612 (2017)","DOI":"10.21437\/Interspeech.2017-950"},{"key":"8_CR16","unstructured":"Pelecanos, J., Sridharan, S.: Feature warping for robust speaker verification. In: Proceedings of 2001 A Speaker Odyssey: The Speaker Recognition Workshop, pp. 213\u2013218. European Speech Communication Association (2001)"},{"key":"8_CR17","unstructured":"Snyder, D., Chen, G., Povey, D.: Musan: A music, speech, and noise corpus. arXiv preprint arXiv:1510.08484 (2015)"},{"key":"8_CR18","doi-asserted-by":"publisher","unstructured":"Snyder, D., Garcia-Romero, D., Sell, G., Povey, D., Khudanpur, S.: X-vectors: robust dnn embeddings for speaker recognition. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5329\u20135333. IEEE (2018). https:\/\/doi.org\/10.1109\/ICASSP.2018.8461375","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Thienpondt, J., Desplanques, B., Demuynck, K.: Integrating frequency translational invariance in tdnns and frequency positional information in 2d resnets to enhance speaker verification. arXiv preprint arXiv:2104.02370 (2021)","DOI":"10.21437\/Interspeech.2021-1570"},{"key":"8_CR20","doi-asserted-by":"publisher","unstructured":"Variani, E., Lei, X., McDermott, E., Moreno, I.L., Gonzalez-Dominguez, J.: Deep neural networks for small footprint text-dependent speaker verification. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4052\u20134056. IEEE (2014). https:\/\/doi.org\/10.1109\/ICASSP.2014.6854363","DOI":"10.1109\/ICASSP.2014.6854363"},{"key":"8_CR21","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems 30 (2017)"},{"key":"8_CR22","unstructured":"Wang, C., et al.: Unispeech: unified speech representation learning with labeled and unlabeled data. In: International Conference on Machine Learning, pp. 10937\u201310947. PMLR (2021)"},{"key":"8_CR23","unstructured":"Yang, Z., Dai, Z., Yang, Y., Carbonell, J., Salakhutdinov, R.R., Le, Q.V.: Xlnet: generalized autoregressive pretraining for language understanding. In: Advances in Neural Information Processing Systems 32 (2019)"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Zhang, Y., et al.: Mfa-conformer: multi-scale feature aggregation conformer for automatic speaker verification. arXiv preprint arXiv:2203.15249 (2022)","DOI":"10.21437\/Interspeech.2022-563"},{"key":"8_CR25","doi-asserted-by":"publisher","unstructured":"Zhou, T., Zhao, Y., Wu, J.: Resnext and res2net structures for speaker verification. In: 2021 IEEE Spoken Language Technology Workshop (SLT), pp. 301\u2013307. IEEE (2021). https:\/\/doi.org\/10.1109\/SLT48900.2021.9383531","DOI":"10.1109\/SLT48900.2021.9383531"}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-53311-2_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T09:54:22Z","timestamp":1731146062000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-53311-2_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031533105","9783031533112"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-53311-2_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"28 January 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Amsterdam","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 January 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 February 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"ConfTool Pro","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"297","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"112","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"38% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.2","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.2","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}