{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T10:36:27Z","timestamp":1763202987910,"version":"3.40.4"},"publisher-location":"Cham","reference-count":22,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031882227","type":"print"},{"value":"9783031882234","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-88223-4_14","type":"book-chapter","created":{"date-parts":[[2025,4,24]],"date-time":"2025-04-24T03:45:09Z","timestamp":1745466309000},"page":"185-194","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Breaking the\u00a0Silence: Detecting AI-Converted Voices in\u00a0the\u00a0Quietest Moments"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-5784-0514","authenticated-orcid":false,"given":"Stefano","family":"Borz\u00ec","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lorenzo","family":"Mongelli","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3865-9911","authenticated-orcid":false,"given":"Filippo","family":"Stanco","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6127-2470","authenticated-orcid":false,"given":"Sebastiano","family":"Battiato","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4819-5340","authenticated-orcid":false,"given":"Dario","family":"Allegra","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,25]]},"reference":[{"key":"14_CR1","doi-asserted-by":"crossref","unstructured":"Borz\u00ec, S., Giudice, O., Stanco, F., Allegra, D.: Is synthetic voice detection research going into the right direction? In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 71\u201380 (2022)","DOI":"10.1109\/CVPRW56347.2022.00017"},{"key":"14_CR2","unstructured":"Brewster, T.: Fraudsters cloned company director\u2019s voice in \\$35 million heist, police find. Forbes (2021). https:\/\/www.forbes.com\/sites\/thomasbrewster\/2021\/10\/14\/huge-bank-fraud-uses-deep-fake-voice-tech-to-steal-millions\/?sh=42bdebd47559"},{"key":"14_CR3","unstructured":"Cochard, D.: RVC-an-AI-powered-voice-changer. medium (2024). https:\/\/medium.com\/axinc-ai\/rvc-an-ai-powered-voice-changer-39927cc83bee"},{"key":"14_CR4","unstructured":"Facebook engineering: Faiss: a library for efficient similarity search (2017). https:\/\/engineering.fb.com\/2017\/03\/29\/data-infrastructure\/faiss-a-library-for-efficient-similarity-search\/"},{"key":"14_CR5","doi-asserted-by":"crossref","unstructured":"Hennequin, R., Khlif, A., Voituret, F., Moussallam, M.: Spleeter: a fast and efficient music source separation tool with pre-trained models. J. Open Source Softw. 5(50), 2154 (2020), deezer Research","DOI":"10.21105\/joss.02154"},{"key":"14_CR6","doi-asserted-by":"crossref","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: Hubert: self-supervised speech representation learning by masked prediction of hidden units (2021)","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"14_CR7","unstructured":"Jordan J.\u00a0Bird, A.L.: Deep voice: real-time detection of AI-generated speech for deepfake voice conversion. Kaggle (2022). https:\/\/www.kaggle.com\/datasets\/birdy654\/deep-voice-deepfake-voice-recognition\/data"},{"key":"14_CR8","unstructured":"Kim, J., Kong, J., Son, J.: Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech (2021). https:\/\/arxiv.org\/abs\/2106.06103"},{"key":"14_CR9","doi-asserted-by":"crossref","unstructured":"Kim, J.W., Salamon, J., Li, P., Bello, J.P.: Crepe: a convolutional representation for pitch estimation (2018). https:\/\/arxiv.org\/abs\/1802.06182","DOI":"10.1109\/ICASSP.2018.8461329"},{"key":"14_CR10","doi-asserted-by":"publisher","first-page":"2507","DOI":"10.1109\/TASLP.2023.3285283","volume":"31","author":"X Liu","year":"2023","unstructured":"Liu, X., et al.: Asvspoof 2021: towards spoofed and deepfake speech detection in the wild. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 31, 2507\u20132522 (2023)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"issue":"1","key":"14_CR11","doi-asserted-by":"publisher","first-page":"2522","DOI":"10.1038\/s42256-019-0138-9","volume":"2","author":"SM Lundberg","year":"2020","unstructured":"Lundberg, S.M., et al.: From local explanations to global understanding with explainable AI for trees. Nat. Mach. Intell. 2(1), 2522\u20135839 (2020)","journal-title":"Nat. Mach. Intell."},{"key":"14_CR12","unstructured":"Lundberg, S.M., Lee, S.I.: A unified approach to interpreting model predictions. In: International Conference on Neural Information Processing Systems, pp. 4768\u20134777 (2017)"},{"key":"14_CR13","doi-asserted-by":"crossref","unstructured":"Mari, D., Latora, F., Milani, S.: The sound of silence: Efficiency of first digit features in synthetic audio detection. In: 2022 IEEE International Workshop on Information Forensics and Security (WIFS), pp.\u00a01\u20136. IEEE (2022)","DOI":"10.1109\/WIFS55849.2022.9975404"},{"key":"14_CR14","doi-asserted-by":"crossref","unstructured":"M\u00fcller, N.M., Czempin, P., Dieckmann, F., Froghyar, A., B\u00f6ttinger, K.: Does audio deepfake detection generalize? Interspeech (2022)","DOI":"10.21437\/Interspeech.2022-108"},{"key":"14_CR15","doi-asserted-by":"crossref","unstructured":"M\u00fcller, N.M., Dieckmann, F., Czempin, P., Canals, R., B\u00f6ttinger, K., Williams, J.: Speech is silver, silence is golden: what do asvspoof-trained models really learn? arXiv preprint arXiv:2106.12914 (2021)","DOI":"10.21437\/ASVSPOOF.2021-9"},{"key":"14_CR16","doi-asserted-by":"crossref","unstructured":"M\u00fcller, N.M., Pizzi, K., Williams, J.: Human perception of audio deepfakes. In: Proceedings of the 1st International Workshop on Deepfake Detection for Audio Multimedia, pp. 85\u201391 (2022)","DOI":"10.1145\/3552466.3556531"},{"key":"14_CR17","doi-asserted-by":"publisher","first-page":"50851","DOI":"10.1109\/ACCESS.2023.3276480","volume":"11","author":"D Salvi","year":"2023","unstructured":"Salvi, D., Hosler, B., Bestagini, P., Stamm, M.C., Tubaro, S.: Timit-tts: a text-to-speech dataset for multimodal synthetic media detection. IEEE Access 11, 50851\u201350866 (2023)","journal-title":"IEEE Access"},{"key":"14_CR18","doi-asserted-by":"crossref","unstructured":"Shim, H.J., Sahidullah, M., Jung, J.W., Watanabe, S., Kinnunen, T.: Beyond silence: bias analysis through loss and asymmetric approach in audio anti-spoofing. arXiv preprint arXiv:2406.17246 (2024)","DOI":"10.21437\/SynData4GenAI.2024-12"},{"key":"14_CR19","unstructured":"Stupp, C.: Fraudsters used AI to mimic ceo\u2019s voice in unusual cybercrime case. wsj (2019). https:\/\/www.wsj.com\/articles\/fraudsters-use-ai-to-mimic-ceos-voice-in-unusual-cybercrime-case-11567157402"},{"key":"14_CR20","doi-asserted-by":"crossref","unstructured":"Wang, L., Yu, L., Zhang, Y., Xie, H.: Generalizable speech spoofing detection against silence trimming with data augmentation and multi-task meta-learning. IEEE\/ACM Trans. Audio Speech Lang. Process. (2024)","DOI":"10.1109\/TASLP.2024.3414340"},{"key":"14_CR21","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Li, Z., Lu, J., Hua, H., Wang, W., Zhang, P.: The impact of silence on speech anti-spoofing. IEEE\/ACM Trans. Audio Speech Lang. Process. (2023)","DOI":"10.1109\/TASLP.2023.3306711"},{"key":"14_CR22","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Wang, W., Zhang, P.: The effect of silence and dual-band fusion in anti-spoofing system. In: Interspeech (2021)","DOI":"10.21437\/Interspeech.2021-1281"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition. ICPR 2024 International Workshops and Challenges"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-88223-4_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,24]],"date-time":"2025-04-24T03:45:33Z","timestamp":1745466333000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-88223-4_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031882227","9783031882234"],"references-count":22,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-88223-4_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"25 April 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kolkata","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}