{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T04:24:21Z","timestamp":1764995061396,"version":"3.46.0"},"reference-count":40,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"12","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,12,1]]},"DOI":"10.1587\/transinf.2025edp7044","type":"journal-article","created":{"date-parts":[[2025,6,23]],"date-time":"2025-06-23T18:06:39Z","timestamp":1750701999000},"page":"1594-1604","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing the Robustness of Speech Anti-Spoofing Countermeasures through Joint Optimization and Transfer Learning"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Yikang","family":"WANG","sequence":"first","affiliation":[{"name":"Integrated Graduate School of Medicine, Engineering, and Agricultural Sciences, University of Yamanashi"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xingming","family":"WANG","sequence":"additional","affiliation":[{"name":"Suzhou Municipal Key Laboratory of Multimodal Intelligent Systems, Duke Kunshan University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chee Siang","family":"LEOW","sequence":"additional","affiliation":[{"name":"Integrated Graduate School of Medicine, Engineering, and Agricultural Sciences, University of Yamanashi"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qishan","family":"ZHANG","sequence":"additional","affiliation":[{"name":"Suzhou Municipal Key Laboratory of Multimodal Intelligent Systems, Duke Kunshan University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"LI","sequence":"additional","affiliation":[{"name":"Suzhou Municipal Key Laboratory of Multimodal Intelligent Systems, Duke Kunshan University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hiromitsu","family":"NISHIZAKI","sequence":"additional","affiliation":[{"name":"Integrated Graduate School of Medicine, Engineering, and Agricultural Sciences, University of Yamanashi"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","unstructured":"[1] J. Yi, C. Wang, J. Tao, X. Zhang, C. Zhang, and Y. Zhao, \u201cAudio Deepfake Detection: A survey,\u201d arXiv preprint arXiv:2308.14970, 2023."},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] H. Delgado, M. Todisco, M. Sahidullah, A.K. Sarkar, N. Evans, T. Kinnunen, and Z.H. Tan, \u201cFurther Optimisations of Constant Q Cepstral Processing for Integrated Utterance and Text-dependent Speaker Verification,\u201d Proc. of SLT, pp.179-185, 2016. 10.1109\/slt.2016.7846262","DOI":"10.1109\/SLT.2016.7846262"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] G. Lavrentyeva, S. Novoselov, A. Tseren, M. Volkova, A. Gorlanov, and A. Kozlov, \u201cSTC Antispoofing Systems for the ASVspoof2019 Challenge,\u201d Proc. of Interspeech, pp.1033-1037, 2019. 10.21437\/interspeech.2019-1768","DOI":"10.21437\/Interspeech.2019-1768"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] M. Alzantot, Z. Wang, and M.B. Srivastava, \u201cDeep Residual Neural Networks for Audio Spoofing Detection,\u201d Proc. of Interspeech, pp.1078-1082, 2019. 10.21437\/interspeech.2019-3174","DOI":"10.21437\/Interspeech.2019-3174"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] F. Chen, S. Deng, T. Zheng, Y. He, and J. Han, \u201cGraph-Based Spectro-Temporal Dependency Modeling for Anti-Spoofing,\u201d Proc. of ICASSP, pp.1-5, 2023. 10.1109\/icassp49357.2023.10096741","DOI":"10.1109\/ICASSP49357.2023.10096741"},{"key":"6","doi-asserted-by":"publisher","unstructured":"[6] F. Scarselli, M. Gori, A.C. Tsoi, M. Hagenbuchner, and G.Monfardini, \u201cThe Graph Neural Network Model,\u201d IEEE Trans. Neural Netw., vol.20, no.1, pp.61-80, 2009. 10.1109\/tnn.2008.2005605","DOI":"10.1109\/TNN.2008.2005605"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] T. Kinnunen, M. Sahidullah, H. Delgado, M. Todisco, N. Evans, J. Yamagishi, and K.A. Lee, \u201cThe ASVspoof 2017 Challenge: Assessing the Limits of Replay Spoofing Attack Detection,\u201d Proc. of Interspeech, pp.2-6, 2017. 10.21437\/interspeech.2017-1111","DOI":"10.21437\/Interspeech.2017-1111"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] M. Todisco, X. Wang, V. Vestman, M. Sahidullah, H. Delgado, A. Nautsch, J. Yamagishi, N. Evans, T.H. Kinnunen, and K.A. Lee, \u201cASVspoof 2019: Future Horizons in Spoofed and Fake Audio Detection,\u201d Proc. of Interspeech, pp.1008-1012, 2019. 10.21437\/interspeech.2019-2249","DOI":"10.21437\/Interspeech.2019-2249"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] J. Xue, C. Fan, J. Yi, C. Wang, Z. Wen, D. Zhang, and Z. Lv, \u201cLearning from Yourself: A Self-Distillation Method for Fake Speech Detection,\u201d Proc. of ICASSP, pp.1-5, 2023. 10.1109\/icassp49357.2023.10096837","DOI":"10.1109\/ICASSP49357.2023.10096837"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] X. Wang and J. Yamagishi, \u201cA Comparative Study on Recent Neural Spoofing Countermeasures for Synthetic Speech Detection,\u201d Proc. of Interspeech, pp.4259-4263, 2021. 10.21437\/interspeech.2021-702","DOI":"10.21437\/Interspeech.2021-702"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] M. Sahidullah, T. Kinnunen, and C. Hanil\u00e7i, \u201cA Comparison of Features for Synthetic Speech Detection,\u201d Proc. of Interspeech, pp.2087-2091, 2015. 10.21437\/interspeech.2015-472","DOI":"10.21437\/Interspeech.2015-472"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] M. Todisco, H. Delgado, K.A. Lee, M. Sahidullah, N. Evans, T. Kinnunen, and J. Yamagishi, \u201cIntegrated Presentation Attack Detection and Automatic Speaker Verification: Common Features and Gaussian Back-end Fusion,\u201d Proc. of Interspeech, pp.77-81, 2018. 10.21437\/interspeech.2018-2289","DOI":"10.21437\/Interspeech.2018-2289"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] H. Tak, J. Patino, M. Todisco, A. Nautsch, N. Evans, and A. Larcher, \u201cEnd-to-end Anti-spoofing with Rawnet2,\u201d Proc. of ICASSP, pp.6369-6373, 2021. 10.1109\/icassp39728.2021.9414234","DOI":"10.1109\/ICASSP39728.2021.9414234"},{"key":"14","doi-asserted-by":"crossref","unstructured":"[14] H. Tak, J.-W. Jung, J. Patino, M. Kamble, M. Todisco, and N. Evans, \u201cEnd-to-End Spectro-Temporal Graph Attention Networks for Speaker Verification Anti-Spoofing and Speech Deepfake Detection,\u201d Proc. of ASVspoof Workshop, pp.1-8, 2021. 10.21437\/asvspoof.2021-1","DOI":"10.21437\/ASVSPOOF.2021-1"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] J. Weon Jung, H.-S. Heo, H. Tak, H. Jin Shim, J.S. Chung, B.-J. Lee, H. Jin Yu, and N. Evans, \u201cAASIST: Audio Anti-Spoofing Using Integrated Spectro-Temporal Graph Attention Networks,\u201d Proc. of ICASSP, pp.6367\u20146371, 2022. 10.1109\/icassp43922.2022.9747766","DOI":"10.1109\/ICASSP43922.2022.9747766"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] X. Liu, M. Liu, L. Wang, K.A. Lee, H. Zhang, and J. Dang, \u201cLeveraging Positional-Related Local-Global Dependency for Synthetic Speech Detection,\u201d Proc. of ICASSP, pp.1-5, 2023. 10.1109\/icassp49357.2023.10096278","DOI":"10.1109\/ICASSP49357.2023.10096278"},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] J. Yamagishi, X. Wang, M. Todisco, M. Sahidullah, J. Patino, A. Nautsch, X. Liu, K.A. Lee, T. Kinnunen, N. Evans, and H. Delgado, \u201cASVspoof 2021: Accelerating Progress in Spoofed and Deepfake Speech Detection,\u201d Proc. of ASVspoof Workshop, pp.47-54, 2021. 10.21437\/asvspoof.2021-8","DOI":"10.21437\/ASVSPOOF.2021-8"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] H. Tak, M.R. Kamble, J. Patino, M. Todisco, and N. Evans, \u201cRawboost: A Raw Data Boosting and Augmentation Method Applied to Automatic Speaker Verification Anti-Spoofing,\u201d Proc. of ICASSP, pp.6382-6386, 2022. 10.1109\/icassp43922.2022.9746213","DOI":"10.1109\/ICASSP43922.2022.9746213"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] Y. Wang, X. Wang, H. Nishizaki, and M. Li, \u201cLow Pass Filtering and Bandwidth Extension for Robust Anti-Spoofing Countermeasure Against Codec Variabilities,\u201d Proc. of ISCSLP, pp.438-442, 2022. 10.1109\/iscslp57327.2022.10038240","DOI":"10.1109\/ISCSLP57327.2022.10038240"},{"key":"20","doi-asserted-by":"publisher","unstructured":"[20] C. Hanil\u00e7i, T. Kinnunen, M. Sahidullah, and A. Sizov, \u201cSpoofing Detection Goes Noisy: An Analysis of Synthetic Speech Detection in the Presence of Additive Noise,\u201d Speech Communication, vol.85, pp.83-97, 2016. 10.1016\/j.specom.2016.10.002","DOI":"10.1016\/j.specom.2016.10.002"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] H. Yu, A. Sarkar, D.A.L. Thomsen, Z.-H. Tan, Z. Ma, and J. Guo, \u201cEffect of Multi-condition Training and Speech Enhancement Methods on Spoofing Detection,\u201d Proc. of SPLINE, pp.1-5, 2016. 10.1109\/splim.2016.7528399","DOI":"10.1109\/SPLIM.2016.7528399"},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] C. Fan, M. Ding, J. Tao, R. Fu, J. Yi, Z. Wen, and Z. Lv, \u201cDual-Branch Knowledge Distillation for Noise-Robust Synthetic Speech Detection,\u201d IEEE\/ACM Trans. on Audio, Speech, and Language Processing, vol.32, pp.2453-2466, 2024. 10.1109\/taslp.2024.3389643","DOI":"10.1109\/TASLP.2024.3389643"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] X. Wang, B. Zeng, S. Hongbin, Y. Wan, and M. Li, \u201cRobust Audio Anti-spoofing Countermeasure with Joint Training of Front-end and Back-end Models,\u201d Proc. of Interspeech, pp.4004-4008, 2023. 10.21437\/interspeech.2023-1166","DOI":"10.21437\/Interspeech.2023-1166"},{"key":"24","doi-asserted-by":"publisher","unstructured":"[24] V. Valimaki, J.D. Parker, L. Savioja, J.O. Smith, and J.S. Abel, \u201cFifty Years of Artificial Reverberation,\u201d IEEE\/ACM Trans. on Audio, Speech, and Language Processing, vol.20, no.5, pp.1421-1448, 2012. 10.1109\/tasl.2012.2189567","DOI":"10.1109\/TASL.2012.2189567"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] J.-H. Kim, J. Heo, H.-J. Shim, and H.-J. Yu, \u201cExtended U-Net for Speaker Verification in Noisy Environments,\u201d Proc. of Interspeech, pp.590-594, 2022. 10.21437\/interspeech.2022-155","DOI":"10.21437\/Interspeech.2022-155"},{"key":"26","unstructured":"[26] A.V. Oppenheim, A.S. Willsky, and S.H. Nawab, Signals and Systems, 2nd ed., ch. 4, sec. 4, 2000."},{"key":"27","doi-asserted-by":"crossref","unstructured":"[27] A. Gulati, J. Qin, C.-C. Chiu, N. Parmar, Y. Zhang, J. Yu, W. Han, S. Wang, Z. Zhang, Y. Wu, and R. Pang, \u201cConformer: Convolution-augmented Transformer for Speech Recognition,\u201d Proc. of Interspeech, pp.5036-5040, 2020. 10.21437\/interspeech.2020-3015","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"28","unstructured":"[28] O. Kuchaiev, J. Li, H. Nguyen, O. Hrinchuk, R. Leary, B.Ginsburg, S. Kriman, S. Beliaev, V. Lavrukhin, J. Cook, et al., \u201cNemo: a Toolkit for Building AI Applications Using Neural Modules,\u201d arXiv:1909.09577, 2019."},{"key":"29","doi-asserted-by":"crossref","unstructured":"[29] Y. Zhang, Z. Lv, H. Wu, S. Zhang, P. Hu, Z. Wu, H.-Y. Lee, and H. Meng, \u201cMFA-Conformer: Multi-scale Feature Aggregation Conformer for Automatic Speaker Verification,\u201d Proc. of Interspeech, pp.306-310, 2022. 10.21437\/interspeech.2022-563","DOI":"10.21437\/Interspeech.2022-563"},{"key":"30","doi-asserted-by":"publisher","unstructured":"[30] D. Cai and M. Li, \u201cLeveraging ASR Pretrained Conformers for Speaker Verification Through Transfer Learning and Knowledge Distillation,\u201d IEEE\/ACM Trans. Audio, Speech, Language Process., vol.32, pp.3532-3545, 2024. 10.1109\/taslp.2024.3419426","DOI":"10.1109\/TASLP.2024.3419426"},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] K. He, X. Zhang, S. Ren, and J. Sun, \u201cDeep Residual Learning for Image Recognition,\u201d Proc. of the CVPR, pp.770-778, 2016. 10.1109\/cvpr.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"32","unstructured":"[32] I.J. Goodfellow, D. Warde-Farley, M. Mirza, A.C. Courville, and Y. Bengio, \u201cMaxout Networks,\u201d Proc. of ICML, pp.1319-1327, 2013."},{"key":"33","doi-asserted-by":"crossref","unstructured":"[33] M. Schuster and K.K. Paliwal, \u201cBidirectional Recurrent Neural Networks,\u201d IEEE Trans. on Signal Processing, vol.45, no.11, pp.2673-2681, 1997. 10.1109\/78.650093","DOI":"10.1109\/78.650093"},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] Z. Wu, T. Kinnunen, N. Evans, J. Yamagishi, Hanil\u00e7i, M. Sahidullah,and A. Sizov, \u201cASVspoof 2015: the First Automatic Speaker Verification Spoofing and Countermeasures Challenge,\u201d Proc. of Interspeech, pp.2037-2041, 2015. 10.21437\/interspeech.2015-462","DOI":"10.21437\/Interspeech.2015-462"},{"key":"35","unstructured":"[35] J. Yamagishi, C. Veaux, and K. MacDonald, \u201cCSTR VCTK Corpus: English Multi-speaker Corpus for CSTR Voice Cloning Toolkit (version 0.92),\u201d https:\/\/doi.org\/10.7488\/ds\/2645, 2019. 10.7488\/ds\/2645"},{"key":"36","unstructured":"[36] D. Snyder, G. Chen, and D. Povey, \u201cMusan: A Music, Speech, and Noise Corpus,\u201d arXiv preprint arXiv:1510.08484, 2015."},{"key":"37","doi-asserted-by":"crossref","unstructured":"[37] Y. Luo and R. Gu, \u201cFast Random Approximation of Multi-channel Room Impulse Response,\u201d Proc. of ICASSP Workshop, pp.449-454, 2024. 10.1109\/icasspw62465.2024.10627689","DOI":"10.1109\/ICASSPW62465.2024.10627689"},{"key":"38","doi-asserted-by":"crossref","unstructured":"[38] T. Ko, V. Peddinti, D. Povey, M.L. Seltzer, and S. Khudanpur, \u201cA Study on Data Augmentation of Reverberant Speech for Robust Speech Recognition,\u201d Proc. of ICASSP, pp.5220-5224, 2017. 10.1109\/icassp.2017.7953152","DOI":"10.1109\/ICASSP.2017.7953152"},{"key":"39","doi-asserted-by":"crossref","unstructured":"[39] J. Hu, L. Shen, and G. Sun, \u201cSqueeze-and-excitation Networks,\u201d Proc. of CVPR, pp.7132-7141, 2018. 10.1109\/cvpr.2018.00745","DOI":"10.1109\/CVPR.2018.00745"},{"key":"40","unstructured":"[40] J. Yamagishi, M. Todisco, M. Sahidullah, H. Delgado, X. Wang, N. Evans, T. Kinnunen, K.A. Lee, V. Vestman, and A. Nautsch, \u201cASVspoof 2019: Automatic Speaker Verification Spoofing and Countermeasures Challenge Evaluation Plan,\u201d Proc. of ASVspoof Workshop, vol.13, 2019."}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/12\/E108.D_2025EDP7044\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T03:27:42Z","timestamp":1764991662000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/12\/E108.D_2025EDP7044\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,1]]},"references-count":40,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2025edp7044","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2025,12,1]]},"article-number":"2025EDP7044"}}