{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:12:58Z","timestamp":1750219978057,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,15]],"date-time":"2021-10-15T00:00:00Z","timestamp":1634256000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62006228"],"award-info":[{"award-number":["62006228"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,15]]},"DOI":"10.1145\/3497623.3497639","type":"proceedings-article","created":{"date-parts":[[2022,2,5]],"date-time":"2022-02-05T00:30:14Z","timestamp":1644021014000},"page":"99-105","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["LightCvT: Audio Forgery Detection via Fusion of Light CNN and Transformer"],"prefix":"10.1145","author":[{"given":"Chenyu","family":"Liu","sequence":"first","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jia","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junxian","family":"Duan","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haifeng","family":"Shen","sequence":"additional","affiliation":[{"name":"AI Enabling Platform, Didi Chuxing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huaibo","family":"Huang","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,2,4]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Non-local neural networks","author":"Wang X.","year":"2017","unstructured":"X. Wang , R. Girshick , A. Gupta , and K. He , \u201c Non-local neural networks ,\u201d 2017 . X. Wang, R. Girshick, A. Gupta, and K. He, \u201cNon-local neural networks,\u201d 2017."},{"key":"e_1_3_2_1_2_1","first-page":"03762","article-title":"Attention is all you need","volume":"1706","author":"Shazeer N.","year":"2017","unstructured":"Vaswani, N. Shazeer , N. Parmar , J. Uszkoreit , L. Jones , A. N. Gomez , L. Kaiser , and I. Polosukhin , \u201c Attention is all you need ,\u201d CoRR , vol. abs\/ 1706 . 03762 , 2017 . arXiv: 1706.03762. [Online]. Available: http:\/\/arxiv.org\/abs\/1706.03762. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, L. Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d CoRR, vol. abs\/1706.03762, 2017. arXiv: 1706.03762. [Online]. Available: http:\/\/arxiv.org\/abs\/1706.03762.","journal-title":"CoRR"},{"key":"e_1_3_2_1_3_1","first-page":"04805","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","volume":"1810","author":"Devlin J.","year":"2018","unstructured":"J. Devlin , M.-W. Chang , K. Lee , and K. Toutanova , \u201c BERT: pre-training of deep bidirectional transformers for language understanding ,\u201d CoRR , vol. abs\/ 1810 . 04805 , 2018 . arXiv: 1810 . 04805. [Online]. Available: http:\/\/arxiv.org\/abs\/1810.04805. J. Devlin, M.-W. Chang, K. Lee, and K. Toutanova, \u201cBERT: pre-training of deep bidirectional transformers for language understanding,\u201d CoRR, vol. abs\/1810.04805, 2018. arXiv: 1810 . 04805. [Online]. Available: http:\/\/arxiv.org\/abs\/1810.04805.","journal-title":"CoRR"},{"key":"e_1_3_2_1_4_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy A.","year":"2020","unstructured":"A. Dosovitskiy , L. Beyer , and A. Kolesnikov , \u201c An image is worth 16x16 words: Transformers for image recognition at scale ,\u201d 2020 . A. Dosovitskiy, L. Beyer, and A. Kolesnikov, \u201cAn image is worth 16x16 words: Transformers for image recognition at scale,\u201d 2020."},{"key":"e_1_3_2_1_5_1","first-page":"1","article-title":"A light cnn for deep face representation with noisy labels","author":"Wu X.","year":"2015","unstructured":"X. Wu , R. He , Z. Sun , and T. Tan , \u201c A light cnn for deep face representation with noisy labels ,\u201d IEEE Transactions on Information Forensics Security , pp. 1 \u2013 1 , 2015 . X. Wu, R. He, Z. Sun, and T. Tan, \u201cA light cnn for deep face representation with noisy labels,\u201d IEEE Transactions on Information Forensics Security, pp. 1\u20131, 2015.","journal-title":"IEEE Transactions on Information Forensics Security"},{"key":"e_1_3_2_1_6_1","first-page":"1072","volume-title":"Sep. 2019","author":"Gomez-Alanis A.","unstructured":"A. Gomez-Alanis , A. Peinado , J. Gonzalez Lopez , and A. Gomez , \u201c A light convolutional gru-rnn deep feature extractor for asv spoofing detection ,\u201d Sep. 2019 , pp. 1068\u2013 1072 . DOI: 10.21437\/Interspeech.2019-2212. 10.21437\/Interspeech.2019-2212 A. Gomez-Alanis, A. Peinado, J. Gonzalez Lopez, and A. Gomez, \u201cA light convolutional gru-rnn deep feature extractor for asv spoofing detection,\u201d Sep. 2019, pp. 1068\u20131072. DOI: 10.21437\/Interspeech.2019-2212."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"e_1_3_2_1_8_1","first-page":"1052","volume-title":"Sep. 2019","author":"Li R.","unstructured":"R. Li , M. Zhao , Z. Li , L. Li , and Q. Hong , \u201c Anti-spoofing speaker verification system with multifeatured integration and multi-task learning ,\u201d Sep. 2019 , pp. 1048\u2013 1052 . DOI: 10.21437\/Interspeech.2019-1698. 10.21437\/Interspeech.2019-1698 R. Li, M. Zhao, Z. Li, L. Li, and Q. Hong, \u201cAnti-spoofing speaker verification system with multifeatured integration and multi-task learning,\u201d Sep. 2019, pp. 1048\u20131052. DOI: 10.21437\/Interspeech.2019-1698."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAAI.2018.00046"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/2999134.2999257"},{"key":"e_1_3_2_1_11_1","volume-title":"Computer Science","author":"Simonyan K.","year":"2014","unstructured":"K. Simonyan and A. Zisserman , \u201c Very deep convolutional networks for large-scale image recognition ,\u201d Computer Science , 2014 . K. Simonyan and A. Zisserman, \u201cVery deep convolutional networks for large-scale image recognition,\u201d Computer Science, 2014."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_13_1","volume-title":"Inception-v4, inception-resnet and the impact of residual connections on learning","author":"Ioffe S.","year":"2016","unstructured":"Szegedy, S. Ioffe , and V. Vanhoucke , \u201c Inception-v4, inception-resnet and the impact of residual connections on learning ,\u201d 2016 . Szegedy, S. Ioffe, and V. Vanhoucke, \u201cInception-v4, inception-resnet and the impact of residual connections on learning,\u201d 2016."},{"key":"e_1_3_2_1_14_1","first-page":"05431","article-title":"Aggregated residual transformations for deep neural networks","volume":"1611","author":"Xie S.","year":"2016","unstructured":"S. Xie , R. B. Girshick , P. Doll\u00b4ar , Z. Tu , and K. He , \u201c Aggregated residual transformations for deep neural networks ,\u201d CoRR , vol. abs\/ 1611 . 05431 , 2016 . arXiv: 1611.05431. [Online]. Available: http:\/\/arxiv.org\/abs\/1611.05431. S. Xie, R. B. Girshick, P. Doll\u00b4ar, Z. Tu, and K. He, \u201cAggregated residual transformations for deep neural networks,\u201d CoRR, vol. abs\/1611.05431, 2016. arXiv: 1611.05431. [Online]. Available: http:\/\/arxiv.org\/abs\/1611.05431.","journal-title":"CoRR"},{"key":"e_1_3_2_1_15_1","volume-title":"Oct.","author":"Hu J.","year":"2018","unstructured":"J. Hu , L. Shen , S. Albanie , and G. Sun , Gather-excite: Exploiting feature context in convolutional neural networks , Oct. 2018 . J. Hu, L. Shen, S. Albanie, and G. Sun, Gather-excite: Exploiting feature context in convolutional neural networks, Oct. 2018."},{"key":"e_1_3_2_1_16_1","first-page":"7141","volume-title":"Jun. 2018","author":"Hu J.","unstructured":"J. Hu , L. Shen , and G. Sun , \u201c Squeeze-and-excitation networks ,\u201d Jun. 2018 , pp. 7132\u2013 7141 . DOI: 10.1109\/CVPR.2018.00745. 10.1109\/CVPR.2018.00745 J. Hu, L. Shen, and G. Sun, \u201cSqueeze-and-excitation networks,\u201d Jun. 2018, pp. 7132\u20137141. DOI: 10.1109\/CVPR.2018.00745."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00338"},{"key":"e_1_3_2_1_18_1","volume-title":"Gcnet: Nonlocal networks meet squeeze-excitation networks and beyond","author":"Cao Y.","year":"2019","unstructured":"Y. Cao , J. Xu , S. Lin , F. Wei , and H. Hu , \u201c Gcnet: Nonlocal networks meet squeeze-excitation networks and beyond ,\u201d arXiv, 2019 . Y. Cao, J. Xu, S. Lin, F. Wei, and H. Hu, \u201cGcnet: Nonlocal networks meet squeeze-excitation networks and beyond,\u201d arXiv, 2019."},{"key":"e_1_3_2_1_19_1","unstructured":"H. Hu J. Gu Z. Zhang J. Dai and Y. Wei \u201cRelation networks for object detection \u201d  H. Hu J. Gu Z. Zhang J. Dai and Y. Wei \u201cRelation networks for object detection \u201d"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"A. Srinivas T. Y. Lin N. Parmar J. Shlens and A.Vaswani \u201cBottleneck transformers for visual recognition \u201d2021.  A. Srinivas T. Y. Lin N. Parmar J. Shlens and A.Vaswani \u201cBottleneck transformers for visual recognition \u201d2021.","DOI":"10.1109\/CVPR46437.2021.01625"},{"key":"e_1_3_2_1_21_1","volume-title":"Training data-efficient image transformers distillation through attention","author":"Touvron H.","year":"2020","unstructured":"H. Touvron , M. Cord , M. Douze , and F. Massa , \u201c Training data-efficient image transformers distillation through attention ,\u201d 2020 . H. Touvron, M. Cord, M. Douze, and F. Massa, \u201cTraining data-efficient image transformers distillation through attention,\u201d 2020."},{"key":"e_1_3_2_1_22_1","volume-title":"Tokens-to-token vit: Training vision transformers from scratch on imagenet","author":"Yuan L.","year":"2021","unstructured":"L. Yuan , Y. Chen , T. Wang , W. Yu , Y. Shi , F. E. Tay , J. Feng , and S. Yan , \u201c Tokens-to-token vit: Training vision transformers from scratch on imagenet ,\u201d 2021 . L. Yuan, Y. Chen, T. Wang, W. Yu, Y. Shi, F. E. Tay, J.Feng, and S. Yan, \u201cTokens-to-token vit: Training vision transformers from scratch on imagenet,\u201d 2021."},{"key":"e_1_3_2_1_23_1","volume-title":"Swin transformer: Hierarchical vision transformer using shifted windows","author":"Liu Z.","year":"2021","unstructured":"Z. Liu , Y. Lin , Y. Cao , H. Hu , Y. Wei , and Z. Zhang , \u201c Swin transformer: Hierarchical vision transformer using shifted windows ,\u201d 2021 . Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, and Z. Zhang, \u201cSwin transformer: Hierarchical vision transformer using shifted windows,\u201d 2021."},{"key":"e_1_3_2_1_24_1","volume-title":"Cswin transformer: A general vision transformer backbone with cross-shaped windows","author":"Dong X.","year":"2021","unstructured":"X. Dong , J. Bao , D. Chen , W. Zhang , N. Yu , L. Yuan , D. Chen , and B. Guo , \u201c Cswin transformer: A general vision transformer backbone with cross-shaped windows ,\u201d 2021 . X. Dong, J. Bao, D. Chen, W. Zhang, N. Yu, L. Yuan, D. Chen, and B. Guo, \u201cCswin transformer: A general vision transformer backbone with cross-shaped windows,\u201d 2021."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_26_1","volume-title":"Cvt: Introducing convolutions to vision transformers","author":"Wu H.","year":"2021","unstructured":"H. Wu , B. Xiao , N. Codella , M. Liu , X. Dai , L. Yuan , and L. Zhang , \u201c Cvt: Introducing convolutions to vision transformers ,\u201d 2021 . H. Wu, B. Xiao, N. Codella, M. Liu, X. Dai, L. Yuan, and L. Zhang, \u201cCvt: Introducing convolutions to vision transformers,\u201d 2021."},{"key":"e_1_3_2_1_27_1","volume-title":"Coatnet: Marrying convolution and attention for all data sizes","author":"Dai Z.","year":"2021","unstructured":"Z. Dai , Liu, Q. V. Le , and M. Tan , \u201c Coatnet: Marrying convolution and attention for all data sizes ,\u201d 2021 . Z. Dai, Liu, Q. V. Le, and M. Tan, \u201cCoatnet: Marrying convolution and attention for all data sizes,\u201d 2021."},{"key":"e_1_3_2_1_28_1","volume-title":"Mobile-former: Bridging mobilenet and transformer","author":"Chen Y.","year":"2021","unstructured":"Y. Chen , X. Dai , D. Chen , M. Liu , X. Dong , L. Yuan , and Z. Liu , Mobile-former: Bridging mobilenet and transformer , 2021 . arXiv: 2108.05895 [cs.CV]. Y. Chen, X. Dai, D. Chen, M. Liu, X. Dong, L. Yuan, and Z. Liu, Mobile-former: Bridging mobilenet and transformer, 2021. arXiv: 2108.05895 [cs.CV]."},{"key":"e_1_3_2_1_29_1","volume-title":"Local-to-global self-attention in vision transformers","author":"Li J.","year":"2021","unstructured":"J. Li , Y. Yan , S. Liao , X. Yang , and L. Shao , \u201c Local-to-global self-attention in vision transformers ,\u201d 2021 . J. Li, Y. Yan, S. Liao, X. Yang, and L. Shao, \u201cLocal-to-global self-attention in vision transformers,\u201d 2021."},{"key":"e_1_3_2_1_30_1","volume-title":"Conformer: Local features coupling global representations for visual recognition","author":"Peng Z.","year":"2021","unstructured":"Z. Peng , W. Huang , S. Gu , L. Xie , and Q. Ye , \u201c Conformer: Local features coupling global representations for visual recognition ,\u201d 2021 . Z. Peng, W. Huang, S. Gu, L. Xie, and Q. Ye, \u201cConformer: Local features coupling global representations for visual recognition,\u201d 2021."},{"key":"e_1_3_2_1_31_1","volume-title":"Conformer: Convolution-augmented transformer for speech recognition","author":"Gulati A.","year":"2020","unstructured":"A. Gulati , J. Qin , C. C. Chiu , N. Parmar , Y. Zhang , J. Yu , W. Han , S. Wang , Z. Zhang , and Y. Wu , \u201c Conformer: Convolution-augmented transformer for speech recognition ,\u201d 2020 . A. Gulati, J. Qin, C. C. Chiu, N. Parmar, Y. Zhang, J. Yu, W. Han, S. Wang, Z. Zhang, and Y. Wu, \u201cConformer: Convolution-augmented transformer for speech recognition,\u201d 2020."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2017.01.001"},{"volume-title":"Light cnn architecture enhancement for different types spoofing attack detection","year":"2019","key":"e_1_3_2_1_33_1","unstructured":"\u201c Light cnn architecture enhancement for different types spoofing attack detection ,\u201d 2019 . \u201cLight cnn architecture enhancement for different types spoofing attack detection,\u201d 2019."},{"volume-title":"Asvspoof 2019: A large-scale public database of synthesized, converted and replayed speech","year":"2019","key":"e_1_3_2_1_34_1","unstructured":"\u201c Asvspoof 2019: A large-scale public database of synthesized, converted and replayed speech ,\u201d 2019 . \u201cAsvspoof 2019: A large-scale public database of synthesized, converted and replayed speech,\u201d 2019."},{"volume-title":"Interspeech","year":"2017","key":"e_1_3_2_1_35_1","unstructured":"Wu, Zhizheng, Kinnunen, Tomi, Evans, Nicholas, Yamagishi, and Junichi , \u201c Asvspoof 2015: The first automatic speaker veri cation spoofi ng and countermeasures challenge ,\u201d in Interspeech , 2017 . Wu, Zhizheng, Kinnunen, Tomi, Evans, Nicholas, Yamagishi, and Junichi, \u201cAsvspoof 2015: The first automatic speaker veri cation spoofi ng and countermeasures challenge,\u201d in Interspeech, 2017."},{"key":"e_1_3_2_1_36_1","volume-title":"Odyssey 2018 - The Speaker and Language Recognition Workshop","author":"Lgado H. De","year":"2018","unstructured":"H. De Lgado , M. Todisco , M. Sahidullah , N. Evans , and J. Yamagishi , \u201c Asvspoof 2017 version 2.0: Meta-data analysis and baseline enhancements ,\u201d in Odyssey 2018 - The Speaker and Language Recognition Workshop , 2018 . H. De Lgado, M. Todisco, M. Sahidullah, N. Evans, and J. Yamagishi, \u201cAsvspoof 2017 version 2.0: Meta-data analysis and baseline enhancements,\u201d in Odyssey 2018 - The Speaker and Language Recognition Workshop, 2018."},{"key":"e_1_3_2_1_37_1","volume-title":"T-dcf: A detection cost function for the tandem assessment of spoofing countermeasures and automatic speaker verification","author":"Kinnunen T.","year":"2018","unstructured":"T. Kinnunen , K. A. Lee , H. Delgado , N. Evans , M. Todisco , M. Sahidullah , and J. Yamagishi , \u201c T-dcf: A detection cost function for the tandem assessment of spoofing countermeasures and automatic speaker verification ,\u201d Jun. 2018 . DOI: 10 . 21437 \/ Odyssey.2018-44. T. Kinnunen, K. A. Lee, H. Delgado, N. Evans, M. Todisco, M. Sahidullah, and J. Yamagishi, \u201cT-dcf: A detection cost function for the tandem assessment of spoofing countermeasures and automatic speaker verification,\u201d Jun. 2018. DOI: 10 . 21437 \/ Odyssey.2018-44."},{"key":"e_1_3_2_1_38_1","first-page":"1047","volume-title":"Sep. 2019","author":"Alluri K. N. R. K.","unstructured":"K. N. R. K. Alluri and A. Vuppala , \u201c Iiit-h spoofing countermeasures for automatic speaker verification spoofing and countermeasures challenge 2019 ,\u201d Sep. 2019 , pp. 1043\u2013 1047 . DOI: 10.21437\/Interspeech.2019-1623. 10.21437\/Interspeech.2019-1623 K. N. R. K. Alluri and A. Vuppala, \u201cIiit-h spoofing countermeasures for automatic speaker verification spoofing and countermeasures challenge 2019,\u201d Sep. 2019, pp. 1043\u20131047. DOI: 10.21437\/Interspeech.2019-1623."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003845"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3174"},{"key":"e_1_3_2_1_41_1","volume-title":"Interspeech","author":"Lavrentyeva G.","year":"2019","unstructured":"G. Lavrentyeva , S. Novoselov , and A. Tseren , \u201c Stc antispoofing systems for the asvspoof2019 challenge ,\u201d in Interspeech , 2019 . G. Lavrentyeva, S. Novoselov, and A. Tseren, \u201cStc antispoofing systems for the asvspoof2019 challenge,\u201d in Interspeech, 2019."}],"event":{"name":"ICCPR '21: 2021 10th International Conference on Computing and Pattern Recognition","acronym":"ICCPR '21","location":"Shanghai China"},"container-title":["2021 10th International Conference on Computing and Pattern Recognition"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3497623.3497639","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3497623.3497639","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:49:23Z","timestamp":1750182563000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3497623.3497639"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,15]]},"references-count":41,"alternative-id":["10.1145\/3497623.3497639","10.1145\/3497623"],"URL":"https:\/\/doi.org\/10.1145\/3497623.3497639","relation":{},"subject":[],"published":{"date-parts":[[2021,10,15]]},"assertion":[{"value":"2022-02-04","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}