{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T16:59:15Z","timestamp":1783789155602,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":30,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819783663","type":"print"},{"value":"9789819783670","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,29]],"date-time":"2024-11-29T00:00:00Z","timestamp":1732838400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,29]],"date-time":"2024-11-29T00:00:00Z","timestamp":1732838400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8367-0_25","type":"book-chapter","created":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T11:56:28Z","timestamp":1732794988000},"page":"419-433","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["EmoFake: An Initial Dataset for\u00a0Emotion Fake Audio Detection"],"prefix":"10.1007","author":[{"given":"Yan","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiangyan","family":"Yi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianhua","family":"Tao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenglong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongfeng","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,29]]},"reference":[{"issue":"10","key":"25_CR1","doi-asserted-by":"publisher","first-page":"1440","DOI":"10.1109\/LSP.2018.2860246","volume":"25","author":"M Chen","year":"2018","unstructured":"Chen, M., He, X., Yang, J., Zhang, H.: 3-D convolutional recurrent neural networks with attention model for speech emotion recognition. IEEE Signal Process. Lett. 25(10), 1440\u20131444 (2018)","journal-title":"IEEE Signal Process. Lett."},{"key":"25_CR2","doi-asserted-by":"publisher","first-page":"110","DOI":"10.1016\/j.specom.2022.09.002","volume":"144","author":"C Fu","year":"2022","unstructured":"Fu, C., Liu, C., Ishi, C.T., Ishiguro, H.: An improved cyclegan-based emotional voice conversion model by augmenting temporal dependency with a transformer. Speech Commun. 144, 110\u2013121 (2022)","journal-title":"Speech Commun."},{"key":"25_CR3","doi-asserted-by":"publisher","unstructured":"Gao, J., Chakraborty, D., Tembine, H., Olaleye, O.: Nonparallel emotional speech conversion. In: Proceedings of the Interspeech 2019, pp. 2858\u20132862 (2019). https:\/\/doi.org\/10.21437\/Interspeech.2019-2878","DOI":"10.21437\/Interspeech.2019-2878"},{"key":"25_CR4","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"25_CR5","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"25_CR6","doi-asserted-by":"publisher","unstructured":"Jiao, W., Yang, H., King, I., et al.: HiGRU: hierarchical gated recurrent units for utterance-level emotion recognition. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 397\u2013406 (2019). https:\/\/doi.org\/10.18653\/v1\/N19-1037","DOI":"10.18653\/v1\/N19-1037"},{"key":"25_CR7","doi-asserted-by":"publisher","unstructured":"Jung, J.w., Heo, H.S., Tak, H., et\u00a0al.: Aasist: audio anti-spoofing using integrated spectro-temporal graph attention networks. In: ICASSP 2022, pp. 6367\u20136371 (2022). https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9747766","DOI":"10.1109\/ICASSP43922.2022.9747766"},{"key":"25_CR8","doi-asserted-by":"publisher","unstructured":"Kinnunen, T., Sahidullah, M., Delgado, H., et\u00a0al.: The ASVspoof 2017 Challenge: Assessing the Limits of Replay Spoofing Attack Detection. In: Proceedings of the Interspeech 2017, pp.\u00a02\u20136 (2017). https:\/\/doi.org\/10.21437\/Interspeech.2017-1111","DOI":"10.21437\/Interspeech.2017-1111"},{"key":"25_CR9","doi-asserted-by":"crossref","unstructured":"M\u00fcller, N.M., et al.: Mlaad: the multi-language audio anti-spoofing dataset. arXiv preprint arXiv:2401.09512 (2024)","DOI":"10.1109\/IJCNN60899.2024.10650962"},{"key":"25_CR10","doi-asserted-by":"publisher","unstructured":"Rizos, G., Baird, A., et\u00a0al.: Stargan for emotional speech conversion: validated by data augmentation of end-to-end emotion recognition. In: ICASSP 2020, pp. 3502\u20133506 (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9054579","DOI":"10.1109\/ICASSP40776.2020.9054579"},{"key":"25_CR11","doi-asserted-by":"publisher","unstructured":"Sahidullah, M., Kinnunen, T., Hanil\u00e7i, C.: A comparison of features for synthetic speech detection. In: Proceedings of the Interspeech 2015, pp. 2087\u20132091 (2015). https:\/\/doi.org\/10.21437\/Interspeech.2015-472","DOI":"10.21437\/Interspeech.2015-472"},{"key":"25_CR12","doi-asserted-by":"publisher","unstructured":"Tak, H., Patino, J., Todisco, M., et\u00a0al.: End-to-end anti-spoofing with rawnet2. In: ICASSP 2021, pp. 6369\u20136373 (2021). https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9414234","DOI":"10.1109\/ICASSP39728.2021.9414234"},{"key":"25_CR13","doi-asserted-by":"publisher","unstructured":"Todisco, M., Wang, X., Vestman, V., et\u00a0al.: ASVspoof 2019: future horizons in spoofed and fake audio detection. In: Proceedings of the Interspeech 2019, pp. 1008\u20131012 (2019). https:\/\/doi.org\/10.21437\/Interspeech.2019-2249","DOI":"10.21437\/Interspeech.2019-2249"},{"key":"25_CR14","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et\u00a0al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol.\u00a030. Curran Associates, Inc. (2017)"},{"key":"25_CR15","doi-asserted-by":"publisher","unstructured":"Wang, X., Yamagishi, J.: A comparative study on recent neural spoofing countermeasures for synthetic speech detection. In: Proceedings of the Interspeech 2021, pp. 4259\u20134263 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-702","DOI":"10.21437\/Interspeech.2021-702"},{"key":"25_CR16","doi-asserted-by":"publisher","unstructured":"Wu, Z., Khodabakhsh, A., Demiroglu, C., et\u00a0al.: Sas: a speaker verification spoofing database containing diverse attacks. In: ICASSP 2015, pp. 4440\u20134444 (2015). https:\/\/doi.org\/10.1109\/ICASSP.2015.7178810","DOI":"10.1109\/ICASSP.2015.7178810"},{"key":"25_CR17","doi-asserted-by":"publisher","unstructured":"Wu, Z., Kinnunen, T., Evans, N., et\u00a0al.: ASVspoof 2015: the first automatic speaker verification spoofing and countermeasures challenge. In: Proceedings of the Interspeech 2015, pp. 2037\u20132041 (2015). https:\/\/doi.org\/10.21437\/Interspeech.2015-462","DOI":"10.21437\/Interspeech.2015-462"},{"key":"25_CR18","doi-asserted-by":"crossref","unstructured":"Xu, Y., Xu, H., Zou, J.: Hgfm: a hierarchical grained and feature model for acoustic emotion recognition. In: ICASSP 2020, pp. 6499\u20136503. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053039"},{"key":"25_CR19","doi-asserted-by":"publisher","unstructured":"Yamagishi, J., Wang, X., Todisco, M., et\u00a0al.: ASVspoof 2021: accelerating progress in spoofed and deepfake speech detection. In: Proceedings of the 2021 Edition of the Automatic Speaker Verification and Spoofing Countermeasures Challenge, pp. 47\u201354 (2021). https:\/\/doi.org\/10.21437\/ASVSPOOF.2021-8","DOI":"10.21437\/ASVSPOOF.2021-8"},{"key":"25_CR20","doi-asserted-by":"publisher","unstructured":"Yi, J., Bai, Y., Tao, J., Ma, H., et\u00a0al.: Half-truth: a partially fake audio detection dataset. In: Procedings of the Interspeech 2021, pp. 1654\u20131658 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-930","DOI":"10.21437\/Interspeech.2021-930"},{"key":"25_CR21","doi-asserted-by":"publisher","unstructured":"Yi, J., Fu, R., Tao, J., et\u00a0al.: Add 2022: the first audio deep synthesis detection challenge. In: ICASSP 2022, pp. 9216\u20139220 (2022). https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9746939","DOI":"10.1109\/ICASSP43922.2022.9746939"},{"key":"25_CR22","unstructured":"Yi, J., Tao, J., Fu, R., et\u00a0al.: Add 2023: the second audio deepfake detection challenge. arXiv preprint arXiv:2305.13774 (2023)"},{"key":"25_CR23","doi-asserted-by":"crossref","unstructured":"Yi, J., et al.: Scenefake: an initial dataset and benchmarks for scene fake audio detection. Pattern Recogn. 152, 110468 (2024)","DOI":"10.1016\/j.patcog.2024.110468"},{"key":"25_CR24","unstructured":"Yi, J., et al.: Audio deepfake detection: a survey. arXiv preprint arXiv:2308.14970 (2023)"},{"key":"25_CR25","doi-asserted-by":"publisher","unstructured":"Zang, Y., Zhang, Y., Heydari, M., Duan, Z.: Singfake: singing voice deepfake detection. In: 2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (ICASSP 2024), pp. 12156\u201312160 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10448184","DOI":"10.1109\/ICASSP48485.2024.10448184"},{"key":"25_CR26","doi-asserted-by":"publisher","unstructured":"Zhou, K., Sisman, B., Li, H.: Transforming spectrum and prosody for emotional voice conversion with non-parallel training data. In: Proceedings of the Speaker and Language Recognition Workshop (Odyssey 2020), pp. 230\u2013237 (2020). https:\/\/doi.org\/10.21437\/Odyssey.2020-33","DOI":"10.21437\/Odyssey.2020-33"},{"key":"25_CR27","doi-asserted-by":"publisher","unstructured":"Zhou, K., Sisman, B., Li, H.: Limited data emotional voice conversion leveraging text-to-speech: two-stage sequence-to-sequence training. In: Proceedings of the Interspeech 2021, pp. 811\u2013815 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-781","DOI":"10.21437\/Interspeech.2021-781"},{"key":"25_CR28","doi-asserted-by":"publisher","unstructured":"Zhou, K., Sisman, B., Liu, R., Li, H.: Seen and unseen emotional style transfer for voice conversion with a new emotional speech dataset. In: ICASSP 2021, pp. 920\u2013924 (2021). https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9413391","DOI":"10.1109\/ICASSP39728.2021.9413391"},{"key":"25_CR29","doi-asserted-by":"publisher","unstructured":"Zhou, K., Sisman, B., Zhang, M., Li, H.: Converting anyone\u2019s emotion: towards speaker-independent emotional voice conversion. In: Proceedings of the Interspeech 2020, pp. 3416\u20133420 (2020). https:\/\/doi.org\/10.21437\/Interspeech.2020-2014","DOI":"10.21437\/Interspeech.2020-2014"},{"key":"25_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.specom.2021.11.006","volume":"137","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Sisman, B., Liu, R., Li, H.: Emotional voice conversion: theory, databases and ESD. Speech Commun. 137, 1\u201318 (2022)","journal-title":"Speech Commun."}],"container-title":["Lecture Notes in Computer Science","Chinese Computational Linguistics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8367-0_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T12:08:41Z","timestamp":1732795721000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8367-0_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,29]]},"ISBN":["9789819783663","9789819783670"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8367-0_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,29]]},"assertion":[{"value":"29 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CCL","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China National Conference on Chinese Computational Linguistics","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taiyuan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 July 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 July 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cncl2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/cips-cl.org\/static\/CCL2024\/en\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}