{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T16:03:49Z","timestamp":1772813029907,"version":"3.50.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2020,4,15]],"date-time":"2020-04-15T00:00:00Z","timestamp":1586908800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,4,15]],"date-time":"2020-04-15T00:00:00Z","timestamp":1586908800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2020,8]]},"DOI":"10.1007\/s11265-020-01532-3","type":"journal-article","created":{"date-parts":[[2020,4,15]],"date-time":"2020-04-15T10:02:57Z","timestamp":1586944977000},"page":"777-791","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Combination of Amplitude and Frequency Modulation Features for Presentation Attack Detection"],"prefix":"10.1007","volume":"92","author":[{"given":"Madhu R.","family":"Kamble","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hemant A.","family":"Patil","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,4,15]]},"reference":[{"issue":"11","key":"1532_CR1","doi-asserted-by":"publisher","first-page":"1039","DOI":"10.1016\/j.specom.2009.04.004","volume":"51","author":"H Zen","year":"2009","unstructured":"Zen, H., Tokuda, K., & Black, A.W. (2009). Statistical parametric speech synthesis. Speech Communication, 51(11), 1039\u20131064.","journal-title":"Speech Communication"},{"key":"1532_CR2","unstructured":"Stylianou, Y., & transformation, V. (2009). A survey. In IEEE international conference on acoustics, speech and signal processing, (ICASSP), Taipei, Taiwan, China (pp. 3585\u20133588)."},{"key":"1532_CR3","unstructured":"Alegre, F.R., Janicki, A., & Evans, N. (2014). Re-assessing the threat of replay spoofing attacks against automatic speaker verification. In IEEE international conference of the biometrics special interest group (BIOSIG), Darmstadt, Germany (pp. 1\u20136)."},{"key":"1532_CR4","doi-asserted-by":"crossref","unstructured":"Kinnunen, T., Sahidullah, M.D., Falcone, M., Costantini, L., Hautamaki, R.G., Thomsen, D.A.L., Sarkar, A.K., Tan, Z.H., Delgado, H., Todisco, M., & et al. (2017). Reddots replayed: a new replay spoofing attack corpus for text-dependent speaker verification research. In IEEE international conference on acoustics, speech and signal processing (ICASSP), New Orleans, Louisiana, USA (pp. 5395\u20135399).","DOI":"10.1109\/ICASSP.2017.7953187"},{"key":"1532_CR5","doi-asserted-by":"crossref","unstructured":"Nagarsheth, P., Khoury, E., Patil, K., & Garland, M. (2017). Replay attack detection using DNN for channel discrimination. In INTERSPEECH, Stockholm, Sweden (pp. 97\u2013101).","DOI":"10.21437\/Interspeech.2017-1377"},{"key":"1532_CR6","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1016\/j.specom.2014.10.005","volume":"66","author":"Z Wu","year":"2015","unstructured":"Wu, Z., Evans, N., Kinnunen, T., Yamagishi, J., Alegre, F., & Li, H. (2015). Spoofing and countermeasures for speaker verification: a survey. Speech Communication, 66, 130\u2013153.","journal-title":"Speech Communication"},{"key":"1532_CR7","unstructured":"Madhu, R., Sailor, H.B., Patil, H.A., & Li, H. (2019). Advances in anti-spoofing: from the perspective of asvspoof challenges. In APSIPA transactions on signal and information processing (in press)."},{"key":"1532_CR8","doi-asserted-by":"crossref","unstructured":"Paul, A., Das, R.K., Sinha, R., & Prasanna, S.R.M. (2016). Countermeasure to handle replay attacks in practical speaker verification systems. In IEEE international conference on signal processing and communications (SPCOM), Bengaluru, India (pp. 1\u20135).","DOI":"10.1109\/SPCOM.2016.7746646"},{"key":"1532_CR9","doi-asserted-by":"crossref","unstructured":"Korshunov, P., Marcel, S., Muckenhirn, H., Gon\u00e7alves, A.R. , Mello, A.G.S., Violato, R.P.V., Sim\u00f5es, F.O., Neto, M.U., de Assis Angeloni, M., Stuchi, J.A., & et al. (2016). Overview of BTAS 2016 speaker anti-spoofing competition. In IEEE international conference on biometrics theory, applications and systems (BTAS), Niagara Falls, New York, USA (pp. 1\u20136).","DOI":"10.1109\/BTAS.2016.7791200"},{"key":"1532_CR10","doi-asserted-by":"crossref","unstructured":"Wu, Z., Gao, S., Cling, E.S., & Li, H. (2014). A study on replay attack and anti-spoofing for text-dependent speaker verification. In IEEE Asia-Pacific signal and information processing association, annual summit and conference (APSIPA), Chiang Mai, Thailand (pp. 1\u20135).","DOI":"10.1109\/APSIPA.2014.7041636"},{"key":"1532_CR11","doi-asserted-by":"crossref","unstructured":"Kinnunen, T., Sahidullah, M.D., Delgado, H., Todisco, M., Evans, N., Yamagishi, J., & Lee, K.A. (2017). The ASVspoof 2017 challenge: assessing the limits of replay spoofing attack detection. In INTERSPEECH, Stockholm, Sweden (pp. 1\u20136).","DOI":"10.21437\/Interspeech.2017-1111"},{"key":"1532_CR12","doi-asserted-by":"crossref","unstructured":"Lee, K.A., Larcher, A., Wang, G., Kenny, P., Br\u00fcmmer, N., van Leeuwen, D.A., Aronowitz, H., Kockmann, M., Vaquero, C., Ma, B., & et al. (2015). The RedDots data collection for speaker recognition. In INTERSPEECH, Dresden, Germany (pp. 2996\u20133000).","DOI":"10.21437\/Interspeech.2015-95"},{"key":"1532_CR13","doi-asserted-by":"crossref","unstructured":"Font, R., Esp\u00edn, J.M., & Cano, M.J. (2017). Experimental analysis of features for replay attack detection results on the ASVspoof 2017 challenge. In INTERSPEECH, Stockholm, Sweden (pp. 7\u201311).","DOI":"10.21437\/Interspeech.2017-450"},{"key":"1532_CR14","doi-asserted-by":"crossref","unstructured":"Patil, H.A., Kamble, M.R., Patel, T.B., & Soni, M. (2017). Novel variable length Teager energy separation based instantaneous frequency features for replay detection. In INTERSPEECH, Stockholm, Sweden (pp. 12\u201316).","DOI":"10.21437\/Interspeech.2017-1362"},{"key":"1532_CR15","doi-asserted-by":"crossref","unstructured":"Jelil, S., Das, R.K., Prasanna, S.R.M., & Sinha, R. (2017). Spoof detection using source, instantaneous frequency and cepstral features. In INTERSPEECH, Stockholm, Sweden (pp. 22\u201326).","DOI":"10.21437\/Interspeech.2017-930"},{"key":"1532_CR16","doi-asserted-by":"crossref","unstructured":"Alluri, K.N.R.K.R., Achanta, S., Kadiri, S.R., Gangashetty, S.V., & Vuppala, A.K. (2017). SFF anti-spoofer: IIIT-H submission for automatic speaker verification spoofing and countermeasures challenge 2017. In INTERSPEECH, Stockholm, Sweden (pp. 107\u2013111).","DOI":"10.21437\/Interspeech.2017-676"},{"key":"1532_CR17","doi-asserted-by":"crossref","unstructured":"Witkowski, M., Kacprzak, S., Zelasko, P., Kowalczyk, K., & Ga\u0142ka, J. (2017). Audio replay attack detection using high-frequency features. In INTERSPEECH, Stockholm, Sweden (pp. 27\u201331).","DOI":"10.21437\/Interspeech.2017-776"},{"key":"1532_CR18","doi-asserted-by":"crossref","unstructured":"Lavrentyeva, G., Novoselov, S., Malykh, E., Kozlov, A., Kudashev, O., & Shchemelinin, V. (2017). Audio replay attack detection with deep learning frameworks. In INTERSPEECH, Stockholm, Sweden (pp. 82\u201386).","DOI":"10.21437\/Interspeech.2017-360"},{"key":"1532_CR19","doi-asserted-by":"crossref","unstructured":"Cai, W., Cai, D., Liu, W., Li, G., & Li, M. (2017). Countermeasures for automatic speaker verification replay spoofing attack: on data augmentation, feature representation, classification and fusion. In INTERSPEECH, Stockholm, Sweden (pp. 17\u201321).","DOI":"10.21437\/Interspeech.2017-906"},{"key":"1532_CR20","doi-asserted-by":"crossref","unstructured":"Chen, Z., Xie, Z., Zhang, W., & Xu, X. (2017). ResNet and model fusion for automatic spoofing detection. In INTERSPEECH 2017, Stockholm, Sweden (pp. 102\u2013106).","DOI":"10.21437\/Interspeech.2017-1085"},{"key":"1532_CR21","doi-asserted-by":"crossref","unstructured":"Kamble, M.R., & Patil, H.A. (2017). Novel energy separation based instantaneous frequency features for spoof speech detection. In IEEE European signal processing conference (EUSIPCO), Kos Island, Greece (pp. 106\u2013110).","DOI":"10.23919\/EUSIPCO.2017.8081178"},{"key":"1532_CR22","unstructured":"Kamble, M.R., & Patil, H.A. (2017). Effectiveness of Mel scale-based ESA-IFCC features for classification of natural vs. spoofed speech. In Shankar, B.U., & et al. (Eds.) PReMI, Lecture Notes in Computer Sciance (LNCS) (pp. 308\u2013316): Springer."},{"key":"1532_CR23","doi-asserted-by":"crossref","unstructured":"Kamble, M.R., Tak, H., & Patil, H.A. (2018). Effectiveness of speech demodulation-based features for replay detection. In INTERSPEECH, Hyderabad, India (pp. 641\u2013645).","DOI":"10.21437\/Interspeech.2018-1675"},{"key":"1532_CR24","doi-asserted-by":"crossref","unstructured":"Kamble, M.R., & Patil, H.A. (2018). Novel variable length energy separation algorithm using instantaneous amplitude features for replay detection. In INTERSPEECH, Hyderabad, India (pp. 646\u2013650).","DOI":"10.21437\/Interspeech.2018-1687"},{"key":"1532_CR25","doi-asserted-by":"crossref","unstructured":"Kamble, M.R., & Patil, H.A. (2018). Novel amplitude weighted frequency modulation features for replay spoof detection. ISCSLP, Taipei, Taiwan, pp. 185\u2013189.","DOI":"10.1109\/ISCSLP.2018.8706673"},{"key":"1532_CR26","doi-asserted-by":"crossref","unstructured":"Kamble, M.R., Tak, H., Maddala, S.K., & Patil, H.A. (2018). Novel demodulation-based features using classifier-level fusion of GMM and CNN for replay detection. In ISCSLP. Taipei, Taiwan (pp. 334\u2013338).","DOI":"10.1109\/ISCSLP.2018.8706648"},{"key":"1532_CR27","doi-asserted-by":"crossref","unstructured":"Kamble, M.R., & Patil, H.A. (2019). Analysis of reverberation via Teager energy features for replay spoof speech detection. In IEEE international conference on acoustics, speech and signal processing (ICASSP), Brighton, UK (pp. 2607\u20132611).","DOI":"10.1109\/ICASSP.2019.8683830"},{"issue":"8","key":"1532_CR28","doi-asserted-by":"publisher","first-page":"1348","DOI":"10.1109\/TASLP.2015.2430815","volume":"23","author":"D Dimitriadis","year":"2015","unstructured":"Dimitriadis, D., & Bocchieri, E. (2015). Use of micro-modulation features in large vocabulary continuous speech recognition tasks. IEEE\/ACM Transactions on Audio, Speech and Language Processing (TASLP), 23(8), 1348\u20131357.","journal-title":"IEEE\/ACM Transactions on Audio, Speech and Language Processing (TASLP)"},{"key":"1532_CR29","doi-asserted-by":"crossref","unstructured":"Maragos, P., Quatieri, T.F., & Kaiser, J.F. (1991). Speech nonlinearities, modulations, and energy operators. In IEEE international conference on acoustics, speech, and signal processing, (ICASSP), Toronto, Ontario, Canada (pp. 421\u2013424).","DOI":"10.1109\/ICASSP.1991.150366"},{"issue":"4","key":"1532_CR30","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","volume":"28","author":"S Davis","year":"1980","unstructured":"Davis, S., & Mermelstein, P. (1980). Comparison of parametric representations for monosyllabic word recognition in continuously spoken sentences. IEEE Transactions on Acoustics, Speech, and Signal Processing, 28(4), 357\u2013366.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"1532_CR31","doi-asserted-by":"crossref","unstructured":"Tak, H., & Patil, H.A. (2018). Novel linear frequency residual cepstral features for replay attack detection. In INTERSPEECH, Hyderabad, India (pp. 726\u2013730).","DOI":"10.21437\/Interspeech.2018-1702"},{"key":"1532_CR32","volume-title":"A wavelet tour of signal processing","author":"S Mallat","year":"1999","unstructured":"Mallat, S. (1999). A wavelet tour of signal processing, 2nd edn. New York: Academic Press.","edition":"2nd edn."},{"issue":"4","key":"1532_CR33","doi-asserted-by":"publisher","first-page":"1532","DOI":"10.1109\/78.212729","volume":"41","author":"P Maragos","year":"1993","unstructured":"Maragos, P., Kaiser, J.F., & Quatieri, T.F. (1993). On amplitude and frequency demodulation using energy operators. IEEE Transactions on Signal Processing, 41(4), 1532\u20131550.","journal-title":"IEEE Transactions on Signal Processing"},{"key":"1532_CR34","doi-asserted-by":"crossref","unstructured":"Maragos, P., Quatieri, T.F., & Kaiser, J.F. (1992). On separating amplitude from frequency modulations using energy operators. In International conference on acoustics, speech, and signal processing (ICASSP), San Francisco, California, USA, (Vol. 2 pp. 1\u20134).","DOI":"10.1109\/ICASSP.1992.226135"},{"key":"1532_CR35","volume-title":"Discrete-time speech signal processing: principles and practice","author":"TF Quatieri","year":"2006","unstructured":"Quatieri, T.F. (2006). Discrete-time speech signal processing: principles and practice, 1st edn. India: Pearson Education.","edition":"1st edn."},{"issue":"5","key":"1532_CR36","doi-asserted-by":"publisher","first-page":"2712","DOI":"10.1152\/jn.01256.2005","volume":"96","author":"H Luo","year":"2006","unstructured":"Luo, H., Wang, Y., Poeppel, D., & Simon, J.Z. (2006). Concurrent encoding of frequency and amplitude modulation in human auditory cortex. Meg evidence, Journal of Neurophysiology, 96(5), 2712\u20132723.","journal-title":"Meg evidence, Journal of Neurophysiology"},{"issue":"1","key":"1532_CR37","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1016\/0165-1684(94)90169-4","volume":"37","author":"A Potamianos","year":"1994","unstructured":"Potamianos, A., & Maragos, P. (1994). A comparison of the energy operator and the Hilbert transform approach to signal and speech demodulation. Signal Processing, 37(1), 95\u2013120.","journal-title":"Signal Processing"},{"key":"1532_CR38","doi-asserted-by":"crossref","unstructured":"Cohen, L., Assaleh, K., & Fineberg, A.D.A.M. (1992). Instantaneous bandwidth and formant bandwidth. In IEEE SP workshop on statistical signal and array processing (pp. 13\u201317).","DOI":"10.1109\/SSAP.1992.246837"},{"issue":"6","key":"1532_CR39","doi-asserted-by":"publisher","first-page":"3795","DOI":"10.1121\/1.414997","volume":"99","author":"A Potamianos","year":"1996","unstructured":"Potamianos, A., & Maragos, P. (1996). Speech formant frequency and bandwidth tracking using multiband energy demodulation. The Journal of the Acoustical Society of America (JASA), 99(6), 3795\u20133806.","journal-title":"The Journal of the Acoustical Society of America (JASA)"},{"key":"1532_CR40","volume-title":"Speech processing \u2013 a dynamic and optimization-oriented approach","author":"D Li","year":"2003","unstructured":"Li, D., & O\u2019Shaughnessy, D. (2003). Speech processing \u2013 a dynamic and optimization-oriented approach, 1st edn. New York: Marcel Dekker Inc.","edition":"1st edn."},{"key":"1532_CR41","doi-asserted-by":"crossref","unstructured":"Kaiser, J.F. (1990). On a simple algorithm to calculate the energy of a signal. In International conference on acoustics, speech, and signal processing (ICASSP), Albuquerque, New Mexico, USA (pp. 381\u2013384).","DOI":"10.1109\/ICASSP.1990.115702"},{"key":"1532_CR42","volume-title":"Discrete cosine transform: algorithms, advantages, Applications","author":"K Ramamohan Rao","year":"2014","unstructured":"Ramamohan Rao, K., & Yip, P. (2014). Discrete cosine transform: algorithms, advantages, Applications. New York: Academic Press."},{"key":"1532_CR43","unstructured":"Kamble, MR, Maddala, S.K., Tak, H., & Patil, H.A. (2019). Comparison of frame and utterance-level classifers for replay attack detection, accepted in Asia-Pacific signal and information processing association, annual summit and conference (APSIPA-ASC), Lanzhou, China."},{"key":"1532_CR44","unstructured":"Delgado, H., Todisco, M., Md, S., Evans, N., Kinnunen, T., Lee, K.A., & Yamagishi, J. (2018). ASVspoof 2017 version 2.0: Meta-data analysis and baseline enhancements. In Odyssey the speaker and language recognition workshop, Les Sables d\u2019Olonne, France (pp. 296\u2013303)."},{"key":"1532_CR45","unstructured":"Objective Control for Talker Verification (OCTAVE), https:\/\/www.octave-project.eu\/, Last Accessed 19 Jan 2019."},{"key":"1532_CR46","doi-asserted-by":"publisher","first-page":"516","DOI":"10.1016\/j.csl.2017.01.001","volume":"45","author":"M Todisco","year":"2017","unstructured":"Todisco, M., Delgado, H., & Evans, N. (2017). Constant Q cepstral coefficients: a spoofing countermeasure for automatic speaker verification. Computer Speech & Language, Elsevier, 45, 516\u2013535.","journal-title":"Computer Speech & Language, Elsevier"},{"issue":"4","key":"1532_CR47","doi-asserted-by":"publisher","first-page":"841","DOI":"10.1109\/JSTSP.2019.2923372","volume":"13","author":"I Rodomagoulakis","year":"2019","unstructured":"Rodomagoulakis, I., & Maragos, P. (2019). Improved frequency modulation features for multichannel distant speech recognition. IEEE Journal of Selected Topics in Signal Processing, 13(4), 841\u2013849.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-020-01532-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11265-020-01532-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-020-01532-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,21]],"date-time":"2022-10-21T10:05:44Z","timestamp":1666346744000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11265-020-01532-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,4,15]]},"references-count":47,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2020,8]]}},"alternative-id":["1532"],"URL":"https:\/\/doi.org\/10.1007\/s11265-020-01532-3","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"value":"1939-8018","type":"print"},{"value":"1939-8115","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,4,15]]},"assertion":[{"value":"19 February 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 March 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 March 2020","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 April 2020","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}