{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T21:35:15Z","timestamp":1776116115938,"version":"3.50.1"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,1,16]],"date-time":"2023-01-16T00:00:00Z","timestamp":1673827200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,16]],"date-time":"2023-01-16T00:00:00Z","timestamp":1673827200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Natural Science Fund Project of China","award":["61301295"],"award-info":[{"award-number":["61301295"]}]},{"name":"Anhui Natural Science Fund Project","award":["1908085MF209"],"award-info":[{"award-number":["1908085MF209"]}]},{"name":"Anhui University Natural Science Research Project","award":["KJ2018A0018"],"award-info":[{"award-number":["KJ2018A0018"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cogn Comput"],"published-print":{"date-parts":[[2023,3]]},"DOI":"10.1007\/s12559-023-10108-9","type":"journal-article","created":{"date-parts":[[2023,1,16]],"date-time":"2023-01-16T13:03:25Z","timestamp":1673874205000},"page":"778-792","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["A Novel Attention-Guided Generative Adversarial Network for Whisper-to-Normal Speech Conversion"],"prefix":"10.1007","volume":"15","author":[{"given":"Teng","family":"Gao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qing","family":"Pan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6509-5520","authenticated-orcid":false,"given":"Jian","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huabin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hon Keung","family":"Kwan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,1,16]]},"reference":[{"key":"10108_CR1","doi-asserted-by":"publisher","first-page":"186","DOI":"10.1109\/LSP.2019.2961213","volume":"27","author":"M Cotescu","year":"2020","unstructured":"Cotescu M, Drugman T, Huybrechts G, Lorenzo-Trueba J, Moinet A. Voice conversion for whispered speech synthesis. IEEE Signal Process Lett. 2020;27:186\u201390.","journal-title":"IEEE Signal Process Lett."},{"key":"10108_CR2","doi-asserted-by":"publisher","first-page":"962242","DOI":"10.3389\/fpsyg.2022.962242","volume":"13","author":"M Xu","year":"2022","unstructured":"Xu M, Shao J, Ding H, Wang L. The effect of aging on identification of Mandarin consonants in normal and whisper registers. Front Psychol. 2022;13:962242.","journal-title":"Front Psychol."},{"issue":"2","key":"10108_CR3","doi-asserted-by":"publisher","first-page":"554","DOI":"10.1097\/AUD.0000000000001114","volume":"43","author":"K Hendrickson","year":"2022","unstructured":"Hendrickson K, Ernest D. The recognition of whispered speech in real-time. Ear Hear. 2022;43(2):554\u201362.","journal-title":"Ear Hear."},{"issue":"5","key":"10108_CR4","doi-asserted-by":"publisher","first-page":"1109","DOI":"10.1016\/j.otc.2007.05.012","volume":"40","author":"AD Rubin","year":"2007","unstructured":"Rubin AD, Sataloff RT. Vocal fold paresis and paralysis. Otolaryngol Clin North Am. 2007;40(5):1109\u201331.","journal-title":"Otolaryngol Clin North Am."},{"issue":"3","key":"10108_CR5","doi-asserted-by":"publisher","first-page":"158","DOI":"10.1007\/s40136-013-0019-4","volume":"1","author":"L Sulica","year":"2013","unstructured":"Sulica L. Vocal fold paresis: an evolving clinical concept. Curr Otorhinolaryngol Rep. 2013;1(3):158\u201362.","journal-title":"Curr Otorhinolaryngol Rep."},{"issue":"5","key":"10108_CR6","doi-asserted-by":"publisher","first-page":"1678","DOI":"10.1121\/1.398598","volume":"86","author":"VC Tartter","year":"1989","unstructured":"Tartter VC. What is in a whisper. J Acoust Soc Am. 1989;86(5):1678\u201383.","journal-title":"J Acoust Soc Am."},{"key":"10108_CR7","doi-asserted-by":"crossref","unstructured":"Wallis L, Jackson-Menaldi C, Holland W, Giraldo A. Vocal fold nodule vs. vocal fold polyp: Answer from surgical pathologist and voice pathologist point of view. J Voice.\u00a02004;18(1):125\u20139.","DOI":"10.1016\/j.jvoice.2003.07.003"},{"issue":"4","key":"10108_CR8","doi-asserted-by":"publisher","first-page":"489","DOI":"10.1016\/S0892-1997(98)80058-1","volume":"12","author":"JA Mattiske","year":"1998","unstructured":"Mattiske JA, Oates JM, Greenwood KM. Vocal problems among teachers: a review of prevalence, causes, prevention, and treatment. J Voice. 1998;12(4):489\u201399.","journal-title":"J Voice."},{"key":"10108_CR9","doi-asserted-by":"crossref","unstructured":"Itoh T, Takeda K, Itakura F. Acoustic analysis and recognition of whispered speech. In: Proceedings of IEEE Workshop on Automatic Speech Recognition and Understanding; 2001. p. 429\u201332.","DOI":"10.1109\/ICASSP.2002.1005758"},{"key":"10108_CR10","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1515\/9781501502415-002","volume":"5","author":"C Zhang","year":"2018","unstructured":"Zhang C, Hansen JHL. Advancements in whispered speech detection for interactive\/speech systems. Signal Acoust Model Speech Commun Disorders. 2018;5:9\u201332.","journal-title":"Signal Acoust Model Speech Commun Disorders."},{"key":"10108_CR11","doi-asserted-by":"crossref","unstructured":"Jin Q, Jou SS, Schultz T. Whispering speaker identification. In: Proceedings of IEEE International Conference on Multimedia and Expo; 2007. p. 1027\u201330.","DOI":"10.1109\/ICME.2007.4284828"},{"key":"10108_CR12","doi-asserted-by":"crossref","unstructured":"Fan X, Hansen JHL. Speaker identification for whispered speech based on frequency warping and score competition. In: Proceedings of INTERSPEECH; 2008. p. 1313\u20136.","DOI":"10.21437\/Interspeech.2008-384"},{"key":"10108_CR13","doi-asserted-by":"crossref","unstructured":"Fan X, Hansen JHL. Speaker Identification with whispered speech based on modified LFCC parameters and feature mapping. In: Proceedings of IEEE Conference on Acoustics, Speech, and Signal Processing; 2009. p. 4553\u20136.","DOI":"10.1109\/ICASSP.2009.4960643"},{"key":"10108_CR14","doi-asserted-by":"crossref","unstructured":"Fan X, Hansen JHL. Acoustic analysis for speaker identification of whispered speech. In: Proceedings of Conference on Acoustics, Speech, and Signal Processing; 2010. p. 5046\u20139.","DOI":"10.1109\/ICASSP.2010.5495059"},{"key":"10108_CR15","doi-asserted-by":"crossref","unstructured":"Fan X, Hansen JHL. Speaker Identification for whispered speech using modified temporal patterns and MFCCs. In: Proceedings of INTERSPEECH; 2009. p. 912\u20135.","DOI":"10.21437\/Interspeech.2009-270"},{"key":"10108_CR16","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1016\/j.specom.2003.10.005","volume":"45","author":"T Ito","year":"2005","unstructured":"Ito T, Takeda K, Itakura F. Analysis and recognition of whispered speech. Speech Commun. 2005;45:139\u201352.","journal-title":"Speech Commun."},{"key":"10108_CR17","doi-asserted-by":"crossref","unstructured":"Tajiri Y, Tanaka K, Toda T, Neubig G, Sakti S, Nakamura S. Non-audible murmur enhancement based on statistical conversion using air-and body-conductive microphones in noisy environments. In: Proceedings of INTERSPEECH; 2015. p. 2769\u201373.","DOI":"10.21437\/Interspeech.2015-583"},{"key":"10108_CR18","doi-asserted-by":"crossref","unstructured":"Ahmadi F, McLoughlin IV, Sharifzadeh HR. Analysis-by-synthesis method for whisper-speech reconstruction. In: 2008 IEEE Asia Pacific Conference on Circuits and Systems; 2008. p. 1280\u20133.","DOI":"10.1109\/APCCAS.2008.4746261"},{"issue":"10","key":"10108_CR19","doi-asserted-by":"publisher","first-page":"2448","DOI":"10.1109\/TBME.2010.2053369","volume":"57","author":"HR Sharifzadeh","year":"2010","unstructured":"Sharifzadeh HR, McLoughlin IV, Ahmadi F. Reconstruction of normal sounding speech for laryngectomy patients through a modified CELP codec. IEEE Trans Biomed Eng. 2010;57(10):2448\u201358.","journal-title":"IEEE Trans Biomed Eng."},{"issue":"24","key":"10108_CR20","doi-asserted-by":"publisher","first-page":"1781","DOI":"10.1049\/el.2014.1645","volume":"50","author":"JJ Li","year":"2014","unstructured":"Li JJ, Mcloughlin IV, Dai LR, Ling ZH. Whisper-to-speech conversion using restricted Boltzmann machine arrays. Electron Lett. 2014;50(24):1781\u20132.","journal-title":"Electron Lett."},{"key":"10108_CR21","doi-asserted-by":"crossref","unstructured":"Janke M, Wand M, Heistermann T, Schultz T, Prahallad K. Fundamental frequency generation for whisper-to-audible speech conversion. In: International Conference on Acoustics, Speech and Signal Processing; 2014. p. 2579\u201383.","DOI":"10.1109\/ICASSP.2014.6854066"},{"key":"10108_CR22","doi-asserted-by":"crossref","unstructured":"Meenakshi GN, Ghosh PK. Whispered speech to neutral speech conversion using bidirectional LSTMs. In: Interspeech. 2018.","DOI":"10.21437\/Interspeech.2018-1487"},{"key":"10108_CR23","doi-asserted-by":"crossref","unstructured":"Heeren, Willemijn FL. Vocalic correlates of pitch in whispered versus normal speech. J Acoust Soc Am. 2015;138(6):3800\u201310.","DOI":"10.1121\/1.4937762"},{"issue":"1","key":"10108_CR24","doi-asserted-by":"publisher","first-page":"395","DOI":"10.1121\/1.4939962","volume":"139","author":"J Clarke","year":"2016","unstructured":"Clarke J, Baskent D, Gaudrain E. Pitch and spectral resolution: a systematic comparison of bottom-up cues for top-down repair of degraded speech. J Acoust Soc Am. 2016;139(1):395\u2013405.","journal-title":"J Acoust Soc Am."},{"issue":"8","key":"10108_CR25","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J. Long short-term memory. Neural Comput. 1997;9(8):1735\u201380.","journal-title":"Neural Comput."},{"key":"10108_CR26","doi-asserted-by":"crossref","unstructured":"Janke M, Wand M, Heistermann T, Schultz T, Prahallad K. Fundamental frequency generation for whisper-to-audible speech conversion. In: Proceedings of IEEE Conference on Acoustics, Speech, and Signal Processing; 2014. p. 2579\u201383.","DOI":"10.1109\/ICASSP.2014.6854066"},{"key":"10108_CR27","doi-asserted-by":"publisher","first-page":"778","DOI":"10.1016\/j.neucom.2022.06.062","volume":"501","author":"J Ji","year":"2022","unstructured":"Ji J, Wang M, Zhang X, Lei M. Relation constraint self-attention for image captioning. Neurocomputing. 2022;501:778\u201389.","journal-title":"Neurocomputing."},{"issue":"8","key":"10108_CR28","first-page":"1","volume":"14","author":"M Guo","year":"2022","unstructured":"Guo M, Liu Z, Mu T, Hu S. Beyond self-attention: External attention using two linear layers for visual tasks. IEEE Trans Pattern Anal Mach Intell. 2022;14(8):1\u201313.","journal-title":"IEEE Trans Pattern Anal Mach Intell."},{"issue":"6","key":"10108_CR29","doi-asserted-by":"publisher","first-page":"1401","DOI":"10.1109\/TNN.2005.852235","volume":"16","author":"DL Wang","year":"2005","unstructured":"Wang DL. The time dimension for scene analysis. IEEE Trans Neural Netw. 2005;16(6):1401\u201326.","journal-title":"IEEE Trans Neural Netw."},{"key":"10108_CR30","doi-asserted-by":"crossref","unstructured":"Subakan C, Ravanelli M, Cornell S, Bronzi M, Zhong J. Attention is all you need in speech separation. In: IEEE International Conference on Acoustics, Speech and Signal Processing. 2021.","DOI":"10.1109\/ICASSP39728.2021.9413901"},{"key":"10108_CR31","doi-asserted-by":"crossref","unstructured":"Kaneko T, Kameoka H. CycleGAN-VC: Non-parallel voice conversion using cycle-consistent adversarial networks. In: Proceedings of 26th European Signal Processing Conference; 2018. p. 2100\u20134.","DOI":"10.23919\/EUSIPCO.2018.8553236"},{"key":"10108_CR32","first-page":"1","volume":"16","author":"D Auerbach Benjamin","year":"2022","unstructured":"Auerbach Benjamin D, Gritton Howard J. Hearing in complex environments: Auditory gain control, attention, and hearing loss. Front Neurosci. 2022;16:1\u201323.","journal-title":"Front Neurosci."},{"key":"10108_CR33","doi-asserted-by":"crossref","unstructured":"Thomassen S, Hartung K, Einh\u00e4ser W, Bendixen A. Low-high-low or high-low-high? Pattern effects on sequential auditory scene analysis. J Acoust Soc Am. 2022;152(5):2758\u201368.","DOI":"10.1121\/10.0015054"},{"key":"10108_CR34","unstructured":"Goodfellow IJ, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y. Generative adversarial nets. In: Proceedings of Conference on Neural Information Processing Systems; 2014. p. 2672\u201380."},{"key":"10108_CR35","doi-asserted-by":"crossref","unstructured":"Shah N, Shah NJ, Patil HA. Effectiveness of generative adversarial network for non-audible murmur-to-whisper speech conversion. In: Proceedings of INTERSPEECH; 2018. p. 3157\u201361.","DOI":"10.21437\/Interspeech.2018-1565"},{"key":"10108_CR36","doi-asserted-by":"crossref","unstructured":"Patel M, Parmar M, Doshi S, Shah N, Patil. Novel inception-GAN for whispered-to-normal speech conversion. In: Proceedings of 10th ISCA Speech Synthesis Workshop; 2019. p. 87\u201392.","DOI":"10.21437\/SSW.2019-16"},{"key":"10108_CR37","doi-asserted-by":"crossref","unstructured":"Purohit M, Patel M, Malaviya H, Patil A, Parmar M, Shah N, Doshi S, Patil HA. Intelligibility improvement of dysarthric speech using MMSE DiscoGAN. In: Proceedings of Conference on Signal Processing and Communications; 2020. p. 1\u20135.","DOI":"10.1109\/SPCOM50965.2020.9179511"},{"key":"10108_CR38","doi-asserted-by":"crossref","unstructured":"Parmar M, Doshi S, Shah NJ, Patel M, Patil HA. Effectiveness of cross-domain architectures for whisper-to-normal speech conversion. In: Proceedings of 27th European Signal Processing Conference; 2019. p. 1\u20135.","DOI":"10.23919\/EUSIPCO.2019.8902961"},{"key":"10108_CR39","unstructured":"Zhang H, Goodfellow I, Metaxas D, Odena A. Self-attention generative adversarial networks. In: Proceedings of International Conference on Machine Learning. 2019 ."},{"key":"10108_CR40","doi-asserted-by":"crossref","unstructured":"Amodio M, Krishnaswamy S. TraVeLGAN: Image-to-image translation by transformation vector learning. In: Proceedings of IEEE Conference on Computer Vision and Pattern Recognition; 2019. p. 8975\u201384.","DOI":"10.1109\/CVPR.2019.00919"},{"key":"10108_CR41","doi-asserted-by":"crossref","unstructured":"Melekhov I, Kannala J, Rahtu E. Siamese network features for image matching. In: Proceedings of 23rd Conference Pattern Recognition; 2016. p. 378\u201383.","DOI":"10.1109\/ICPR.2016.7899663"},{"key":"10108_CR42","doi-asserted-by":"crossref","unstructured":"Gao Y, Singh R, Raj B. Voice impersonation using generative adversarial networks. In: Proceedings of Conference on Acoustics, Speech, and Signal Processing; 2018. p. 2506\u201310.","DOI":"10.1109\/ICASSP.2018.8462018"},{"key":"10108_CR43","doi-asserted-by":"crossref","unstructured":"Zhu J, Park T, Isola P, Efros AA. Unpaired image-to-image translation using cycle-consistent adversarial networks. In: Proceedings of IEEE Conference on Computer Vision; 2017. p. 2242\u201351.","DOI":"10.1109\/ICCV.2017.244"},{"key":"10108_CR44","unstructured":"Taigman Y, Polyak A, Wolf L. Unsupervised cross-domain image generation. In: Proceedings of the International Conference on Learning Representations. 2016."},{"key":"10108_CR45","volume-title":"CSTR NAM TIMIT Plus, [dataset]","author":"J Yamagishi","year":"2021","unstructured":"Yamagishi J, Brown G, Yang CY, Clark R, King S. CSTR NAM TIMIT Plus, [dataset]. Centre for Speech Technology Research: University of Edinburgh; 2021."},{"issue":"9","key":"10108_CR46","doi-asserted-by":"publisher","first-page":"2505","DOI":"10.1109\/TASL.2012.2205241","volume":"20","author":"T Toda","year":"2012","unstructured":"Toda T, Nakagiri M, Shikano K. Statistical voice conversion techniques for body-conducted unvoiced speech enhancement. IEEE Trans Audio Speech Language Process. 2012;20(9):2505\u201317.","journal-title":"IEEE Trans Audio Speech Language Process."},{"key":"10108_CR47","doi-asserted-by":"crossref","unstructured":"Meenakshi GN, Ghosh PK. Whispered speech to neutral speech conversion using bidirectional LSTMs. In: Proceedings of INTERSPEECH; 2018. p. 491\u20135.","DOI":"10.21437\/Interspeech.2018-1487"},{"key":"10108_CR48","doi-asserted-by":"crossref","unstructured":"Griffin D, Lim J. Signal estimation from modified short-time Fourier transform. In: Proceedings of IEEE Conference on Acoustics, Speech, and Signal Processing; 1983. p. 804\u20137.","DOI":"10.1109\/ICASSP.1983.1172092"},{"key":"10108_CR49","doi-asserted-by":"crossref","unstructured":"Erro D, Sainz I, Navas E, Hernaez I. Improved HNM based vocoder for statistical synthesizers. In: INTERSPEECH, Florence, Italy; 2011. p.\u00a01809\u201312.","DOI":"10.21437\/Interspeech.2011-35"},{"key":"10108_CR50","doi-asserted-by":"crossref","unstructured":"Taal CH, Hendriks RC, Heusdens R, Jensen J. A short-time objective intelligibility measure for time-frequency weighted noisy speech. In: Proceedings of Conference on Acoustics, Speech, and Signal Processing; 2010. p. 4214\u20137.","DOI":"10.1109\/ICASSP.2010.5495701"},{"key":"10108_CR51","doi-asserted-by":"crossref","unstructured":"Rix AW, Beerends JG, Hollier M, Hekstra AP. Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs. In: Proceedings of IEEE International Conference on Acoustics, Speech, and Signal Processing. 2001. p.749\u2013752.","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"10108_CR52","doi-asserted-by":"crossref","unstructured":"Kubichek R. Mel-cepstral distance measure for objective speech quality assessment. In: Proceedings of IEEE Pacific Rim Conference on Communications Computers and Signal Processing; 1993. p. 125\u20138.","DOI":"10.1109\/PACRIM.1993.407206"},{"issue":"5","key":"10108_CR53","doi-asserted-by":"publisher","first-page":"380","DOI":"10.1109\/TASSP.1976.1162849","volume":"24","author":"A Gray","year":"1976","unstructured":"Gray A, Markel J. Distance measures for speech processing. IEEE Trans Acoust Speech Signal Process. 1976;24(5):380\u201391.","journal-title":"IEEE Trans Acoust Speech Signal Process."},{"key":"10108_CR54","doi-asserted-by":"crossref","unstructured":"Malfait L, Berger J, Kastner M. P.563 - The ITU-T standard for single-ended speech quality assessment. IEEE Trans Audio Speech Language Process. 2006;14(6):1924\u201334.","DOI":"10.1109\/TASL.2006.883177"}],"container-title":["Cognitive Computation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-023-10108-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s12559-023-10108-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-023-10108-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,12]],"date-time":"2024-10-12T09:03:42Z","timestamp":1728723822000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s12559-023-10108-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,1,16]]},"references-count":54,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,3]]}},"alternative-id":["10108"],"URL":"https:\/\/doi.org\/10.1007\/s12559-023-10108-9","relation":{},"ISSN":["1866-9956","1866-9964"],"issn-type":[{"value":"1866-9956","type":"print"},{"value":"1866-9964","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,1,16]]},"assertion":[{"value":"7 February 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 January 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 January 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This article does not contain any studies with human participants or animals performed by any of the authors.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"The authors declare no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}]}}