{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T07:34:06Z","timestamp":1777620846023,"version":"3.51.4"},"reference-count":57,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2018,12,12]],"date-time":"2018-12-12T00:00:00Z","timestamp":1544572800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2019,3]]},"DOI":"10.1007\/s10772-018-09579-1","type":"journal-article","created":{"date-parts":[[2018,12,12]],"date-time":"2018-12-12T13:47:06Z","timestamp":1544622426000},"page":"99-110","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Enhancement of esophageal speech obtained by a voice conversion technique using time dilated Fourier cepstra"],"prefix":"10.1007","volume":"22","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6345-5366","authenticated-orcid":false,"given":"Imen","family":"Ben Othmane","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joseph","family":"Di Martino","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ka\u00efs","family":"Ouni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,12,12]]},"reference":[{"issue":"2","key":"9579_CR2","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1250\/ast.11.71","volume":"11","author":"M Abe","year":"1990","unstructured":"Abe, M., et al. (1990). Voice conversion through vector quantization. Journal of the Acoustical Society of Japan (E), 11(2), 71\u201376.","journal-title":"Journal of the Acoustical Society of Japan (E)"},{"key":"9579_CR3","unstructured":"Arya, S. (1996). Nearest neighbor searching and applications. Univ. of Maryland at College Park, MD."},{"issue":"6","key":"9579_CR4","doi-asserted-by":"publisher","first-page":"1337","DOI":"10.1002\/j.1538-7305.1959.tb01591.x","volume":"38","author":"HL Barney","year":"1959","unstructured":"Barney, H. L., Haworth, F. E., & Dunn, H. K. (1959). An experimental transistorized artificial larynx. Bell Labs Technical Journal, 38(6), 1337\u20131356.","journal-title":"Bell Labs Technical Journal"},{"key":"9579_CR5","unstructured":"Beauregard, G. T., Zhu, X., & Wyse, L. (2005) An efficient algorithm for real-time spectrogram inversion. In Proceedings of the 8th international conference on digital audio effects."},{"issue":"2","key":"9579_CR6","doi-asserted-by":"publisher","first-page":"113120","DOI":"10.1109\/TASSP.1979.1163209","volume":"27","author":"SF Boll","year":"1979","unstructured":"Boll, S. F. (1979). Suppression of acoustic noise in speech using spectral subtraction. IEEE Transactions on Acoustics, Speech and Signal Processing, 27(2), 113120.","journal-title":"IEEE Transactions on Acoustics, Speech and Signal Processing"},{"issue":"3","key":"9579_CR7","doi-asserted-by":"publisher","first-page":"138","DOI":"10.1097\/00020840-200006000-00002","volume":"8","author":"K Chenausky","year":"2000","unstructured":"Chenausky, K., & MacAuslan, J. (2000). Utilization of microprocessors in voice quality improvement: The electrolarynx. Current Opinion in Otolaryngology & Head and Neck Surgery, 8(3), 138\u2013142.","journal-title":"Current Opinion in Otolaryngology & Head and Neck Surgery"},{"issue":"10","key":"9579_CR8","doi-asserted-by":"publisher","first-page":"1428","DOI":"10.1109\/PROC.1977.10747","volume":"65","author":"DG Childers","year":"1977","unstructured":"Childers, D. G., Skinner, D. P., & Kemerait, R. C. (1977). The cepstrum: A guide to processing. Proceedings of the IEEE, 65(10), 1428\u20131443.","journal-title":"Proceedings of the IEEE"},{"key":"9579_CR9","doi-asserted-by":"crossref","unstructured":"Cole, D., et al. (1997). Application of noise reduction techniques for alaryngeal speech enhancement. In TENCON\u201997. IEEE region 10 annual conference. Speech and image technologies for computing and telecommunications., Proceedings of IEEE (Vol. 2). IEEE.","DOI":"10.1109\/TENCON.1997.648252"},{"key":"9579_CR10","unstructured":"Del Pozo, A., & Young, S. (2006). Continuous tracheoesophageal speech repair. In Signal processing conference, 2006 14th European. IEEE."},{"key":"9579_CR11","doi-asserted-by":"crossref","unstructured":"Del Pozo, A., & Young, S. (2008). Repairing tracheoesophageal speech duration. In Proc Speech Prosody.","DOI":"10.21437\/SpeechProsody.2008-41"},{"key":"9579_CR13","doi-asserted-by":"crossref","unstructured":"Desai, S., et al. (2009). Voice conversion using artificial neural networks. In Acoustics, speech and signal processing, 2009. ICASSP 2009. IEEE international conference on. IEEE.","DOI":"10.1109\/ICASSP.2009.4960478"},{"issue":"5","key":"9579_CR12","doi-asserted-by":"publisher","first-page":"954","DOI":"10.1109\/TASL.2010.2047683","volume":"18","author":"S Desai","year":"2010","unstructured":"Desai, S., et al. (2010). Spectral mapping using artificial neural networks for voice conversion. IEEE Transactions on Audio, Speech, and Language Processing, 18(5), 954\u2013964.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9579_CR14","doi-asserted-by":"crossref","unstructured":"Deza, M. M., & Deza, E. (2009). Encyclopedia of distances. In Encyclopedia of distances (pp. 1\u2013583). Springer, Berlin.","DOI":"10.1007\/978-3-642-00234-2_1"},{"issue":"9","key":"9579_CR15","doi-asserted-by":"publisher","first-page":"2472","DOI":"10.1587\/transinf.E93.D.2472","volume":"93","author":"H Doi","year":"2010","unstructured":"Doi, H. (2010). Esophageal speech enhancement based on statistical voice conversion with Gaussian mixture models. IEICE Transaction on Information and Systems, 93(9), 2472\u20132482.","journal-title":"IEICE Transaction on Information and Systems"},{"key":"9579_CR16","doi-asserted-by":"crossref","unstructured":"Doi, H., Nakamura, K., Toda, T., Saruwatari, H., & Shikano, K. (May 2011). An evaluation of alaryngeal speech enhancement methods based on voice conversion techniques. In Proc. ICASSP (pp. 5136\u20135139).","DOI":"10.1109\/ICASSP.2011.5947513"},{"key":"9579_CR17","doi-asserted-by":"crossref","unstructured":"Doi, H., et al. (2010). Statistical approach to enhancing esophageal speech based on Gaussian mixture models. In Acoustics speech and signal processing (ICASSP), 2010 IEEE international conference on. IEEE.","DOI":"10.1109\/ICASSP.2010.5495676"},{"issue":"1","key":"9579_CR18","doi-asserted-by":"publisher","first-page":"63","DOI":"10.2466\/pms.1975.40.1.63","volume":"40","author":"MD Filter","year":"1975","unstructured":"Filter, M. D., & Hyman, M. (1975). Relationship of acoustic parameters and perceptual ratings of esophageal speech. Perceptual and Motor Skills, 40(1), 63\u201368.","journal-title":"Perceptual and Motor Skills"},{"key":"9579_CR19","unstructured":"Garc\u00eda, B., Vicente, J., & Aramendi, E. (2002). Time-spectral technique for esophageal speech regeneration. In 11th EUSIPCO (European Signal Processing Conference). IEEE, Toulouse, France."},{"key":"9579_CR20","doi-asserted-by":"crossref","unstructured":"Garc\u00eda, B., et al. (2005). Esophageal voices: Glottal flow restoration. In IEEE international conference on acoustics, speech, and signal processing, 2005. Proceedings (ICASSP\u201905) (Vol. 4). IEEE.","DOI":"10.1109\/ICASSP.2005.1415965"},{"issue":"2","key":"9579_CR21","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin, D., & Lim, J. (1984). Signal estimation from modified short-time Fourier transform. IEEE Transactions on Acoustics, Speech, and Signal Processing, 32(2), 236\u2013243.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"9579_CR22","unstructured":"Hisada, A., & Sawada, H. (2002). Real-time clarification of esophageal speech using a comb filter. In International conference on disability, virtual reality and associated technologies (pp. 39\u201346)."},{"issue":"1","key":"9579_CR23","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/0021-9924(69)90049-5","volume":"2","author":"HR Hoops","year":"1969","unstructured":"Hoops, H. R., & Noll, J. D. (1969). Relationship of selected acoustic variables to judgments of esophageal speech. Journal of Communication Disorders, 2(1), 1\u201313.","journal-title":"Journal of Communication Disorders"},{"key":"9579_CR1","unstructured":"http:\/\/www.cancerresearchuk.org\/about-cancer ."},{"key":"9579_CR26","unstructured":"Ben Othmane, I., Di Martino, J., & Ouni, K. (2017). Enhancement of esophageal speech using voice conversion techniques. In International conference on natural language, signal and speech processing-ICNLSSP."},{"key":"9579_CR25","doi-asserted-by":"crossref","unstructured":"Ben Othmane, I., Di Martino, J., & Ouni, K. (2018). Improving the computational performance of standard GMM-based voice conversion systems used in real-time applications. In ICECOCS18\u20141st international conference on electronics, control, optimization and computer science, Dec 2018, Kenitra, Morocco. IEEE.","DOI":"10.1109\/ICECOCS.2018.8610514"},{"issue":"1","key":"9579_CR24","first-page":"10","volume":"1","author":"I Ben Othmane","year":"2018","unstructured":"Ben Othmane, I., Di Martino, J., & Ouni, K. (2018). Enhancement of esophageal speech using statistical and neuromimetic voice conversion techniques. Journal of International Science and General Applications, 1(1), 10.","journal-title":"Journal of International Science and General Applications"},{"key":"9579_CR27","doi-asserted-by":"crossref","unstructured":"Kain, A., & Macon, M. W. (1998). Spectral voice conversion for text-to-speech synthesis. In Acoustics, speech and signal processing, 1998. Proceedings of the 1998 IEEE international conference on (Vol. 1). IEEE.","DOI":"10.1109\/ICASSP.1998.674423"},{"issue":"7","key":"9579_CR28","doi-asserted-by":"publisher","first-page":"881","DOI":"10.1109\/TPAMI.2002.1017616","volume":"24","author":"T Kanungo","year":"2002","unstructured":"Kanungo, T. (2002). An efficient k-means clustering algorithm: Analysis and implementation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 24(7), 881\u2013892.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"9579_CR29","doi-asserted-by":"crossref","unstructured":"Kawahara, H., et al. (1999). Fixed point analysis of frequency to instantaneous frequency mapping for accurate estimation of F0 and periodicity. In Sixth european conference on speech communication and technology.","DOI":"10.21437\/Eurospeech.1999-613"},{"key":"9579_CR30","doi-asserted-by":"crossref","unstructured":"Ling-HuiChen, Z.-H., & YanSong, L.-R. (2013). Joint spectral distribution modeling using restricted boltzmann machines for voice conversion.","DOI":"10.21437\/Interspeech.2013-666"},{"issue":"5","key":"9579_CR31","first-page":"865874","volume":"53","author":"H Liu","year":"2006","unstructured":"Liu, H., Zhao, Q., Wan, M. X., & Wang, S. P. (2006). Enhancement of electrolarynx speech based on auditory masking. IEEE Transactions on Biomedical Engineering, 53(5), 865874.","journal-title":"IEEE Transactions on Biomedical Engineering"},{"issue":"1","key":"9579_CR32","doi-asserted-by":"crossref","first-page":"56","DOI":"10.22201\/icat.16656423.2010.8.01.475","volume":"8","author":"A Mantilla-Caeiros","year":"2010","unstructured":"Mantilla-Caeiros, A., Nakano-Miyatake, M., & Perez-Meana, H. (2010). A pattern recognition based esophageal speech enhancement system. Journal of Applied Research and Technology, 8(1), 56\u201370.","journal-title":"Journal of Applied Research and Technology"},{"issue":"2","key":"9579_CR33","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1250\/ast.23.69","volume":"23","author":"K Matsui","year":"2002","unstructured":"Matsui, K., et al. (2002). Enhancement of esophageal speech using formant synthesis. Acoustical Science and Technology, 23(2), 69\u201376.","journal-title":"Acoustical Science and Technology"},{"key":"9579_CR34","unstructured":"Matui, K., Hara, N., Kobayashi, N., & Hirose, H. (May, 1999). Enhancement of esophageal speech using formant synthesis. In Proc. ICASSP (pp. 1831\u20131834), Phoenix, Arizona."},{"issue":"3","key":"9579_CR35","doi-asserted-by":"publisher","first-page":"952","DOI":"10.1109\/TSA.2005.857790","volume":"14","author":"A Mouchtaris","year":"2006","unstructured":"Mouchtaris, A., Van der Spiegel, J., & Mueller, P. (2006). Nonparallel training for voice conversion based on a parameter adaptation approach. IEEE Transactions on Audio, Speech, and Language Processing, 14(3), 952\u2013963.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9579_CR36","unstructured":"Nair, V., & Hinton, G. E. (2010). Rectified linear units improve restricted boltzmann machines. In Proceedings of the 27th international conference on machine learning (ICML-10)."},{"issue":"1","key":"9579_CR37","first-page":"134146","volume":"54","author":"K Nakamura","year":"2012","unstructured":"Nakamura, K., Toda, T., Saruwatari, H., & Shikano, K. (2012). Speaking-aid systems using GMM-based voice conversion for electrolaryngeal speech. SPECOM, 54(1), 134146.","journal-title":"SPECOM"},{"key":"9579_CR39","doi-asserted-by":"crossref","unstructured":"Nakashika, T., Takiguchi, T., & Ariki, Y. (2014). High-order sequence modeling using speaker-dependent recurrent temporal restricted Boltzmann machines for voice conversion. In Fifteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2014-447"},{"key":"9579_CR38","doi-asserted-by":"crossref","unstructured":"Nakashika, T., et al. (2013). Voice conversion in high-order eigen space using deep belief nets. In Interspeech.","DOI":"10.21437\/Interspeech.2013-102"},{"key":"9579_CR40","unstructured":"Nankaku, Y., et al. (2007). Spectral conversion based on statistical models including time-sequence matching."},{"issue":"2","key":"9579_CR41","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1016\/0167-6393(94)00058-I","volume":"16","author":"M Narendranath","year":"1995","unstructured":"Narendranath, M. (1995). Transformation of formants for voice conversion using artificial neural networks. Speech Communication, 16(2), 207\u2013216.","journal-title":"Speech Communication"},{"key":"9579_CR42","doi-asserted-by":"crossref","unstructured":"Park, S. H. (2011). Simple linear regression. In International Encyclopedia of Statistical Science (pp. 1327\u20131328). Springer, Berlin.","DOI":"10.1007\/978-3-642-04898-2_517"},{"issue":"5","key":"9579_CR43","doi-asserted-by":"publisher","first-page":"2461","DOI":"10.1121\/1.413279","volume":"98","author":"Y Qi","year":"1995","unstructured":"Qi, Y., Weinberg, B., & Bi, N. (1995). Enhancement of female esophageal and tracheoesophageal speech. The Journal of the Acoustical Society of America, 98(5), 2461\u20132465.","journal-title":"The Journal of the Acoustical Society of America"},{"issue":"2","key":"9579_CR44","doi-asserted-by":"publisher","first-page":"202","DOI":"10.1044\/jshd.4902.202","volume":"49","author":"J Robbins","year":"1984","unstructured":"Robbins, J., et al. (1984). A comparative acoustic study of normal, esophageal, and tracheoesophageal speech production. The Journal of Speech and Hearing Disorders, 49(2), 202\u2013210.","journal-title":"The Journal of Speech and Hearing Disorders"},{"issue":"10","key":"9579_CR45","doi-asserted-by":"publisher","first-page":"670","DOI":"10.1001\/archotol.1984.00800360042009","volume":"110","author":"J Robbins","year":"1984","unstructured":"Robbins, J., et al. (1984). Selected acoustic features of tracheoesophageal, esophageal, and laryngeal speech. Archives of Otolaryngology, 110(10), 670\u2013672.","journal-title":"Archives of Otolaryngology"},{"issue":"10","key":"9579_CR46","doi-asserted-by":"publisher","first-page":"2448","DOI":"10.1109\/TBME.2010.2053369","volume":"57","author":"HR Sharifzadeh","year":"2010","unstructured":"Sharifzadeh, H. R., McLoughlin, I. V., & Ahmadi, F. (2010). Reconstruction of normal sounding speech for laryngectomy patients through a modified CELP codec. IEEE Transactions on Biomedical Engineering, 57(10), 2448\u20132458.","journal-title":"IEEE Transactions on Biomedical Engineering"},{"key":"9579_CR47","doi-asserted-by":"publisher","first-page":"417","DOI":"10.1044\/jshr.1003.417","volume":"10","author":"T Shipp","year":"1967","unstructured":"Shipp, T. (1967). Frequency, duration, and perceptual measures in relation to judgments of alaryngeal speech acceptability. Journal of Speech, Language, and Hearing Research, 10, 417\u2013427.","journal-title":"Journal of Speech, Language, and Hearing Research"},{"issue":"3","key":"9579_CR48","doi-asserted-by":"publisher","first-page":"623","DOI":"10.1177\/000348945906800302","volume":"68","author":"JC Snidecor","year":"1959","unstructured":"Snidecor, J. C., & Curry, E. T. (1959). XLIV temporal and pitch aspects of superior esophageal speech. Annals of Otology, Rhinology & Laryngology, 68(3), 623\u2013636.","journal-title":"Annals of Otology, Rhinology & Laryngology"},{"key":"9579_CR50","unstructured":"Srivastava, N. (2013). Improving neural networks with dropout. University of Toronto 182."},{"issue":"1","key":"9579_CR49","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N. (2014). Dropout: A simple way to prevent neural networks from overfitting. Journal of Machine Learning Research, 15(1), 1929\u20131958.","journal-title":"Journal of Machine Learning Research"},{"issue":"2","key":"9579_CR51","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1109\/89.661472","volume":"6","author":"OC Stylianou","year":"1998","unstructured":"Stylianou, O. C., & Moulines, E. (1998). Continuous probabilistic transform for voice conversion. IEEE Transactions on Speech and Audio Processing, 6(2), 131\u2013142.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"issue":"8","key":"9579_CR52","doi-asserted-by":"publisher","first-page":"2222","DOI":"10.1109\/TASL.2007.907344","volume":"15","author":"T Toda","year":"2007","unstructured":"Toda, T., Black, A. W., & Tokuda, K. (2007). Voice conversion based on maximum likelihood estimation of spectral parameter trajectory. IEEE Transactions on Audio, Speech, and Language, 15(8), 2222\u20132235.","journal-title":"IEEE Transactions on Audio, Speech, and Language"},{"key":"9579_CR53","doi-asserted-by":"crossref","unstructured":"T\u00fcurkmen, H. I., & Karsligil, M. E. (2008). Reconstruction of dysphonic speech by melp. In Iberoamerican congress on pattern recognition. Springer, Berlin.","DOI":"10.1007\/978-3-540-85920-8_93"},{"key":"9579_CR54","doi-asserted-by":"crossref","unstructured":"Werghi, A., Di Martino, J., & Jebara, S. B. (2010). On the use of an iterative estimation of continuous probabilistic transforms for voice conversion. In I\/V Communications and mobile network (ISVC), 2010 5th international symposium on. IEEE.","DOI":"10.1109\/ISVC.2010.5656149"},{"key":"9579_CR55","doi-asserted-by":"crossref","unstructured":"Wu, Z., Chng, E. S., & Li, H. (2013). Conditional restricted boltzmann machine for voice conversion. In Signal and information processing (ChinaSIP), 2013 IEEE China summit & international conference on. IEEE.","DOI":"10.1109\/ChinaSIP.2013.6625307"},{"key":"9579_CR56","doi-asserted-by":"crossref","unstructured":"Zhang, M., et al. (2008). Text-independent voice conversion based on state mapped codebook. In Acoustics, speech and signal processing, 2008. ICASSP 2008. IEEE international conference on. IEEE.","DOI":"10.1109\/ICASSP.2008.4518682"},{"issue":"5","key":"9579_CR57","doi-asserted-by":"publisher","first-page":"1645","DOI":"10.1109\/TASL.2007.899236","volume":"15","author":"X Zhu","year":"2007","unstructured":"Zhu, X., Beauregard, G. T., & Wyse, L. L. (2007). Real-time signal estimation from modified short-time Fourier transform magnitude spectra. IEEE Transactions on Audio, Speech, and Language Processing, 15(5), 1645\u20131653.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-018-09579-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-09579-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-09579-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,13]],"date-time":"2024-07-13T07:31:00Z","timestamp":1720855860000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-018-09579-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,12,12]]},"references-count":57,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2019,3]]}},"alternative-id":["9579"],"URL":"https:\/\/doi.org\/10.1007\/s10772-018-09579-1","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,12,12]]},"assertion":[{"value":"25 June 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 November 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 December 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}