{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T07:19:16Z","timestamp":1785395956829,"version":"3.56.0"},"reference-count":33,"publisher":"Elsevier BV","issue":"3-4","license":[{"start":{"date-parts":[[1999,4,1]],"date-time":"1999-04-01T00:00:00Z","timestamp":922924800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Speech Communication"],"published-print":{"date-parts":[[1999,4]]},"DOI":"10.1016\/s0167-6393(98)00085-5","type":"journal-article","created":{"date-parts":[[2003,4,5]],"date-time":"2003-04-05T03:57:58Z","timestamp":1049515078000},"page":"187-207","source":"Crossref","is-referenced-by-count":1128,"title":["Restructuring speech representations using a pitch-adaptive time\u2013frequency smoothing and an instantaneous-frequency-based F0 extraction: Possible role of a repetitive structure in sounds"],"prefix":"10.1016","volume":"27","author":[{"given":"Hideki","family":"Kawahara","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ikuyo","family":"Masuda-Katsuse","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alain","family":"de Cheveign\u00e9","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"9","key":"10.1016\/S0167-6393(98)00085-5_BIB1","first-page":"1188","article-title":"Harmonics estimation based on instantaneous frequency and its application to pitch determination","volume":"E78-D","author":"Abe","year":"1995","journal-title":"IEICE Trans. Information and Systems"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB2","doi-asserted-by":"crossref","unstructured":"Abe, T., Kobayashi, T., Imai, S., 1996. Robust pitch estimation with harmonics enhancement in noisy environments based on instantaneous frequency. In: Proc. ICSLP 96, Philadelphia, pp. 1277\u20131280","DOI":"10.1109\/ICSLP.1996.607843"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB3","doi-asserted-by":"crossref","unstructured":"Abrantes, A.J., Marques, J.S., Trancoso, I.M., 1991. Hybrid sinusoidal modeling of speech without voicing decision. In: Proc. Eurospeech 91, Paris, pp. 231\u2013234","DOI":"10.21437\/Eurospeech.1991-52"},{"issue":"2 pt.2","key":"10.1016\/S0167-6393(98)00085-5_BIB4","doi-asserted-by":"crossref","first-page":"637","DOI":"10.1121\/1.1912679","article-title":"Speech analysis and synthesis by linear prediction of speech wave","volume":"50","author":"Atal","year":"1971","journal-title":"J. Acoust. Soc. Amer."},{"issue":"5","key":"10.1016\/S0167-6393(98)00085-5_BIB5","doi-asserted-by":"crossref","first-page":"1478","DOI":"10.1121\/1.381841","article-title":"Group delay distortion in electroacoustical systems","volume":"63","author":"Blauert","year":"1978","journal-title":"J. Acoust. Soc. Amer."},{"issue":"4","key":"10.1016\/S0167-6393(98)00085-5_BIB6","doi-asserted-by":"crossref","first-page":"520","DOI":"10.1109\/5.135376","article-title":"Estimating and interpreting the instantaneous frequency of a signal \u2013 part 1: Fundamentals","volume":"80","author":"Boashash","year":"1992","journal-title":"Proc. IEEE"},{"issue":"4","key":"10.1016\/S0167-6393(98)00085-5_BIB7","doi-asserted-by":"crossref","first-page":"550","DOI":"10.1109\/5.135378","article-title":"Estimating and interpreting the instantaneous frequency of a signal \u2013 part 2: Algorithms and applications","volume":"80","author":"Boashash","year":"1992","journal-title":"Proc. IEEE"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB8","doi-asserted-by":"crossref","unstructured":"Bregman, A.S., 1990. Auditory Scene Analysis. MIT Press, Cambridge, MA","DOI":"10.7551\/mitpress\/1486.001.0001"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB9","doi-asserted-by":"crossref","unstructured":"Caspers, B., Atal, B., 1987. Role of multi-pulse excitation in synthesis of natural-sounding voiced speech In: Proc. IEEE Internat. Conf. Acoust. Speech and Signal Processing Vol. 4, pp. 2388\u20132391","DOI":"10.1109\/ICASSP.1987.1169921"},{"issue":"7","key":"10.1016\/S0167-6393(98)00085-5_BIB10","doi-asserted-by":"crossref","first-page":"941","DOI":"10.1109\/5.30749","article-title":"Time-frequency distributions \u2013 a review","volume":"77","author":"Cohen","year":"1989","journal-title":"Proc. IEEE"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB11","unstructured":"Cooke, M.P., 1993. Modelling Auditory Processing and Organisation. Cambridge University Press, London"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB12","unstructured":"de Cheveign\u00e9, A., 1996. Speech fundamental frequency estimation. Technical Report TR-H-195, ATR-HIP"},{"issue":"3","key":"10.1016\/S0167-6393(98)00085-5_BIB13","doi-asserted-by":"crossref","first-page":"1261","DOI":"10.1121\/1.423232","article-title":"Cancellation model of pitch perception","volume":"103","author":"de Cheveign\u00e9","year":"1998","journal-title":"J. Acoust. Soc. Amer."},{"issue":"2","key":"10.1016\/S0167-6393(98)00085-5_BIB14","doi-asserted-by":"crossref","first-page":"169","DOI":"10.1121\/1.1916020","article-title":"Remaking speech","volume":"11","author":"Dudley","year":"1939","journal-title":"J. Acoust. Soc. Amer."},{"key":"10.1016\/S0167-6393(98)00085-5_BIB15","doi-asserted-by":"crossref","unstructured":"Dutoit, T., Leich, H., 1993. An analysis of the performance of the MBE model when used in the context of a text-to-speech system. In: Proc. Eurospeech 93, Berlin, pp. 531\u2013534","DOI":"10.21437\/Eurospeech.1993-28"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB16","doi-asserted-by":"crossref","first-page":"411","DOI":"10.1109\/78.80824","article-title":"Discrete all-pole modeling","volume":"39","author":"El-Jaroudi","year":"1991","journal-title":"IEEE Trans. SP"},{"issue":"8","key":"10.1016\/S0167-6393(98)00085-5_BIB17","doi-asserted-by":"crossref","first-page":"1223","DOI":"10.1109\/29.1651","article-title":"Multiband excitation vocoder","volume":"36","author":"Griffin","year":"1988","journal-title":"IEEE Trans. on Acoustics Speech and Signal Processing"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB18","unstructured":"Itakura, F., Saito, S., 1970. A statistical method for estimation of speech spectral density and formant frequencies. Trans. IECE Japan, 53-A, 36\u201343 (in Japanese)"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB19","unstructured":"Kawahara, H., 1997. Speech representation and transformation using adaptive interpolation of weighted spectrum: Vocoder revisited. In: Proc. IEEE Internat. Conf. Acoust. Speech and Signal Processing 2, M\u00fcnich, 1303\u20131306"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB20","unstructured":"Kawahara, H., Masuda, I., 1996. Speech representation and transformation based on adaptive time-frequency interpolation. Technical Report of IEICE, EA96-28, pp. 9\u201316 (in Japanese)"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB21","unstructured":"Kawahara, H., Williams, J.C., 1996. Effects of auditory feedback on voice pitch. In: Davis, P.J., Fletcher, N.H. (Eds.), Vocal Fold Physiology. Singular, M\u00fcnich, Chapter 18, pp. 263\u2013278"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB22","unstructured":"Kawahara, H., Tsuzaki, M., Patterson, Roy D., 1996. A method to shape a class of all-pass filters and their perceptual correlates. Tech. Com. Psycho. Physio. the Acoust. Soc. Jpn., H-96-79, 1\u20138 (in Japanese)"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB23","unstructured":"Marr, D., 1982. Vision: A Computational Investigation into Human Representation and Processing of Visual Information. Freeman, New York"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB24","doi-asserted-by":"crossref","first-page":"744","DOI":"10.1109\/TASSP.1986.1164910","article-title":"Speech analysis\/synthesis based on a sinusoidal representation","volume":"34","author":"McAulay","year":"1986","journal-title":"IEEE Trans. ASSP"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB25","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1016\/0167-6393(94)00058-I","article-title":"Transformation of formants for voice conversion using artificial neural networks","volume":"16","author":"Narendranath","year":"1995","journal-title":"Speech Communication"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB26","unstructured":"Oppenheim, A., Schafer, R., 1989. Discrete-Time Signal Processing. Prentice-Hall, Englewood Cliffs, NJ"},{"issue":"5","key":"10.1016\/S0167-6393(98)00085-5_BIB27","doi-asserted-by":"crossref","first-page":"1560","DOI":"10.1121\/1.395146","article-title":"A pulse ribbon model of monaural phase perception","volume":"82","author":"Patterson","year":"1987","journal-title":"J. Acoust. Soc. Amer."},{"key":"10.1016\/S0167-6393(98)00085-5_BIB28","doi-asserted-by":"crossref","first-page":"1418","DOI":"10.1121\/1.1918360","article-title":"Pitch of the residue","volume":"34","author":"Schouten","year":"1962","journal-title":"J. Acoust. Soc. Amer."},{"key":"10.1016\/S0167-6393(98)00085-5_BIB29","doi-asserted-by":"crossref","unstructured":"Secrest, B.G., Doddington, G.R., 1983. An integrated pitch tracking algorithm for speech systems. In: Proc. IEEE ICASSP83, pp. 1352\u20131355","DOI":"10.1109\/ICASSP.1983.1172016"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB30","doi-asserted-by":"crossref","unstructured":"Slaney, M., Covell, M., Lassiter, B., 1996. Automatic audio morphing. In: Proc. IEEE Internat. Conf. Acoust. Speech and Signal Processing, Atlanta, pp. 1\u20134","DOI":"10.1109\/ICASSP.1996.543292"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB31","doi-asserted-by":"crossref","unstructured":"Stylianou, Y., Laroche, J., Moulines, E., 1995. High-quality speech modification based on a harmonic+noise model. In: Proc. Eurospeech 95, Madrid, pp. 451\u2013454","DOI":"10.21437\/Eurospeech.1995-122"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB32","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1016\/0167-6393(95)00044-5","article-title":"Time-scale and pitch modifications of speech signals and resynthesis from the discrete short-time Fourier transform","volume":"18","author":"Veldhuis","year":"1996","journal-title":"Speech Communication"},{"key":"10.1016\/S0167-6393(98)00085-5_BIB33","doi-asserted-by":"crossref","unstructured":"Webster, D.B., Popper, A.N., Fay, R.R., 1992. The Mammalian Auditory Pathway: Neuroanatomy. Springer, Berlin, 1992","DOI":"10.1007\/978-1-4612-4416-5"}],"container-title":["Speech Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639398000855?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167639398000855?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2023,4,16]],"date-time":"2023-04-16T06:21:17Z","timestamp":1681626077000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167639398000855"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[1999,4]]},"references-count":33,"journal-issue":{"issue":"3-4","published-print":{"date-parts":[[1999,4]]}},"alternative-id":["S0167639398000855"],"URL":"https:\/\/doi.org\/10.1016\/s0167-6393(98)00085-5","relation":{},"ISSN":["0167-6393"],"issn-type":[{"value":"0167-6393","type":"print"}],"subject":[],"published":{"date-parts":[[1999,4]]}}}