{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,10]],"date-time":"2025-10-10T21:37:55Z","timestamp":1760132275837},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2013,12,19]],"date-time":"2013-12-19T00:00:00Z","timestamp":1387411200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2014,3]]},"DOI":"10.1007\/s11265-013-0862-z","type":"journal-article","created":{"date-parts":[[2013,12,18]],"date-time":"2013-12-18T01:06:56Z","timestamp":1387328816000},"page":"423-435","source":"Crossref","is-referenced-by-count":5,"title":["Pitch-Scaled Spectrum Based Excitation Model for HMM-based Speech Synthesis"],"prefix":"10.1007","volume":"74","author":[{"given":"Zhengqi","family":"Wen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianhua","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shifeng","family":"Pan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,12,19]]},"reference":[{"issue":"11","key":"862_CR1","doi-asserted-by":"crossref","first-page":"1039","DOI":"10.1016\/j.specom.2009.04.004","volume":"51","author":"H Zen","year":"2009","unstructured":"Zen, H., Tokuda, K., & Black, A. (2009). Statistical parametric speech synthesis. Speech Communication, 51(11), 1039\u20131064.","journal-title":"Speech Communication"},{"key":"862_CR2","unstructured":"[online] HMM-based Speech Synthesis System (HTS). http:\/\/hts.sp.nitech.ac.jp\/ ."},{"key":"862_CR3","unstructured":"Stylianou, Y. (1996). Harmonic plus Noise Model for Speech, combined with Statistical Methods, for Speech and Speaker Modification. P.h.D. thesis, Ecole Nationale Sup\u00e8rieure des T\u00e9l\u00e9communications. Paris, France."},{"issue":"11","key":"862_CR4","doi-asserted-by":"crossref","first-page":"820","DOI":"10.1109\/LSP.2007.898854","volume":"14","author":"K Hermus","year":"2007","unstructured":"Hermus, K., Van Hamme, H., & Irhimeh, S. (2007). Estimation of the voicing cut-off frequency contour based on a cumulative harmonicity score. IEEE Signal Processing Letters, 14(11), 820\u2013823.","journal-title":"IEEE Signal Processing Letters"},{"issue":"5","key":"862_CR5","doi-asserted-by":"crossref","first-page":"187","DOI":"10.1016\/S0167-6393(98)00085-5","volume":"27","author":"H Kawahara","year":"1999","unstructured":"Kawahara, H., Masuda-Katsuse, I., & de Cheveign\u00e9, A. (1999). Restructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency-based F0 extraction: possible role of a repetitive structure in sounds. Speech Communication, 27(5), 187\u2013207.","journal-title":"Speech Communication"},{"issue":"1","key":"862_CR6","doi-asserted-by":"crossref","first-page":"21","DOI":"10.1109\/89.890068","volume":"9","author":"Y Stylianou","year":"2001","unstructured":"Stylianou, Y. (2001). Applying the harmonic plus noise model in concatenative speech synthesis. IEEE Transactions on Speech Audio Processing, 9(1), 21\u201329.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"key":"862_CR7","unstructured":"Hemptinne, C. (2006). Integration of the harmonic plus noise model (HNM) into the hidden markov model-based speech synthesis, system (HTS). Master thesis. IDIAP Research Institute, IDIAP-RR 69, Switzerland."},{"key":"862_CR8","doi-asserted-by":"crossref","unstructured":"Zen, H., Toda, T., Nakamura, M., & Tokuda, K. (2007). Details of the Nitech HMM-based speech synthesis for Blizzard Challenge 2005. IEICE Transactions on Information and Systems, E90(D), 325\u2013333.","DOI":"10.1093\/ietisy\/e90-1.1.325"},{"key":"862_CR9","unstructured":"Cabral, J. P., Renals, S., Yamagishi, J., & Richmond, K. (2011). HMM-based Speech Synthesizer Using the LF-model of the Glottal Source. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 4704\u20134707."},{"key":"862_CR10","volume-title":"A four-parameter model of glottal flow","author":"G Fant","year":"1985","unstructured":"Fant, G., Liljencrants, J., & Lin, Q. (1985). A four-parameter model of glottal flow. Stockholm: STL-QPSR, KTH."},{"issue":"1","key":"862_CR11","doi-asserted-by":"crossref","first-page":"153","DOI":"10.1109\/TASL.2010.2045239","volume":"19","author":"T Raitio","year":"2010","unstructured":"Raitio, T., Suni, J., Yamagishi, H., Pulakka, A., Nurminen, J., Vainio, M., & Alku, P. (2010). HMM-based speech synthesis utilizing glottal inverse filtering. IEEE Transactions on Speech Audio Processing, 19(1), 153\u2013165.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"issue":"5","key":"862_CR12","doi-asserted-by":"crossref","first-page":"569","DOI":"10.1109\/89.784109","volume":"7","author":"MD Plumpe","year":"1999","unstructured":"Plumpe, M. D., Quatieri, T. F., & Reynolds, D. A. (1999). Modeling of the glottal flow derivative waveform with application to speaker identification. IEEE Transactions on Speech Audio Processing, 7(5), 569\u2013585.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"key":"862_CR13","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Tokuda, K., Masuko, T., & Kitamura, T. (2001). Mixed excitation for HMM-based speech synthesis. 9th European Conference on Speech Communication and Technology, 2263\u20132266.","DOI":"10.21437\/Eurospeech.2001-539"},{"key":"862_CR14","unstructured":"Macree, A. V., Truong, K., George, E. B., Barnwell, T. P., & Viswanathan, V. (1996). A 2.4 kbitsfs MELP Coder Candidate for the New US. Federal Standard. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 200\u2013203."},{"key":"862_CR15","doi-asserted-by":"crossref","first-page":"1779","DOI":"10.21437\/Interspeech.2009-148","volume":"2009","author":"T Drugman","year":"2009","unstructured":"Drugman, T., Wilfart, G., & Dutoit, T. (2009). A deterministic plus stochastic model of the residual signal for improved parametric speech synthesis. Proceedings of Interspeech, 2009, 1779\u20131782.","journal-title":"Proceedings of Interspeech"},{"key":"862_CR16","unstructured":"Maia, R., Toda, T., Zen, H., Nankaku, Y., & Tokuda, K. (2007). An excitation model for HMM-based speech synthesis based on residual modeling. 6the ISCA Workshop on Speech Synthesis, 131\u2013136."},{"issue":"4","key":"862_CR17","doi-asserted-by":"crossref","first-page":"361","DOI":"10.1109\/89.848218","volume":"8","author":"J Skoglund","year":"2000","unstructured":"Skoglund, J., & Bastiaan, W. K. (2000). On time-frequency masking in voiced speech. IEEE Transactions on Speech Audio Processing, 8(4), 361\u2013369.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"issue":"34","key":"862_CR18","first-page":"744","volume":"4","author":"JM Robert","year":"1986","unstructured":"Robert, J. M., & Thomas, F. Q. (1986). Speech analysis\/synthesis based on a sinusoidal representation. IEEE Transactions on Speech Audio Processing, 4(34), 744\u2013754.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"issue":"7","key":"862_CR19","doi-asserted-by":"crossref","first-page":"713","DOI":"10.1109\/89.952489","volume":"9","author":"PJB Jackson","year":"2001","unstructured":"Jackson, P. J. B., & Shadle, C. H. (2001). Pitch-scaled estimation of simultaneous voiced and trubulence-noise components in speech. IEEE Transactions on Speech Audio Processing, 9(7), 713\u2013726.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"key":"862_CR20","doi-asserted-by":"crossref","first-page":"1805","DOI":"10.21437\/Interspeech.2011-34","volume":"2011","author":"ZQ Wen","year":"2011","unstructured":"Wen, Z. Q., & Tao, J. H. (2011). Inverse filtering based harmonic plus noise excitation model for HMM-based speech synthesis. Proceedings of Interspeech, 2011, 1805\u20131808.","journal-title":"Proceedings of Interspeech"},{"key":"862_CR21","doi-asserted-by":"crossref","first-page":"38","DOI":"10.21437\/Interspeech.2010-6","volume":"2010","author":"H Kawahara","year":"2010","unstructured":"Kawahara, H., Morise, M., Takahashi, T., Banno, H., Nisimura, R., & Irino, T. (2010). Simplification and extension of non-periodic excitation source representations for high-quality speech manipulation systems. Proceedings of Interspeech, 2010, 38\u201341.","journal-title":"Proceedings of Interspeech"},{"key":"862_CR22","volume-title":"Acoustic Theory of Speech Production","author":"G Fant","year":"1960","unstructured":"Fant, G. (1960). Acoustic Theory of Speech Production. The Hague: Mouton."},{"issue":"1","key":"862_CR23","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/89.650304","volume":"6","author":"B Yegnanarayana","year":"1998","unstructured":"Yegnanarayana, B., d\u2019Alessandro, C., & Darsinos, V. (1998). An iterative algorithm for decomposition of speech signals into periodic and aperiodic components. IEEE Transactions on Speech Audio Processing, 6(1), 1\u201311.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"issue":"1","key":"862_CR24","doi-asserted-by":"crossref","first-page":"34","DOI":"10.1109\/TASL.2006.876878","volume":"15","author":"P Naylor","year":"2007","unstructured":"Naylor, P., Kounoudes, A., Gudnason, J., & Brookes, M. (2007). Estimation of glottal closure instants in voiced speech using the DYPSA algorithm. IEEE Transactions on Speech Audio Processing, 15(1), 34\u201343.","journal-title":"IEEE Transactions on Speech Audio Processing"},{"issue":"1","key":"862_CR25","doi-asserted-by":"crossref","first-page":"84","DOI":"10.1109\/TASSP.1981.1163506","volume":"29","author":"AH Nuttal","year":"1981","unstructured":"Nuttal, A. H. (1981). Some windows with very good sidelobe behavior. IEEE Transactions on Acoustics, Speech and Audio Processing, 29(1), 84\u201391.","journal-title":"IEEE Transactions on Acoustics, Speech and Audio Processing"},{"key":"862_CR26","unstructured":"Drugman, T., Moinet, A., Dutoit, T., & Wilfart, G. (2009). Using a pitch-synchronous redisual codebook for hybrid HMM\/frame selection speech synthesis. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 3793\u20133796."},{"key":"862_CR27","unstructured":"Raitio, T., Suni, A., Pulakka, H., Vainio, M., & Alku, P. (2011). Utilizing glottal source pulse library for generation improved excitation signal for HMM-based speech synthesis. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 4564\u20134567."},{"key":"862_CR28","doi-asserted-by":"crossref","unstructured":"Fodor, I. K. (2002). A survey of dimension reduction techniques. Technical Report UCRL-ID-148494, Lawrence Livermore National Laboratory. Center for Applied Scientific Computing, USA.","DOI":"10.2172\/15002155"},{"key":"862_CR29","doi-asserted-by":"crossref","DOI":"10.1002\/0471725331","volume-title":"A User\u2019s Guide to Principal components","author":"JE Jackson","year":"1991","unstructured":"Jackson, J. E. (1991). A User\u2019s Guide to Principal components. New York: John Wiley and Sons."},{"key":"862_CR30","unstructured":"Wen, Z. Q., & Tao, J. H. (2011). An excitation model based on inverse filtering for speech analysis and synthesis. IEEE International Workshop on Machine Learning for Signal Processing."},{"key":"862_CR31","doi-asserted-by":"crossref","unstructured":"Linde, Y., Buzo, A., & Gray, R. M. (1980). An algorithm for vector quantizer design. IEEE Transaction on Communications, 28(1), 84\u201385.","DOI":"10.1109\/TCOM.1980.1094577"},{"key":"862_CR32","doi-asserted-by":"crossref","unstructured":"Wen, Z. Q., Tao, J. H., & Hain, H. U. (2012). Pitch-scaled spectrum based excitation model for HMM-based speech synthesis. IEEE 11th International Conference on Signal Processing.","DOI":"10.1109\/ICoSP.2012.6491561"},{"issue":"4","key":"862_CR33","doi-asserted-by":"crossref","first-page":"561","DOI":"10.1109\/PROC.1975.9792","volume":"63","author":"M John","year":"1975","unstructured":"John, M. (1975). Linear prediction: a tutorial review. Proceedings of the IEEE, 63(4), 561\u2013580.","journal-title":"Proceedings of the IEEE"},{"issue":"3","key":"862_CR34","first-page":"455","volume":"E85-D","author":"K Tokuda","year":"2002","unstructured":"Tokuda, K., Masuko, T., Miyazaki, N., & Kobayashi, T. (2002). Multi-space probability distribution HMM. IEICE Transactions on Information and Systems, E85-D(3), 455\u2013464.","journal-title":"IEICE Transactions on Information and Systems"},{"issue":"2","key":"862_CR35","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1250\/ast.21.79","volume":"21","author":"K Shinoda","year":"2000","unstructured":"Shinoda, K., & Watanabe, T. (2000). MDL-based context-dependent subword modeling for speech recognition. The Journal of the Acoustical Society of Japan (e), 21(2), 79\u201386.","journal-title":"The Journal of the Acoustical Society of Japan (e)"},{"key":"862_CR36","doi-asserted-by":"crossref","unstructured":"Tokuda, K., Kobayashi, T., & Imai, S. (1995). Speech parameter generation from HMM using dynamic features. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 660\u2013663","DOI":"10.1109\/ICASSP.1995.479684"},{"key":"862_CR37","unstructured":"Kawahra, H., Morise, M., Takahashi, T., Nisimura, R., & Irino, T. (2006). Tandem-STRAIGHT: A temporally stable power spectral representation for periodic signals and applications to interference-free spectrum, F0, and aperiodicity estimation. Proceedings of ICASSP, 3933\u20133936."},{"issue":"5","key":"862_CR38","doi-asserted-by":"crossref","first-page":"453","DOI":"10.1016\/0167-6393(90)90021-Z","volume":"9","author":"E Moulines","year":"1990","unstructured":"Moulines, E., & Charpentier, F. (1990). Pitch-synchronous waveform processing techniques for text-to-speech synthesis using diphones. Speech Communication, 9(5), 453\u2013467.","journal-title":"Speech Communication"},{"issue":"S1","key":"862_CR39","doi-asserted-by":"crossref","first-page":"S35","DOI":"10.1121\/1.1995189","volume":"57","author":"F Itakura","year":"1975","unstructured":"Itakura, F. (1975). Line spectrum representation of linear predictor coefficients of speech signals. Journal of the Acoustical Society of America, 57(S1), S35\u2013S35.","journal-title":"Journal of the Acoustical Society of America"}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-013-0862-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-013-0862-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-013-0862-z","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,9]],"date-time":"2023-07-09T03:44:48Z","timestamp":1688874288000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-013-0862-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,12,19]]},"references-count":39,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2014,3]]}},"alternative-id":["862"],"URL":"https:\/\/doi.org\/10.1007\/s11265-013-0862-z","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"value":"1939-8018","type":"print"},{"value":"1939-8115","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,12,19]]}}}