{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T14:38:02Z","timestamp":1740148682575,"version":"3.37.3"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2017,9,26]],"date-time":"2017-09-26T00:00:00Z","timestamp":1506384000000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"the National High-Tech Research and Development Program of China (863 Program)","award":["2015AA016305"],"award-info":[{"award-number":["2015AA016305"]}]},{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China (NSFC)","doi-asserted-by":"crossref","award":["61305003"],"award-info":[{"award-number":["61305003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China (NSFC)","doi-asserted-by":"crossref","award":["61425017"],"award-info":[{"award-number":["61425017"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China (NSFC)","doi-asserted-by":"crossref","award":["61403386"],"award-info":[{"award-number":["61403386"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the Strategic Priority Research Program of the CAS","award":["Grant XDB02080006"],"award-info":[{"award-number":["Grant XDB02080006"]}]},{"name":"the Major Program for the National Social Science Fund of China","award":["13&ZD189"],"award-info":[{"award-number":["13&ZD189"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2018,7]]},"DOI":"10.1007\/s11265-017-1290-2","type":"journal-article","created":{"date-parts":[[2017,9,25]],"date-time":"2017-09-25T22:13:06Z","timestamp":1506377586000},"page":"1039-1052","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Investigating Deep Neural Network Adaptation for Generating Exclamatory and Interrogative Speech in Mandarin"],"prefix":"10.1007","volume":"90","author":[{"given":"Yibin","family":"Zheng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ya","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengqi","family":"Wen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianhua","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,9,26]]},"reference":[{"key":"1290_CR1","doi-asserted-by":"crossref","unstructured":"Black, A. & Cambpbell, N. (1995). Optimising selection of units from speech database for concatenative synthesis. In Proc. European Conference on Speech Communication and Technology, EUROSPEECH\u201995, pp. 581\u2013584.","DOI":"10.21437\/Eurospeech.1995-148"},{"key":"1290_CR2","doi-asserted-by":"crossref","unstructured":"Hunt, A., & Black, A. (1996). Unit selection in a concatenative speech synthesis system using a large speech database. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201996, pp. 373\u2013376.","DOI":"10.1109\/ICASSP.1996.541110"},{"issue":"3","key":"1290_CR3","doi-asserted-by":"crossref","first-page":"223","DOI":"10.1006\/csla.1999.0123","volume":"13","author":"R Donovan","year":"1999","unstructured":"Donovan, R., & Woodland, P. (1999). A hidden Markov-model-based trainable speech synthesizer. Computer Speech & Language, 13(3), 223\u2013241.","journal-title":"Computer Speech & Language"},{"key":"1290_CR4","doi-asserted-by":"crossref","unstructured":"Merritt, T., Clark, R. A. J., Wu, Z., Yamagishi, J., & King. S. (2016). Deep neural network-guided unit selection synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201916, pp. 333\u2013336.","DOI":"10.1109\/ICASSP.2016.7472658"},{"key":"1290_CR5","doi-asserted-by":"crossref","unstructured":"Black, A. (2003). Unit selection and emotional speech. In Proc. European Conference on Speech Communication and Technology, EUROSPEECH\u201903, pp. 1649\u20131652.","DOI":"10.21437\/Eurospeech.2003-473"},{"issue":"12","key":"1290_CR6","first-page":"2184","volume":"J79-D-II","author":"T Masuko","year":"1996","unstructured":"Masuko, T., Tokuda, K., Kobayashi, T., & Imai, S. (1996). HMM-based speech synthesis using dynamic features (in Japanese). IEICE Transactions, J79-D-II(12), 2184\u20132190.","journal-title":"IEICE Transactions"},{"issue":"11","key":"1290_CR7","first-page":"2099","volume":"J83-D-II","author":"T Yoshimura","year":"2000","unstructured":"Yoshimura, T., Tokuda, K., Masuko, T., Kobayashi, T., & Kitamura, T. (2000). Simultaneous modeling of spectrum, pitch and duration in HMM-based speech synthesis, (in Japanese). IEICE Transactions, J83-D-II(11), 2099\u20132107.","journal-title":"IEICE Transactions"},{"key":"1290_CR8","doi-asserted-by":"crossref","unstructured":"Tokuda, K., Yoshimura, T., Masuko, T., Kobayashi, T., & Kitamura, T. (2000). Speech parameter generation algorithms for HMM-based speech synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201900, pp. 1315\u20131318.","DOI":"10.1109\/ICASSP.2000.861820"},{"issue":"3","key":"1290_CR9","doi-asserted-by":"crossref","first-page":"503","DOI":"10.1093\/ietisy\/e88-d.3.502","volume":"E88-D","author":"J Yamagishi","year":"2005","unstructured":"Yamagishi, J., Onishi, K., Masuko, T., & Kobayashi, T. (2005). Acoustic modeling of speaking styles and emotional expressions in HMM-based speech synthesis. IEICE Transactions on Information and Systems, E88-D(3), 503\u2013509.","journal-title":"IEICE Transactions on Information and Systems"},{"issue":"3","key":"1290_CR10","doi-asserted-by":"crossref","first-page":"1092","DOI":"10.1093\/ietisy\/e89-d.3.1092","volume":"E89-D","author":"M Tachibana","year":"2006","unstructured":"Tachibana, M., Yamagishi, J., Masuko, T., & Kobayashi, T. (2006). A style adaptation technique for speech synthesis using HSMM and supra segmental features. IEICE Transactions on Information and Systems, E89-D(3), 1092\u20131099.","journal-title":"IEICE Transactions on Information and Systems"},{"issue":"11","key":"1290_CR11","doi-asserted-by":"crossref","first-page":"2484","DOI":"10.1093\/ietisy\/e88-d.11.2484","volume":"E88-D","author":"M Tachibana","year":"2005","unstructured":"Tachibana, M., Yamagishi, J., Masuko, T., & Kobayashi, T. (2005). Speech synthesis with various emotional expressions and speaking styles by style interpolation and morphing. IEICE Transactions on Information and Systems, E88-D(11), 2484\u20132491.","journal-title":"IEICE Transactions on Information and Systems"},{"issue":"9","key":"1290_CR12","doi-asserted-by":"crossref","first-page":"1406","DOI":"10.1093\/ietisy\/e90-d.9.1406","volume":"E90-D","author":"T Nose","year":"2007","unstructured":"Nose, T., Yamagishi, J., & Kobayashi, T. (2007). A style control technique for HMM-based expressive speech synthesis. IEICE Transactions on Information and Systems, E90-D(9), 1406\u20131413.","journal-title":"IEICE Transactions on Information and Systems"},{"key":"1290_CR13","doi-asserted-by":"crossref","unstructured":"Masuko, T., Tokuda, K., Kobayashi, T., & Imai, S. (1997) Voice characteristic conversion for HMM-based speech synthesis system. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201997, pp. 1611\u20131614.","DOI":"10.1109\/ICASSP.1997.598807"},{"key":"1290_CR14","doi-asserted-by":"crossref","unstructured":"Tamura, M., Masuko, T., Tokuda, K., & Kobayashi, T. (2001). Adaptation of pitch and spectrum for HMM-based speech synthesis using MLLR. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201901, pp. 805\u2013808.","DOI":"10.1109\/ICASSP.2001.941037"},{"issue":"8","key":"1290_CR15","first-page":"1956","volume":"E86-A","author":"J Yamagishi","year":"2003","unstructured":"Yamagishi, J., Tamura, M., Masuko, T., Tokuda, K., & Kobayashi, T. (Aug. 2003). A training method of average voice model for HMM-based speech synthesis. IEICE Transactions on Fundamentals, E86-A(8), 1956\u20131963.","journal-title":"IEICE Transactions on Fundamentals"},{"issue":"2","key":"1290_CR16","doi-asserted-by":"crossref","first-page":"533","DOI":"10.1093\/ietisy\/e90-d.2.533","volume":"E90-D","author":"J Yamagishi","year":"2007","unstructured":"Yamagishi, J., & Kobayashi, T. (Feb. 2007). Average-voice-based speech synthesis using HSMM-based speaker adaptation and adaptive training. IEICE Transactions on Information and Systems, E90-D(2), 533\u2013543.","journal-title":"IEICE Transactions on Information and Systems"},{"key":"1290_CR17","unstructured":"Tamura, M., Masuko, T., Tokuda, K., & Kobayashi, T. (1998). Speaker adaptation for HMM-based speech synthesis system using MLLR. In Proc. in Proc. IEEE Speech Synthesis Workshop, pp. 273\u2013276."},{"issue":"2","key":"1290_CR18","doi-asserted-by":"crossref","first-page":"171","DOI":"10.1006\/csla.1995.0010","volume":"9","author":"C Leggetter","year":"1995","unstructured":"Leggetter, C., & Woodland, P. (1995). Maximum likelihood linear regression for speaker adaptation of continuous density hidden Markov models. Computer Speech & Language, 9(2), 171\u2013185.","journal-title":"Computer Speech & Language"},{"issue":"12","key":"1290_CR19","first-page":"2509","volume":"J83-D-II","author":"T Masuko","year":"2000","unstructured":"Masuko, T., Tamura, M., & Tokuda, K. (2000). Voice characteristics conversion for HMM-based speech synthesis system using MAP-VFS (in Japanese). IEICE Transactions, J83-D-II(12), 2509\u20132516.","journal-title":"IEICE Transactions"},{"issue":"4","key":"1290_CR20","first-page":"545","volume":"J85-D-II","author":"M Tamura","year":"2002","unstructured":"Tamura, M., Masuko, T., Tokuda, K., & Kobayashi, T. (2002). Speaker adaptation of pitch and spectrum for HMM-based speech synthesis, (in Japanese). IEICE Transactions, J85-D-II(4), 545\u2013553.","journal-title":"IEICE Transactions"},{"issue":"5","key":"1290_CR21","doi-asserted-by":"crossref","first-page":"825","DOI":"10.1093\/ietisy\/e90-d.5.825","volume":"E90-D","author":"H Zen","year":"2007","unstructured":"Zen, H., Tokuda, K., Masuko, T., Kobayashi, T., & Kitamura, T. (May 2007). A hidden semi-Markov model-based speech synthesis system. IEICE Transactions on Information and Systems, E90-D(5), 825\u2013834.","journal-title":"IEICE Transactions on Information and Systems"},{"key":"1290_CR22","unstructured":"Fang, S., Wen, Z., & Tao, J. (2015). Speech synthesis of questions based on adaptive training. In Proc. NCMMSC, 2015, pp. 092\u2013095."},{"key":"1290_CR23","doi-asserted-by":"crossref","unstructured":"Yamagishi, J., Kobayashi, T., & Isogai, J. (2009). Analysis of speaker adaptation algorithms for HMM-based speech synthesis and a constrained SMAPLR Adaptation algorithm. IEICE Transactions on Information and Systems, 17(1), 66\u201384.","DOI":"10.1109\/TASL.2008.2006647"},{"key":"1290_CR24","doi-asserted-by":"crossref","unstructured":"Fan, Y., Qian, Y., Soong, F. K., & He, L. (2015). Multi-speaker modeling and speaker adaptation for DNN-based TTS synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201915, pp. 4475\u20134479.","DOI":"10.1109\/ICASSP.2015.7178817"},{"key":"1290_CR25","doi-asserted-by":"crossref","unstructured":"Zen, H., Senior, A., & Schuster, M (2013). Statistical parametric speech synthesis using deep neural networks. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201913","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"1290_CR26","doi-asserted-by":"crossref","unstructured":"Tokuda, K., & Zen, H. (2015). Directly modeling speech waveforms by neural networks for statistical parametric speech synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201915 pp. 4215\u20134219.","DOI":"10.1109\/ICASSP.2015.7178765"},{"issue":"8","key":"1290_CR27","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., & Schmidhuber, J. (1997). Long short-term memory. Neural Computation, 9(8), 1735\u20131780.","journal-title":"Neural Computation"},{"key":"1290_CR28","unstructured":"Chung, J., Gulcehre, C., Cho, K., & Bengio, Y. (2014). Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv preprint arXiv:1412.3555."},{"key":"1290_CR29","doi-asserted-by":"crossref","unstructured":"Zen, H., & Sak, H., (2015). Unidirectional long short-term memory recurrent neural network with recurrent output layer for low-latency speech synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201915, pp. 4470\u20134474.","DOI":"10.1109\/ICASSP.2015.7178816"},{"key":"1290_CR30","doi-asserted-by":"crossref","unstructured":"Wu, Z. Z., Valentini-Botinhao, C., Watts, O., & King, S. (2015). Deep neural networks employing multi-task learning and stacked bottleneck features for speech synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201915, pp. 4460\u20134464.","DOI":"10.1109\/ICASSP.2015.7178814"},{"key":"1290_CR31","doi-asserted-by":"crossref","unstructured":"Wu, Z. & King, S. (2016). Investigating gated recurrent networks for speech synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201916, pp. 5140\u20135144.","DOI":"10.1109\/ICASSP.2016.7472657"},{"key":"1290_CR32","doi-asserted-by":"crossref","unstructured":"Wu, Z. Z., Swietojanski, P., Veaux, C., & King, S., (2015). A study of speaker adaptation for DNN-based speech synthesis. In Proc. INTERSPEECH\u201915, pp. 879\u2013883.","DOI":"10.21437\/Interspeech.2015-270"},{"key":"1290_CR33","unstructured":"Potard, B., Motlicek, P., & Imseng, D. (2015). Preliminary work on speaker adaptation for DNN-based speech synthesis. Tech. Rep., No. EPFL-REPORT- 204660, Idiap, 2015"},{"key":"1290_CR34","unstructured":"Yuchen Fan, Y. Q., Song, F. K., & He, L. (2015). Unsupervised speaker adaptation for DNN-based tts synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201915 pp. 5135\u20135139."},{"key":"1290_CR35","doi-asserted-by":"crossref","unstructured":"Fan, Y., Y. Qian, F.K. Soong, & L. He, (2016). Speaker and language factorization in DNN-based TTS synthesis. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201916, pp. 1005\u20131008.","DOI":"10.1109\/ICASSP.2016.7472737"},{"key":"1290_CR36","unstructured":"Yu, Q., Liu, P., & Cai, L. (2016). Learning cross-lingual information with multilingual BLTSM for speech synthesis of low-resource language. In Proc. IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201916, pp. 1233\u20131236."},{"issue":"5","key":"1290_CR37","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1109\/89.466659","volume":"3","author":"V Digalakis","year":"1995","unstructured":"Digalakis, V., Rtischev, D., & Neumeyer, L. (1995). Speaker adaptation using constrained reestimation of Gaussian mixtures. IEEE Trans. Speech Audio Process., 3(5), 357\u2013366.","journal-title":"IEEE Trans. Speech Audio Process."},{"issue":"2","key":"1290_CR38","doi-asserted-by":"crossref","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"M Gales","year":"1998","unstructured":"Gales, M. (1998). Maximum likelihood linear transformations for HMM-based speech recognition. Computer Speech & Language, 12(2), 75\u201398.","journal-title":"Computer Speech & Language"},{"issue":"3\u20134","key":"1290_CR39","doi-asserted-by":"crossref","first-page":"187","DOI":"10.1016\/S0167-6393(98)00085-5","volume":"27","author":"H Kawahara","year":"1999","unstructured":"Kawahara, H., Masuda-Katsuse, I., & de Cheveigne, A. (1999). Restructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency based F0 extraction: Possible role of a repetitive structure in sounds. Speech Communication, 27(3\u20134), 187\u2013207.","journal-title":"Speech Communication"},{"key":"1290_CR40","unstructured":"Theano. (2016). A Python framework for fast computation of mathematical expressions. [OL] [2016\u201305-09] http:\/\/deeplearning.net\/software\/theano\/ . Accessed 7 July 2016."},{"key":"1290_CR41","unstructured":"Tokuda, K., Zen, H., & Black, A.W. (2002). An HMM-based speech synthesis system applied to English. In Proc. IEEE Speech Synthesis Workshop, pp. 227\u2013230."}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-017-1290-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-017-1290-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-017-1290-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,26]],"date-time":"2023-08-26T10:24:34Z","timestamp":1693045474000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-017-1290-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,9,26]]},"references-count":41,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2018,7]]}},"alternative-id":["1290"],"URL":"https:\/\/doi.org\/10.1007\/s11265-017-1290-2","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"type":"print","value":"1939-8018"},{"type":"electronic","value":"1939-8115"}],"subject":[],"published":{"date-parts":[[2017,9,26]]}}}