{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T07:28:58Z","timestamp":1758266938037,"version":"3.41.0"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2017,10,2]],"date-time":"2017-10-02T00:00:00Z","timestamp":1506902400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"name":"National High-Tech Research and Development Program of China (863 Program)","award":["2015AA016305"],"award-info":[{"award-number":["2015AA016305"]}]},{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China (NSFC)","doi-asserted-by":"crossref","award":["61403386"],"award-info":[{"award-number":["61403386"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the Strategic Priority Research Program of the CAS","award":["XDB02080006"],"award-info":[{"award-number":["XDB02080006"]}]},{"name":"the Major Program for the National Social Science Fund of China","award":["13&ZD189"],"award-info":[{"award-number":["13&ZD189"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2018,7]]},"DOI":"10.1007\/s11265-017-1293-z","type":"journal-article","created":{"date-parts":[[2017,10,2]],"date-time":"2017-10-02T05:57:50Z","timestamp":1506923870000},"page":"1025-1037","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Improving Deep Neural Network Based Speech Synthesis through Contextual Feature Parametrization and Multi-Task Learning"],"prefix":"10.1007","volume":"90","author":[{"given":"Zhengqi","family":"Wen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kehuang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chin-Hui","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianhua","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,10,2]]},"reference":[{"key":"1293_CR1","doi-asserted-by":"crossref","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"G Hinton","year":"2006","unstructured":"Hinton, G., Osindero, S., & Teh, Y. (2006). A Fast Learning Algorithm for Deep Belief Nets. Neural Computation, 18, 1527\u20131554.","journal-title":"Neural Computation"},{"key":"1293_CR2","doi-asserted-by":"crossref","first-page":"428","DOI":"10.1016\/j.tics.2007.09.004","volume":"11","author":"G-E Hinton","year":"2007","unstructured":"Hinton, G.-E. (2007). Learning multiple layers of representation. Trends in Cognitive Sciences, 11, 428\u2013434.","journal-title":"Trends in Cognitive Sciences"},{"key":"1293_CR3","unstructured":"LeCun, Y., Boser, B., Denker, J., Henderson, D., Howard, R., Hubbard, W., & Jackel, L. (1990) Handwritten digit recognition with a back-propagation network. In Advances in Neural Information Processing Systems (NIPS), pp. 396\u2013404."},{"issue":"4","key":"1293_CR4","doi-asserted-by":"crossref","first-page":"541","DOI":"10.1162\/neco.1989.1.4.541","volume":"1","author":"Y LeCun","year":"1989","unstructured":"LeCun, Y., Boser, B., Denker, J. S., Henderson, D., Howard, R. E., Hubbard, W., & Jackel, L. D. (1989). Backpropagation applied to handwritten zip code recognition. Neural Computation, 1(4), 541\u2013551.","journal-title":"Neural Computation"},{"issue":"11","key":"1293_CR5","doi-asserted-by":"crossref","first-page":"2673","DOI":"10.1109\/78.650093","volume":"45","author":"S Mike","year":"1997","unstructured":"Mike, S., & Paliwal, K. (1997). Bidirectional recurrent neural networks. IEEE Transactions on Signal Processing, 45(11), 2673\u20132681.","journal-title":"IEEE Transactions on Signal Processing"},{"issue":"8","key":"1293_CR6","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"H Sepp","year":"1997","unstructured":"Sepp, H., & J\u00fcrgen, S. (1997). Long short-term memory. Neural Computation, 9(8), 1735\u20131780.","journal-title":"Neural Computation"},{"issue":"6","key":"1293_CR7","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G. E., Mohamed, A., Jaitly, N., Senior, A., Vanhoucke, V., Nguyen, P., Sainath, T. N., & Kingsbury, B. (2012). Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups. IEEE Signal Processing Magazine, 29(6), 82\u201397.","journal-title":"IEEE Signal Processing Magazine"},{"key":"1293_CR8","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A., Hinton, G. (2013). Speech Recognition with Deep Recurrent Neural Network. In Proc. of ICASSP, pp. 6645\u20136649.","DOI":"10.1109\/ICASSP.2013.6638947"},{"issue":"10","key":"1293_CR9","doi-asserted-by":"crossref","first-page":"1533","DOI":"10.1109\/TASLP.2014.2339736","volume":"22","author":"O Abdel-Hamid","year":"2014","unstructured":"Abdel-Hamid, O., Mohamed, A., Jiang, H., Deng, L., Penn, G., & Yu, D. (2014). Convolutional Neural Networks for Speech Recognition. In IEEE\/ACM Trans. on Audio, Speech and Language Processing, 22(10), 1533\u20131545.","journal-title":"In IEEE\/ACM Trans. on Audio, Speech and Language Processing"},{"issue":"2","key":"1293_CR10","first-page":"1097","volume":"1","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. (2012). ImageNet Classification with Deep Convolutional Neural Networks. Proc. of NIPS, 1(2), 1097\u20131105.","journal-title":"Proc. of NIPS"},{"key":"1293_CR11","doi-asserted-by":"crossref","unstructured":"K.M. He, X.Y. Zhang, S.Q. Ren and J. Sun, Deep Residual Learning for Image Recognition. In Proc. of CVPR, pp. 770\u2013778, 2015.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1293_CR12","first-page":"2493","volume":"12","author":"R Collobert","year":"2011","unstructured":"Collobert, R., Weston, J., Bottou, L., Karlen, M., Kavukcuoglu, K., & Kuksa, P. (2011). Natural Language Processing (Almost) from Scratch. Journal of Machind Learning Research, 12, 2493\u20132537.","journal-title":"Journal of Machind Learning Research"},{"key":"1293_CR13","doi-asserted-by":"crossref","unstructured":"Kim, Y. (2014) Convolutional Neural Networks for Sentence Classification. In: Proc. of EMNLP, pp. 1746\u20131751.","DOI":"10.3115\/v1\/D14-1181"},{"key":"1293_CR14","doi-asserted-by":"crossref","unstructured":"Cho, K., Merrienboer, B., Bahdanau, D., Bougares, F., Schwenk, H., Bengio, Y. (2014) Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation. In: Proc. of EMNLP.","DOI":"10.3115\/v1\/D14-1179"},{"key":"1293_CR15","doi-asserted-by":"crossref","unstructured":"Hunt, A. J., Black, A. W. (1996) Unit selection in a concatenative speech synthesis system using a large speech database. In Proc. of ICASSP, pp. 373\u2013376.","DOI":"10.1109\/ICASSP.1996.541110"},{"key":"1293_CR16","unstructured":"H. Kawai, T. Toda, J. Ni, et al. (2004) XIMERA: A new TTS from ATR based on corpus-based technologies. In Proc. of Fifth ISCA Workshop on Speech Synthesis."},{"key":"1293_CR17","doi-asserted-by":"crossref","unstructured":"Ling, Z. H., Wang, R. H. (2007) HMM-based hierarchical unit selection combining Kullback-Leibler divergence with likelihood criterion. In Proc. of ICASSP, pp. 1245\u20131248.","DOI":"10.1109\/ICASSP.2007.367302"},{"key":"1293_CR18","first-page":"1229","volume":"4","author":"AW Black","year":"2007","unstructured":"Black, A. W., Zen, H., & Tokuda, K. (2007). Statistical parametric speech synthesis. Proc. ICASSP, 4, 1229\u20131232.","journal-title":"Proc. ICASSP"},{"issue":"11","key":"1293_CR19","doi-asserted-by":"crossref","first-page":"1039","DOI":"10.1016\/j.specom.2009.04.004","volume":"51","author":"H Zen","year":"2009","unstructured":"Zen, H., Tokuda, K., & Black, A. W. (2009). Statistical Parametric Speech Synthesis. Speech Communication, 51(11), 1039\u20131064.","journal-title":"Speech Communication"},{"key":"1293_CR20","doi-asserted-by":"crossref","unstructured":"T. Yoshimura, K. Tokuda, T. Masuko, T. Kobayashi, and T. Kitamura (1999) Simultaneous modeling of spectrum, pitch and duration in HMM-based speech synthesis. In Proc. of Eurospeech, pp. 2347\u20132350.","DOI":"10.21437\/Eurospeech.1999-513"},{"key":"1293_CR21","doi-asserted-by":"crossref","first-page":"35","DOI":"10.1109\/MSP.2014.2359987","volume":"32","author":"ZH Ling","year":"2015","unstructured":"Ling, Z. H., Kang, S. Y., Zen, H., Senior, A., Schuster, M., Qian, X. J., Meng, H., & Deng, L. (2015). Deep Learning for Acoustic Modeling in Parametric Speech Generation. Journal of IEEE Signal Processing Magazine, 32, 35\u201352.","journal-title":"Journal of IEEE Signal Processing Magazine"},{"key":"1293_CR22","unstructured":"Bengio, Y., Ducharme, R., Vincent, P., & Jauvin, C. (2003). A Neural probabilistic language model. Journal of Machine Learning Research, 1137\u20131155."},{"key":"1293_CR23","unstructured":"Mikolov, T., Karafiat, M., Burget, L. (2010) J. \u201cHonza\u201d Cernocky and S. Khudanpur, \u201cRecurrent neural network based language model. In Proc. of INTERSPEECH, pp. 1045\u20131048."},{"key":"1293_CR24","first-page":"2493","volume":"12","author":"R Collobert","year":"2011","unstructured":"Collobert, R., Weston, J., Bottou, L., Karlen, M., Kavukcuoglu, K., & Kuksa, P. (2011). Natural language processing (almost) from scratch. Journal of Machine Learning Research, 12, 2493\u20132537.","journal-title":"Journal of Machine Learning Research"},{"key":"1293_CR25","first-page":"873","volume":"1","author":"EH Huang","year":"2012","unstructured":"Huang, E. H., Socher, R., Manning, C. D., & Ng, A. Y. (2012). Improving word representations via global context and multiple word prototypes. Proc. of ACL, 1, 873\u2013882.","journal-title":"Proc. of ACL"},{"key":"1293_CR26","unstructured":"T. Mikolov, K. Chen, G. Corrado and J. Dean (2013) Efficient estimation of word representations in vector space. In Proc. of CoRR."},{"key":"1293_CR27","doi-asserted-by":"crossref","unstructured":"Kang, S., Qian, X., & Meng, H. (2013). Multi-distribution deep belief network for speech synthesis. In Proc. of ICASSP, pp.7962\u20137966.","DOI":"10.1109\/ICASSP.2013.6639225"},{"key":"1293_CR28","doi-asserted-by":"crossref","unstructured":"Ling, Z.-H., Deng, L., & Yu, D. (2013) Modeling spectral envelopes using restricted Boltzmann machines for statistical parametric speech synthesis. In Proc. of ICASSP, pp. 7825\u20137829.","DOI":"10.1109\/ICASSP.2013.6639187"},{"key":"1293_CR29","doi-asserted-by":"crossref","unstructured":"Fan, Y.-C., Qian, Y., Xie, F.-L. & Soong, F. K. (2014) TTS Synthesis with Bidirectional LSTM based Recurrent Neural Networks. In Proc. of Interspeech, pp.1964\u20131968.","DOI":"10.21437\/Interspeech.2014-443"},{"key":"1293_CR30","doi-asserted-by":"crossref","first-page":"148","DOI":"10.1016\/j.neucom.2012.11.008","volume":"106","author":"SM Siniscalchi","year":"2013","unstructured":"Siniscalchi, S. M., Yu, D., Deng, L., & Lee, C.-H. (2013). Exploiting Deep Neural Networks for Detection-Based Speech Recognition. Neurocomputing, 106, 148\u2013157.","journal-title":"Neurocomputing"},{"key":"1293_CR31","doi-asserted-by":"crossref","unstructured":"C.-H. Lee and S. M. Siniscalchi, \u201cAn Information-Extraction Approach to Speech Processing: Analysis, Detection, Verification and Recognition,\u201d Proceedings of the IEEE, Vol. 101, No. 5, pp. 1089\u20131115, May 2013.","DOI":"10.1109\/JPROC.2013.2238591"},{"key":"1293_CR32","doi-asserted-by":"crossref","first-page":"41","DOI":"10.1023\/A:1007379606734","volume":"28","author":"R Caruana","year":"1997","unstructured":"Caruana, R. (1997). Multitask learning. Machine Learning Journal, 28, 41\u201375.","journal-title":"Machine Learning Journal"},{"key":"1293_CR33","doi-asserted-by":"crossref","unstructured":"Wu, Z., Valentini-Botinhao, C., Watts, O., & King, S. (2015). Deep neural networks employing multi-task learning and stacked bottleneck features for speech synthesis. In Proc. of ICASSP, pp. 4460\u20134464.","DOI":"10.1109\/ICASSP.2015.7178814"},{"key":"1293_CR34","doi-asserted-by":"crossref","unstructured":"Tokuda, K., Kobayashi, T., & Imai, S. (1995). Speech parameter generation from HMM using dynamic features. Proc. of ICASSP, pp. 660\u2013663.","DOI":"10.1109\/ICASSP.1995.479684"},{"key":"1293_CR35","doi-asserted-by":"crossref","unstructured":"Song, E., Joo, Y.-S., & Kang, H.-G. (2015) Improved Time-Frequency Trajectory Excitation Modeling for a Statistical Parametric Speech Synthesis System. In Proc. of ICASSP.","DOI":"10.1109\/ICASSP.2015.7178912"},{"key":"1293_CR36","unstructured":"Fan, B., Lee, S.-W., Tian, X.-H., Xie, L., & Dong, M.-H. (2015). A Waveform Representation Framework for High-Qaulity Statistical Parametric Speech Synthesis. In Proc. of APASIPA."},{"key":"1293_CR37","doi-asserted-by":"crossref","unstructured":"Hu, Q., Yamagishi, J., Richmond, K., Subramanian K., & Stylianou, Y. (2016) Initial Investigation of Speech Synthesis based on Complex-Valued Neural Networks. In Proc. of ICASSP, pp. 5630\u20135634.","DOI":"10.1109\/ICASSP.2016.7472755"},{"key":"1293_CR38","doi-asserted-by":"crossref","unstructured":"Wen, Z. Q., Kawahara, H., & Tao, J. H., (2012) Pitch-Scaled Analysis based Residual Reconstruction for Speech Analysis and Synthesis. In Proc. of INTERSPEECH, pp. 374\u2013377.","DOI":"10.21437\/Interspeech.2012-136"},{"issue":"7","key":"1293_CR39","doi-asserted-by":"crossref","first-page":"713","DOI":"10.1109\/89.952489","volume":"9","author":"PJB Jackson","year":"2001","unstructured":"Jackson, P. J. B., & Shadle, C. H. (2001). Pitch-Scaled Estimation of Simultaneous Voiced and Trubulence-Noise Components in Speech. IEEE Trans. On Speech Audio Processing, 9(7), 713\u2013726.","journal-title":"IEEE Trans. On Speech Audio Processing"},{"key":"1293_CR40","first-page":"1.10.1","volume":"1","author":"F-K Soong","year":"1984","unstructured":"Soong, F.-K., & Juang, B.-H. (1984). Line spectrum pair (UP) and speech data compression. Proc. of ICASSP, San Diego, 1, 1.10.1\u20131.10.4.","journal-title":"Proc. of ICASSP, San Diego"},{"key":"1293_CR41","unstructured":"Watts, O. (2013). Unsupervised learning for text-to-speech synthesis. PhD dissertation."},{"key":"1293_CR42","unstructured":"Chen, X., Xu, L., Liu, Z., Sun, M., & Luan, H. (2015). Joint learning of character and word embeddings. In International Joint Conference on Artificial Intelligence."},{"key":"1293_CR43","doi-asserted-by":"crossref","unstructured":"Wen, Z. Q., Li, Y., & Tao, J. H. (2016). The Parameterized Phoneme Identity Feature as a Continuous Real-Valued Vector for Neural Network based Speech Synthesis. In Proc. of INTERSPEECH.","DOI":"10.21437\/Interspeech.2016-222"},{"key":"1293_CR44","doi-asserted-by":"crossref","unstructured":"Collobert, R., & Weston, J. (2008). A unified architecture for natural language processing: Deep neural networks with multitask learning. In International Conference on Machine Learning.","DOI":"10.1145\/1390156.1390177"},{"key":"1293_CR45","doi-asserted-by":"crossref","unstructured":"Sun, F., Guo, J., Lan, Y., Xu, J., & Cheng, X. (2016). Inside out: Two jointly predictive models for word representations and phrase representations. In Proceedings of the 30th AAAI conference.","DOI":"10.1609\/aaai.v30i1.10338"},{"key":"1293_CR46","doi-asserted-by":"crossref","first-page":"1771","DOI":"10.1162\/089976602760128018","volume":"14","author":"GE Hinton","year":"2002","unstructured":"Hinton, G. E. (2002). Training products of experts by minimizing contrastive divergence. Neural Computation, 14, 1771\u20131800.","journal-title":"Neural Computation"},{"key":"1293_CR47","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4615-3210-1","volume-title":"Connectionist speech recognition","author":"H Bourlard","year":"1994","unstructured":"Bourlard, H., & Morgan, N. (1994). Connectionist speech recognition. Dordrecht: Kluwer Academic Publishers."},{"key":"1293_CR48","volume-title":"Nonlinear Programming","author":"DP Bertsekas","year":"1999","unstructured":"Bertsekas, D. P. (1999). Nonlinear Programming (2nd ed.). Belmont: Athena Scientific.","edition":"2"},{"issue":"6088","key":"1293_CR49","doi-asserted-by":"crossref","first-page":"533","DOI":"10.1038\/323533a0","volume":"323","author":"D-E Rumelhart","year":"1986","unstructured":"Rumelhart, D.-E., Hinton, G.-E., & Williams, R.-J. (1986). Learning representations by back-propagating errors. Nature, 323(6088), 533\u2013536.","journal-title":"Nature"},{"key":"1293_CR50","unstructured":"Zheng, Y. B., Wen, Z. Q., Liu, B., Li, Y. & Tao, J. H. (2016). An Initial Research Towards Accurate Pitch Extraction for Speech Synthesis based on Bidirectional Long Short-Term Memory Recurrent Neural Network. In Proc. of ICSP."},{"key":"1293_CR51","doi-asserted-by":"crossref","unstructured":"Su, H., Zhang, H., Zhang, X. L., & Gao, G. G. (2016). Convolutional Neural Network for Robust Pitch Determination. In Proc. of ICASSP, pp. 579\u2013583.","DOI":"10.1109\/ICASSP.2016.7471741"},{"issue":"5","key":"1293_CR52","doi-asserted-by":"crossref","first-page":"187","DOI":"10.1016\/S0167-6393(98)00085-5","volume":"27","author":"H Kawahara","year":"1999","unstructured":"Kawahara, H., Masuda-Katsuse, I., & de Cheveign\u00e9, A. (1999). Restructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency-based F0 extraction: Possible role of a repetitive structure in sounds. Speech Communication, 27(5), 187\u2013207.","journal-title":"Speech Communication"},{"key":"1293_CR53","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., Silovsky, J., Stemmer, G., & Vesely, K. (2011) The Kaldi Speech Recognition Toolkit. In Proc. IEEE Workshop on Automatic Speech Recognition and Understanding."},{"issue":"2","key":"1293_CR54","doi-asserted-by":"crossref","first-page":"443","DOI":"10.1109\/TASSP.1985.1164550","volume":"ASSp-33","author":"Y Ephraim","year":"1985","unstructured":"Ephraim, Y., & Malah, D. (1985). Speech Enhancement using a Minimum Mean-Square Error Log-Spectral Amplitude Estimator. IEEE Transactions on Acoustics, Speech, and Signal Processing, ASSp-33(2), 443\u2013445.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"1293_CR55","unstructured":"Blin, L., Boeffard, O. & Barreaud, V. (2008). WEB-based listening test system for speech synthesis and speech conversion evaluation. In Proc. of LREC (Marrakech (Morocco))."}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-017-1293-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-017-1293-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-017-1293-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T23:39:22Z","timestamp":1750894762000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-017-1293-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,10,2]]},"references-count":55,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2018,7]]}},"alternative-id":["1293"],"URL":"https:\/\/doi.org\/10.1007\/s11265-017-1293-z","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"type":"print","value":"1939-8018"},{"type":"electronic","value":"1939-8115"}],"subject":[],"published":{"date-parts":[[2017,10,2]]}}}