{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,16]],"date-time":"2025-10-16T10:00:40Z","timestamp":1760608840687,"version":"3.40.5"},"publisher-location":"Cham","reference-count":34,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319136226"},{"type":"electronic","value":"9783319136233"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2014]]},"DOI":"10.1007\/978-3-319-13623-3_5","type":"book-chapter","created":{"date-parts":[[2014,11,14]],"date-time":"2014-11-14T10:28:38Z","timestamp":1415960918000},"page":"40-48","source":"Crossref","is-referenced-by-count":1,"title":["Statistical Text-to-Speech Synthesis of Spanish Subtitles"],"prefix":"10.1007","author":[{"given":"S.","family":"Piqueras","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"M. A.","family":"del-Agua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"A.","family":"Gim\u00e9nez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"J.","family":"Civera","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"A.","family":"Juan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"5_CR1","unstructured":"Ahocoder, http:\/\/aholab.ehu.es\/ahocoder"},{"key":"5_CR2","unstructured":"Coursera, http:\/\/www.coursera.org"},{"key":"5_CR3","unstructured":"HMM-Based Speech Synthesis System (HTS), http:\/\/hts.sp.nitech.ac.jp"},{"key":"5_CR4","unstructured":"Khan Academy, http:\/\/www.khanacademy.org"},{"key":"5_CR5","unstructured":"Axelrod, A., He, X., Gao, J.: Domain adaptation via pseudo in-domain data selection. In: Proc. of EMNLP, pp. 355\u2013362 (2011)"},{"key":"5_CR6","unstructured":"Bottou, L.: Stochastic gradient learning in neural networks. In: Proceedings of Neuro-N\u00eemes 1991. EC2, Nimes, France (1991)"},{"issue":"1","key":"5_CR7","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"G.E. Dahl","year":"2012","unstructured":"Dahl, G.E., Yu, D., Deng, L., Acero, A.: Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Transactions on Audio, Speech, and Language Processing\u00a020(1), 30\u201342 (2012)","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"issue":"2","key":"5_CR8","doi-asserted-by":"publisher","first-page":"184","DOI":"10.1109\/JSTSP.2013.2283471","volume":"8","author":"D. Erro","year":"2014","unstructured":"Erro, D., Sainz, I., Navas, E., Hernaez, I.: Harmonics plus noise model based vocoder for statistical parametric speech synthesis. IEEE Journal of Selected Topics in Signal Processing\u00a08(2), 184\u2013194 (2014)","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"5_CR9","doi-asserted-by":"crossref","unstructured":"Fan, Y., Qian, Y., Xie, F., Soong, F.: TTS synthesis with bidirectional LSTM based recurrent neural networks. In: Proc. of Interspeech (submitted 2014)","DOI":"10.21437\/Interspeech.2014-443"},{"issue":"6","key":"5_CR10","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G. Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G.E., Mohamed, A.R., Jaitly, N., Senior, A., Vanhoucke, V., Nguyen, P., Sainath, T.N., et al.: Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups. IEEE Signal Processing Magazine\u00a029(6), 82\u201397 (2012)","journal-title":"IEEE Signal Processing Magazine"},{"key":"5_CR11","doi-asserted-by":"crossref","unstructured":"Hunt, A.J., Black, A.W.: Unit selection in a concatenative speech synthesis system using a large speech database. In: Proc. of ICASSP, vol.\u00a01, pp. 373\u2013376 (1996)","DOI":"10.1109\/ICASSP.1996.541110"},{"key":"5_CR12","doi-asserted-by":"crossref","unstructured":"King, S.: Measuring a decade of progress in text-to-speech. Loquens\u00a01(1), e006 (2014)","DOI":"10.3989\/loquens.2014.006"},{"key":"5_CR13","doi-asserted-by":"crossref","unstructured":"Koehn, P.: Statistical Machine Translation. Cambridge University Press (2010)","DOI":"10.1017\/CBO9780511815829"},{"key":"5_CR14","unstructured":"Kominek, J., Schultz, T., Black, A.W.: Synthesizer voice quality of new languages calibrated with mean mel cepstral distortion. In: Proc. of SLTU, pp. 63\u201368 (2008)"},{"key":"5_CR15","doi-asserted-by":"crossref","unstructured":"Lopez, A.: Statistical machine translation. ACM Computing Surveys 40(3), 8:1\u20138:49 (2008)","DOI":"10.1145\/1380584.1380586"},{"key":"5_CR16","unstructured":"poliMedia: The polimedia video-lecture repository (2007), http:\/\/media.upv.es"},{"key":"5_CR17","unstructured":"Sainz, I., Erro, D., Navas, E., Hern\u00e1ez, I., S\u00e1nchez, J., Saratxaga, I.: Aholab speech synthesizer for albayzin 2012 speech synthesis evaluation. In: Proc. of IberSPEECH, pp. 645\u2013652 (2012)"},{"key":"5_CR18","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., Yu, D.: Feature engineering in context-dependent dnn for conversational speech transcription. In: Proc. of ASRU, pp. 24\u201329 (2011)","DOI":"10.1109\/ASRU.2011.6163899"},{"issue":"2","key":"5_CR19","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1250\/ast.21.79","volume":"21","author":"K. Shinoda","year":"2000","unstructured":"Shinoda, K., Watanabe, T.: MDL-based context-dependent subword modeling for speech recognition. Journal of the Acoustical Society of Japan\u00a021(2), 79\u201386 (2000)","journal-title":"Journal of the Acoustical Society of Japan"},{"key":"5_CR20","unstructured":"Silvestre-Cerd\u00e0, J.A., et al.: Translectures. In: Proc. of IberSPEECH, pp. 345\u2013351 (2012)"},{"key":"5_CR21","unstructured":"TED Ideas worth spreading, http:\/\/www.ted.com"},{"key":"5_CR22","unstructured":"The transLectures-UPV Team.: The transLectures-UPV toolkit (TLK), http:\/\/translectures.eu\/tlk"},{"key":"5_CR23","unstructured":"Toda, T., Black, A.W., Tokuda, K.: Mapping from articulatory movements to vocal tract spectrum with Gaussian mixture model for articulatory speech synthesis. In: Proc. of ISCA Speech Synthesis Workshop (2004)"},{"key":"5_CR24","doi-asserted-by":"crossref","unstructured":"Tokuda, K., Kobayashi, T., Imai, S.: Speech parameter generation from hmm using dynamic features. In: Proc. of ICASSP, vol.\u00a01, pp. 660\u2013663 (1995)","DOI":"10.1109\/ICASSP.1995.479684"},{"issue":"3","key":"5_CR25","first-page":"455","volume":"85","author":"K. Tokuda","year":"2002","unstructured":"Tokuda, K., Masuko, T., Miyazaki, N., Kobayashi, T.: Multi-space probability distribution HMM. IEICE Transactions on Information and Systems\u00a085(3), 455\u2013464 (2002)","journal-title":"IEICE Transactions on Information and Systems"},{"key":"5_CR26","unstructured":"transLectures: D3.1.2: Second report on massive adaptation, http:\/\/www.translectures.eu\/wp-content\/uploads\/2014\/01\/transLectures-D3.1.2-15Nov2013.pdf"},{"key":"5_CR27","unstructured":"Turr\u00f3, C., Ferrando, M., Busquets, J., Ca\u00f1ero, A.: Polimedia: a system for successful video e-learning. In: Proc. of EUNIS (2009)"},{"key":"5_CR28","unstructured":"Videolectures.NET: Exchange ideas and share knowledge, http:\/\/www.videolectures.net"},{"key":"5_CR29","doi-asserted-by":"crossref","unstructured":"Wu, Y.J., King, S., Tokuda, K.: Cross-lingual speaker adaptation for HMM-based speech synthesis. In: Proc. of ISCSLP, pp. 1\u20134 (2008)","DOI":"10.1109\/CHINSL.2008.ECP.14"},{"key":"5_CR30","unstructured":"Yamagishi, J.: An introduction to HMM-based speech synthesis. Tech. rep. Centre for Speech Technology Research (2006), https:\/\/wiki.inf.ed.ac.uk\/twiki\/pub\/CSTR\/TrajectoryModelling\/HTS-Introduction.pdf"},{"key":"5_CR31","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Tokuda, K., Masuko, T., Kobayashi, T., Kitamura, T.: Simultaneous modeling of spectrum, pitch and duration in HMM-based speech synthesis. In: Proc. of Eurospeech, pp. 2347\u20132350 (1999)","DOI":"10.21437\/Eurospeech.1999-513"},{"key":"5_CR32","doi-asserted-by":"crossref","unstructured":"Zen, H., Senior, A.: Deep mixture density networks for acoustic modeling in statistical parametric speech synthesis. In: Proc. of ICASSP, pp. 3872\u20133876 (2014)","DOI":"10.1109\/ICASSP.2014.6854321"},{"key":"5_CR33","doi-asserted-by":"crossref","unstructured":"Zen, H., Senior, A., Schuster, M.: Statistical parametric speech synthesis using deep neural networks. In: Proc. of ICASSP, pp. 7962\u20137966 (2013)","DOI":"10.1109\/ICASSP.2013.6639215"},{"issue":"11","key":"5_CR34","doi-asserted-by":"publisher","first-page":"1039","DOI":"10.1016\/j.specom.2009.04.004","volume":"51","author":"H. Zen","year":"2009","unstructured":"Zen, H., Tokuda, K., Black, A.W.: Statistical parametric speech synthesis. Speech Communication\u00a051(11), 1039\u20131064 (2009)","journal-title":"Speech Communication"}],"container-title":["Lecture Notes in Computer Science","Advances in Speech and Language Technologies for Iberian Languages"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-13623-3_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,13]],"date-time":"2025-05-13T17:52:01Z","timestamp":1747158721000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-13623-3_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014]]},"ISBN":["9783319136226","9783319136233"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-13623-3_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2014]]}}}