{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,10]],"date-time":"2025-06-10T22:40:03Z","timestamp":1749595203420,"version":"3.41.0"},"reference-count":36,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"10","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2016]]},"DOI":"10.1587\/transinf.2016slp0011","type":"journal-article","created":{"date-parts":[[2016,9,30]],"date-time":"2016-09-30T22:23:29Z","timestamp":1475274209000},"page":"2471-2480","source":"Crossref","is-referenced-by-count":1,"title":["Investigation of Using Continuous Representation of Various Linguistic Units in Neural Network Based Text-to-Speech Synthesis"],"prefix":"10.1587","volume":"E99.D","author":[{"given":"Xin","family":"WANG","sequence":"first","affiliation":[{"name":"National Institute of Informatics"},{"name":"SOKENDAI"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"TAKAKI","sequence":"additional","affiliation":[{"name":"National Institute of Informatics"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junichi","family":"YAMAGISHI","sequence":"additional","affiliation":[{"name":"National Institute of Informatics"},{"name":"SOKENDAI"},{"name":"Centre for Speech Technology Research (CSTR), University of Edinburgh"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"doi-asserted-by":"crossref","unstructured":"[1] P. Taylor, Text-to-Speech Synthesis, Cambridge University Press, 2009.","key":"1","DOI":"10.1017\/CBO9780511816338"},{"unstructured":"[2] R. Collobert, J. Weston, L. Bottou, M. Karlen, K. Kavukcuoglu, and P. Kuksa, \u201cNatural language processing (almost) from scratch,\u201d J. Mach. Learn. Res., vol.12, pp.2493-2537, 2011.","key":"2"},{"unstructured":"[3] T. Mikolov, W.t. Yih, and G. Zweig, \u201cLinguistic regularities in continuous space word representations,\u201d HLT-NAACL, pp.746-751, 2013.","key":"3"},{"unstructured":"[4] T. Mikolov, I. Sutskever, K. Chen, G.S. Corrado, and J. Dean, \u201cDistributed representations of words and phrases and their compositionality,\u201d NIPS-2013, pp.3111-3119, 2013.","key":"4"},{"unstructured":"[5] P. Wang, Y. Qian, F.K. Soong, L. He, and H. Zhao, \u201cWord embedding for recurrent neural network based tts synthesis,\u201d ICASSP-2015, pp.4879-4883, IEEE, 2015.","key":"5"},{"unstructured":"[6] P. Zhu, L. Xie, and Y. Chen, \u201cArticulatory movement prediction using deep bidirectional long short-term memory based recurrent neural networks and word\/phone embeddings,\u201d INTERSPEECH-2015, pp.2192-2196, 2015.","key":"6"},{"unstructured":"[7] M.A.K. Halliday, An Introduction to Functional Grammar, 2nd ed., Edward Arnold, London, UK, 1994.","key":"7"},{"doi-asserted-by":"crossref","unstructured":"[8] A.J. Hunt and A.W. Black, \u201cUnit selection in a concatenative speech synthesis system using a large speech database,\u201d ICASSP-1996, pp.373-376, IEEE, 1996.","key":"8","DOI":"10.1109\/ICASSP.1996.541110"},{"doi-asserted-by":"crossref","unstructured":"[9] K. Tokuda, Y. Nankaku, T. Toda, H. Zen, J. Yamagishi, and K. Oura, \u201cSpeech synthesis based on hidden Markov models,\u201d Proc. IEEE, vol.101, no.5, pp.1234-1252, 2013.","key":"9","DOI":"10.1109\/JPROC.2013.2251852"},{"unstructured":"[10] P. Taylor, \u201cHidden markov models for grapheme to phoneme conversion,\u201d INTERSPEECH-2005, pp.1973-1976, 2005.","key":"10"},{"unstructured":"[11] A.W. Black, K. Lenzo, and V. Pagel, \u201cIssues in building general letter to sound rules,\u201d SSW3-1998, 1998.","key":"11"},{"doi-asserted-by":"crossref","unstructured":"[12] Y. Xu, A. Lee, S. Prom-on, and F. Liu, \u201cExplaining the penta model: a reply to arvaniti and ladd,\u201d Phonology, vol.32, no.3, pp.505-535, 2015.","key":"12","DOI":"10.1017\/S0952675715000299"},{"doi-asserted-by":"crossref","unstructured":"[13] K.E.A. Silverman, M.E. Beckman, J.F. Pitrelli, M. Ostendorf, C.W. Wightman, P. Price, J.B. Pierrehumbert, and J. Hirschberg, \u201cTOBI: a standard for labeling English prosody,\u201d ICSLP-1992, pp.867-870, 1992.","key":"13","DOI":"10.21437\/ICSLP.1992-260"},{"doi-asserted-by":"crossref","unstructured":"[14] J. Hirschberg, \u201cPitch accent in context predicting intonational prominence from text,\u201d Artificial Intelligence, vol.63, no.1, pp.305-340, 1993.","key":"14","DOI":"10.1016\/0004-3702(93)90020-C"},{"doi-asserted-by":"crossref","unstructured":"[15] J. Kupiec, \u201cRobust part-of-speech tagging using a hidden markov model,\u201d Computer Speech &amp; Language, vol.6, no.3, pp.225-242, 1992.","key":"15","DOI":"10.1016\/0885-2308(92)90019-Z"},{"doi-asserted-by":"crossref","unstructured":"[16] M. Collins, \u201cHead-driven statistical models for natural language parsing,\u201d Computational linguistics, vol.29, no.4, pp.589-637, 2003.","key":"16","DOI":"10.1162\/089120103322753356"},{"unstructured":"[17] M. Ostendorf, P.J. Price, and S. Shattuck-Hufnagel, \u201cThe boston university radio news corpus,\u201d Linguistic Data Consortium, pp.1-19, 1995.","key":"17"},{"doi-asserted-by":"crossref","unstructured":"[18] M.P. Marcus, M.A. Marcinkiewicz, and B. Santorini, \u201cBuilding a large annotated corpus of english: The penn treebank,\u201d Computational linguistics, vol.19, no.2, pp.313-330, 1993.","key":"18","DOI":"10.21236\/ADA273556"},{"doi-asserted-by":"crossref","unstructured":"[19] D. Hirst, \u201cThe phonology and phonetics of speech prosody: between acoustics and interpretation,\u201d Speech Prosody, pp.163-170, 2004.","key":"19","DOI":"10.21437\/SpeechProsody.2004-39"},{"doi-asserted-by":"crossref","unstructured":"[20] E. Shriberg and A. Stolcke, \u201cProsody modeling for automatic speech recognition and understanding,\u201d in Mathematical Foundations of Speech and Language Processing, vol.138, pp.105-114, Springer, 2004.","key":"20","DOI":"10.1007\/978-1-4419-9017-4_5"},{"doi-asserted-by":"crossref","unstructured":"[21] A. Batliner and B. M\u00f6bius, \u201cProsodic models, automatic speech understanding, and speech synthesis: Towards the common ground?,\u201d in The integration of phonetic knowledge in speech technology, vol.25, pp.21-44, Springer, 2005.","key":"21","DOI":"10.1007\/1-4020-2637-4_3"},{"doi-asserted-by":"crossref","unstructured":"[22] C.W. Wightman, \u201cToBI or not ToBI?,\u201d Speech Prosody, pp.25-29, 2002.","key":"22","DOI":"10.21437\/SpeechProsody.2002-4"},{"unstructured":"[23] H. Zen, A. Senior, and M. Schuster, \u201cStatistical parametric speech synthesis using deep neural networks,\u201d ICASSP-2013, pp.7962-7966, IEEE, 2013.","key":"23"},{"unstructured":"[24] A. Graves, Supervised sequence labelling with recurrent neural networks, Ph.D. Thesis, Technische Universit\u00e4t M\u00fcnchen, 2008.","key":"24"},{"doi-asserted-by":"crossref","unstructured":"[25] Y. Bengio, \u201cLearning deep architectures for ai,\u201d Found. Trends Mach. Learn., vol.2, no.1, pp.1-127, Jan. 2009.","key":"25","DOI":"10.1561\/2200000006"},{"unstructured":"[26] J. Bian, B. Gao, and T.-Y. Liu, \u201cKnowledge-powered deep learning for word embedding,\u201d in Machine Learning and Knowledge Discovery in Databases, vol.8724, pp.132-148, Springer, 2014.","key":"26"},{"unstructured":"[27] Q. Le and T. Mikolov, \u201cDistributed representations of sentences and documents,\u201d ICML-14, pp.1188-1196, 2014.","key":"27"},{"unstructured":"[28] O.S. Watts, Unsupervised learning for text-to-speech synthesis, Ph.D. Thesis, University of Edinburgh, 2013.","key":"28"},{"doi-asserted-by":"crossref","unstructured":"[29] D.R. Ladd, M.E. Beckman, and J.B. Pierrehumbert, \u201cIntonational structure in Japanese and English,\u201d Phonology, vol.3, no.01, pp.255-309, 1986.","key":"29","DOI":"10.1017\/S095267570000066X"},{"unstructured":"[30] HTS Working Group, \u201cThe English TTS System \u201dFlite+hts_engine\u201d,\u201d 2014.","key":"30"},{"doi-asserted-by":"crossref","unstructured":"[31] O. Levy, Y. Goldberg, and I. Dagan, \u201cImproving distributional similarity with lessons learned from word embeddings,\u201d Transactions of the Association for Computational Linguistics, vol.3, pp.211-225, 2015.","key":"31","DOI":"10.1162\/tacl_a_00134"},{"unstructured":"[32] J. Turian, L. Ratinov, and Y. Bengio, \u201cWord representations: a simple and general method for semi-supervised learning,\u201d Proc. 48th Annual Meeting of the Association for Computational Linguistics, pp.384-394, 2010.","key":"32"},{"doi-asserted-by":"crossref","unstructured":"[33] H. Kawahara, I. Masuda-Katsuse, and A. de Cheveign\u00e9, \u201cRestructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency-based F0 extraction: Possible role of a repetitive structure in sounds,\u201d Speech Communication, vol.27, no.3-4, pp.187-207, 1999.","key":"33","DOI":"10.1016\/S0167-6393(98)00085-5"},{"unstructured":"[34] D. O&apos;Shaughnessy, Speech communications: human and machine, Institute of Electrical and Electronics Engineers, 2000.","key":"34"},{"doi-asserted-by":"crossref","unstructured":"[35] D. Rubinstein, E. Levi, R. Schwartz, and A. Rappoport, \u201cHow well do distributional models capture different types of semantic knowledge?,\u201d Proc. ACL, pp.726-730, 2015.","key":"35","DOI":"10.3115\/v1\/P15-2119"},{"doi-asserted-by":"crossref","unstructured":"[36] C. Xu, Y. Bai, J. Bian, B. Gao, G. Wang, X. Liu, and T.-Y. Liu, \u201cRc-net: A general framework for incorporating knowledge into word representations,\u201d CIKM-14, pp.1219-1228, 2014.","key":"36","DOI":"10.1145\/2661829.2662038"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E99.D\/10\/E99.D_2016SLP0011\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,10]],"date-time":"2025-06-10T22:12:56Z","timestamp":1749593576000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E99.D\/10\/E99.D_2016SLP0011\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"references-count":36,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2016]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2016slp0011","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2016]]}}}