{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T15:40:04Z","timestamp":1717256404123},"reference-count":36,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2014,8,17]],"date-time":"2014-08-17T00:00:00Z","timestamp":1408233600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2015,3]]},"DOI":"10.1007\/s10772-014-9247-3","type":"journal-article","created":{"date-parts":[[2014,8,16]],"date-time":"2014-08-16T12:00:18Z","timestamp":1408190418000},"page":"25-36","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["$$\\hbox {F}_{0}$$ F 0 contour generation and synthesis using Bengali Hmm-based speech synthesis system"],"prefix":"10.1007","volume":"18","author":[{"given":"Sankar","family":"Mukherjee","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shyamal Kumar Das","family":"Mandal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,8,17]]},"reference":[{"key":"9247_CR1","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.1109\/ICSLP.1996.607807","volume":"2","author":"T Anastasakos","year":"1996","unstructured":"Anastasakos, T., McDonough, J., Schwartz, R., & Makhoul, J. (1996). A compact model for speaker-adaptive training. Proceedings of Fourth International Conference on Spoken Language Processing, 2, 1137\u20131140.","journal-title":"Proceedings of Fourth International Conference on Spoken Language Processing"},{"key":"9247_CR2","doi-asserted-by":"crossref","unstructured":"Black, A. (2002). Perfect synthesis for all of the people all of the time. In Proceedings of IEEE Speech Synthesis Workshop.","DOI":"10.1109\/WSS.2002.1224400"},{"key":"9247_CR3","doi-asserted-by":"crossref","unstructured":"Black, A., & Taylor, P. (1997). Automatically clustering similar units for unit selection in speech synthesis. In Proceedings of Eurospeech (pp. 601\u2013604).","DOI":"10.21437\/Eurospeech.1997-219"},{"key":"9247_CR4","doi-asserted-by":"crossref","unstructured":"Bulyko, I., Ostendorf, M., & Bilmes, J. (2002). Robust splicing costs and efficient search with BMM models for concatenative speech synthesis. In Proceedings of ICASSP (pp. 461\u2013464).","DOI":"10.1109\/ICASSP.2002.1005776"},{"key":"9247_CR5","first-page":"279","volume":"3","author":"N Campbell","year":"1996","unstructured":"Campbell, N., & Black, A. (1996). Prosody and the selection of source units for concatenative synthesis. Progress in Speech Synthesis, 3, 279\u2013292.","journal-title":"Progress in Speech Synthesis"},{"key":"9247_CR6","doi-asserted-by":"crossref","unstructured":"Chen, C.-P., Huang, Y.-C., Wu, C.-H. & Lee, K.-D. (2012). Cross-lingual frame selection method for polyglot speech synthesis. In Proceeding of IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 4521\u20134524).","DOI":"10.1109\/ICASSP.2012.6288923"},{"key":"9247_CR7","doi-asserted-by":"crossref","unstructured":"Dines, J., & Sridharan, S. (2001). Trainable speech synthesis with trended hidden Markov models. In Proceedings of ICASSP (pp. 833\u2013837).","DOI":"10.1109\/ICASSP.2001.941044"},{"key":"9247_CR8","doi-asserted-by":"crossref","unstructured":"Donovan, R., & Woodland, P. (1995). Improvements in an HMM-based speech synthesiser. In Proceedings of Eurospeech (pp. 573\u2013576).","DOI":"10.21437\/Eurospeech.1995-146"},{"issue":"4","key":"9247_CR9","doi-asserted-by":"crossref","first-page":"233","DOI":"10.1250\/ast.5.233","volume":"5","author":"H Fujisaki","year":"1984","unstructured":"Fujisaki, H., & Hirose, K. (1984). Analysis of voice fundamental frequency contours for declarative sentences of japanese. Journal of the Acoustical Society of Japan (E), 5(4), 233\u2013242.","journal-title":"Journal of the Acoustical Society of Japan (E)"},{"key":"9247_CR10","first-page":"137","volume":"1","author":"T Fukada","year":"1992","unstructured":"Fukada, T., Tokuda, K., Kobayashi, T., & Imai, S. (1992). An adaptive algorithm for mel-cepstral analysis of speech. Proceeding of IEEE International Conference on Acoustics, Speech, and Signal Processing, 1, 137\u2013140.","journal-title":"Proceeding of IEEE International Conference on Acoustics, Speech, and Signal Processing"},{"key":"9247_CR11","doi-asserted-by":"crossref","unstructured":"Gao, B.-H., Qian, Y., Wu, Z.-Z., & Soong, F.-K. (2008). Duration refinement by jointly optimizing state and longer unit likelihood. In Proceedings of Interspeech (pp. 2266\u20132269).","DOI":"10.21437\/Interspeech.2008-556"},{"issue":"9","key":"9247_CR12","doi-asserted-by":"crossref","first-page":"1245","DOI":"10.1109\/TC.2007.1079","volume":"56","author":"C-C Hsia","year":"2007","unstructured":"Hsia, C.-C., Wu, C.-H., & Wu, J.-Q. (2007). Conversion function clustering and selection using linguistic and spectral information for emotional voice conversion. IEEE Transactions on Computers, 56(9), 1245\u20131254.","journal-title":"IEEE Transactions on Computers"},{"key":"9247_CR13","doi-asserted-by":"crossref","unstructured":"Hunt, A., & Black, A. (1996). Unit selection in a concatenative speech synthesis system using a large speech database. In Proceedings of ICASSP (pp. 373\u2013376).","DOI":"10.1109\/ICASSP.1996.541110"},{"key":"9247_CR14","doi-asserted-by":"crossref","unstructured":"Kawai, H., & Tsuzaki, M. (2002). A study on time-dependent voice quality variation in a large-scale single speaker speech corpus used for speech synthesis. In Proceedings of IEEE Speech Synthesis Workshop.","DOI":"10.1109\/WSS.2002.1224362"},{"key":"9247_CR15","doi-asserted-by":"crossref","unstructured":"Ling, Z.-H., Wu, Y.-J., Wang, Y.-P., Qin, L., & Wang, R.-H. (2006). USTC system for Blizzard challenge 2006\u2014an improved HMM-based speech synthesis method. In Proceedings of Blizzard Challenge Workshop.","DOI":"10.21437\/Blizzard.2006-6"},{"key":"9247_CR16","first-page":"49","volume":"6","author":"SD Mandal","year":"2005","unstructured":"Mandal, S. D., Saha, A., & Datta, A. (2005). Annotated speech corpora development in Indian languages. Vishwa Bharat, 6, 49\u201364.","journal-title":"Vishwa Bharat"},{"key":"9247_CR17","unstructured":"Mandal, S. D., Warsi, A. H., Basu, T., Hirose, K., & Fujisaki, H. (2010). Analysis and synthesis of $$\\text{ F }_{0}$$ F 0 contours for bangla readout speech. In Proceedings of Oriental COCOSDA, Kathmandu, Nepal."},{"key":"9247_CR18","unstructured":"Mukherjee, S. & Mandal, S. D. (2012). A Bengali HMM based speech synthesis system. In Proceeding of International Conference on Speech Database and Assessments (Oriental COCOSDA) (pp. 225\u2013259)."},{"key":"9247_CR19","doi-asserted-by":"crossref","unstructured":"Mukherjee, S. & Mandal, S. D. (2013). Bengali parts-of-speech tagging using global linear model. In Proceeding of IEEE INDICON -2013.","DOI":"10.1109\/INDCON.2013.6726132"},{"key":"9247_CR20","unstructured":"Oura, K., Zen, H., Nankaku, Y., Lee, A., & Tokuda, K. (2007). Postfiltering for HMM-based speech synthesis using mel-LSPs. In Proceedings of Autumn Meeting of ASJ (pp. 367\u2013368) (in Japanese)."},{"key":"9247_CR21","doi-asserted-by":"crossref","DOI":"10.1037\/e526112012-054","volume-title":"Affective computing","author":"RW Picard","year":"1997","unstructured":"Picard, R. W. (1997). Affective computing. Cambridge: MIT Press."},{"key":"9247_CR22","doi-asserted-by":"crossref","unstructured":"Qian, Y., Xu, J., & Soong, F. K. (2011). A frame mapping based HMM approach to cross-lingual voice transformation. In Proceeding of IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 5120\u20135123).","DOI":"10.1109\/ICASSP.2011.5947509"},{"key":"9247_CR23","doi-asserted-by":"crossref","unstructured":"Shi, Y., Chang, E., Peng, H., & Chu, M. (2002). Power spectral density based channel equalization of large speech database for concatenative TTS system. In Proceedings of ICSLP (pp. 2369\u20132372).","DOI":"10.21437\/ICSLP.2002-97"},{"key":"9247_CR24","doi-asserted-by":"crossref","unstructured":"Stylianou, Y. (1999). Assessment and correction of voice quality variabilities in large speech databases for concatenative speech synthesis. In Proceedings of ICASSP (pp. 377\u2013380).","DOI":"10.1109\/ICASSP.1999.758141"},{"key":"9247_CR25","unstructured":"Sun, J.-W., Ding, F., & Wu, Y.-H. (2009). Polynomial segment model based statistical parametric speech synthesis system. In Proceedings of ICASSP (pp. 4021\u20134024)."},{"issue":"5","key":"9247_CR26","doi-asserted-by":"crossref","first-page":"816","DOI":"10.1093\/ietisy\/e90-d.5.816","volume":"90","author":"T Toda","year":"2007","unstructured":"Toda, T., & Tokuda, K. (2007). A speech parameter generation algorithm considering global variance for HMM-based speech synthesis. IEICE Transactions on Information and Systems E Series D, 90(5), 816\u2013824.","journal-title":"IEICE Transactions on Information and Systems E Series D"},{"key":"9247_CR27","first-page":"1315","volume":"3","author":"K Tokuda","year":"2000","unstructured":"Tokuda, K., Yoshimura, T., Masuko, T., Kobayashi, T., & Kitamura, T. (2000). Speech parameter generation algorithms for HMM-based speech synthesis. Proceeding of IEEE International Conference on Acoustics, Speech, and Signal Processing, 3, 1315\u20131318.","journal-title":"Proceeding of IEEE International Conference on Acoustics, Speech, and Signal Processing"},{"key":"9247_CR28","unstructured":"Tokuda, K., Zen, H., & Black, A.W. (2002). An HMM-based speech synthesis system applied to english. In Proceedings of IEEE Workshop on Speech Synthesis (pp. 227\u2013230)."},{"key":"9247_CR29","unstructured":"Tseng, C.-Y. (2006). Higher level organization and discourse prosody. The Second International Symposium on Tonal Aspects of Languages (pp. 23\u201334)."},{"key":"9247_CR30","unstructured":"Warsi, A. H., Basu, T., Hirose, K., & Fujisaki, H. (2012). Analysis and synthesis of $$\\text{ F }_{0}$$ F 0 contours of declarative, interrogative, and imperative utterances of bangla. In Proceeding of International Conference on Speech Database and Assessments (Oriental COCOSDA) (pp. 56\u201361)."},{"key":"9247_CR31","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Tokuda, K., Masuko, T., Kobayashi, T., & Kitamura, T. (1998). Duration modeling for HMM-based speech synthesis. In Proceedings of ICSLP (pp. 29\u201332).","DOI":"10.21437\/ICSLP.1998-6"},{"key":"9247_CR32","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Tokuda, K., Masuko, T., Kobayashi, T., & Kitamura, T. (1999). Simultaneous modeling of spectrum, pitch and duration in HMM-based speech synthesis. In Proceedings of Eurospeech (pp. 2347\u20132350).","DOI":"10.21437\/Eurospeech.1999-513"},{"key":"9247_CR33","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Tokuda, K., Masuko, T., Kobayashi, T., & Kitamura, T. (2001). Mixed excitation for HMM-based speech synthesis. In Proceedings of Eurospeech (pp. 2263\u20132266).","DOI":"10.21437\/Eurospeech.2001-539"},{"key":"9247_CR34","doi-asserted-by":"crossref","unstructured":"Zen, H., Toda, T., & Tokuda, K. (2006). The Nitech-NAIST HMM-based speech synthesis system for the Blizzard challenge 2006. In Proceedings of Blizzard Challenge Workshop.","DOI":"10.21437\/Blizzard.2006-3"},{"issue":"5","key":"9247_CR35","doi-asserted-by":"crossref","first-page":"825","DOI":"10.1093\/ietisy\/e90-d.5.825","volume":"90","author":"H Zen","year":"2007","unstructured":"Zen, H., Tokuda, K., Masuko, T., Kobayashi, T., & Kitamura, T. (2007a). A hidden semi-Markov model-based speech synthesis system. IEICE Transactions on Information and Systems E Series D, 90(5), 825\u2013834.","journal-title":"IEICE Transactions on Information and Systems E Series D"},{"key":"9247_CR36","unstructured":"Zen, H., Nose, T., Yamagishi, J., Sako, S., Masuko, T., Black, A. W., & Tokuda, K. (2007b). The HMM-based speech synthesis system (hts) version 2.0. In Proceedings of Sixth ISCA Workshop on Speech Synthesis (pp. 294\u2013299)."}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-014-9247-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-014-9247-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-014-9247-3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T15:02:31Z","timestamp":1717254151000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-014-9247-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,8,17]]},"references-count":36,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2015,3]]}},"alternative-id":["9247"],"URL":"https:\/\/doi.org\/10.1007\/s10772-014-9247-3","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,8,17]]}}}