{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,16]],"date-time":"2026-03-16T10:10:51Z","timestamp":1773655851022,"version":"3.50.1"},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2015,11,21]],"date-time":"2015-11-21T00:00:00Z","timestamp":1448064000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2016,3]]},"DOI":"10.1007\/s10772-015-9320-6","type":"journal-article","created":{"date-parts":[[2015,11,21]],"date-time":"2015-11-21T09:52:52Z","timestamp":1448099572000},"page":"9-18","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":26,"title":["Automatic speech segmentation in syllable centric speech recognition system"],"prefix":"10.1007","volume":"19","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2464-4217","authenticated-orcid":false,"given":"Soumya Priyadarsini","family":"Panda","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ajit Kumar","family":"Nayak","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,11,21]]},"reference":[{"key":"9320_CR1","doi-asserted-by":"crossref","first-page":"85","DOI":"10.1016\/j.specom.2013.07.008","volume":"56","author":"L Besacier","year":"2014","unstructured":"Besacier, L., Barnard, E., Karpov, A., & Schultz, T. (2014). Automatic speech recognition for under-resourced languages: A survey. Speech Communication, 56, 85\u2013100.","journal-title":"Speech Communication"},{"issue":"4","key":"9320_CR2","doi-asserted-by":"crossref","first-page":"653","DOI":"10.1109\/TCE.2014.7027339","volume":"60","author":"J Ga\u0142ka","year":"2014","unstructured":"Ga\u0142ka, J., Masior, M., & Salasa, M. (2014). Voice authentication embedded solution for secured access control. IEEE Transactions on Consumer Electronics, 60(4), 653\u2013661.","journal-title":"IEEE Transactions on Consumer Electronics"},{"key":"9320_CR3","doi-asserted-by":"crossref","first-page":"109","DOI":"10.1016\/j.dsp.2014.06.002","volume":"32","author":"Y He","year":"2014","unstructured":"He, Y., Han, J., Zheng, T., & Sun, G. (2014). A new framework for robust speech recognition in complex channel environments. Digital Signal Processing, 32, 109\u2013123.","journal-title":"Digital Signal Processing"},{"issue":"1","key":"9320_CR4","doi-asserted-by":"crossref","first-page":"96","DOI":"10.1109\/TASSP.1986.1164784","volume":"34","author":"SM Kay","year":"1986","unstructured":"Kay, S. M., & Sudhaker, R. (1986). A zero crossing-based spectrum analyzer. IEEE Transactions on Acoustics, Speech, and Signal Processing, 34(1), 96\u2013104.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"issue":"5","key":"9320_CR5","doi-asserted-by":"crossref","first-page":"1068","DOI":"10.1016\/j.csl.2012.12.005","volume":"27","author":"F Kelly","year":"2013","unstructured":"Kelly, F., Drygajlo, A., & Harte, N. (2013). Speaker verification in score-ageing-quality classification space. Computer Speech & Language, 27(5), 1068\u20131084.","journal-title":"Computer Speech & Language"},{"issue":"3","key":"9320_CR6","doi-asserted-by":"crossref","first-page":"769","DOI":"10.1016\/j.csl.2013.09.009","volume":"28","author":"N Kitaoka","year":"2014","unstructured":"Kitaoka, N., Enami, D., & Nakagawa, S. (2014). Effect of acoustic and linguistic contexts on human and machine speech recognition. Computer Speech & Language, 28(3), 769\u2013787.","journal-title":"Computer Speech & Language"},{"issue":"2","key":"9320_CR7","doi-asserted-by":"crossref","first-page":"265","DOI":"10.1007\/s10772-012-9139-3","volume":"15","author":"SG Koolagudi","year":"2012","unstructured":"Koolagudi, S. G., & Rao, K. S. (2012). Emotion recognition from speech using source, system, and prosodic features. International Journal of Speech Technology, 15(2), 265\u2013289.","journal-title":"International Journal of Speech Technology"},{"issue":"1","key":"9320_CR8","doi-asserted-by":"crossref","first-page":"320","DOI":"10.1109\/TASSP.1985.1164503","volume":"33","author":"YK Lau","year":"1985","unstructured":"Lau, Y. K., & Chan, C. K. (1985). Speech recognition based on zero crossing rate and energy. IEEE Transactions on Acoustics, Speech, and Signal Processing, 33(1), 320\u2013323.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"issue":"1","key":"9320_CR9","doi-asserted-by":"crossref","first-page":"151","DOI":"10.1016\/j.csl.2012.01.008","volume":"27","author":"M Li","year":"2013","unstructured":"Li, M., Han, K. J., & Narayanan, S. (2013). Automatic speaker age and gender recognition using acoustic and prosodic level information fusion. Computer Speech & Language, 27(1), 151\u2013167.","journal-title":"Computer Speech & Language"},{"issue":"2","key":"9320_CR10","doi-asserted-by":"crossref","first-page":"175","DOI":"10.1016\/0167-6393(95)00043-7","volume":"18","author":"CH Lin","year":"1996","unstructured":"Lin, C. H., Wu, C. H., Ting, P. Y., & Wang, H. M. (1996). Frameworks for recognition of Mandarin syllables with tones using sub-syllabic units. Speech Communication, 18(2), 175\u2013190.","journal-title":"Speech Communication"},{"issue":"1","key":"9320_CR11","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/S0167-6393(97)00021-6","volume":"22","author":"RP Lippmann","year":"1997","unstructured":"Lippmann, R. P. (1997). Speech recognition by machines and humans. Speech Communication, 22(1), 1\u201315.","journal-title":"Speech Communication"},{"issue":"8","key":"9320_CR12","doi-asserted-by":"crossref","first-page":"2203","DOI":"10.1109\/TMM.2014.2360798","volume":"16","author":"Q Mao","year":"2014","unstructured":"Mao, Q., Dong, M., Huang, Z., & Zhan, Y. (2014). Learning Salient Features for Speech Emotion Recognition Using Convolutional Neural Networks. IEEE Transactions on Multimedia, 16(8), 2203\u20132213.","journal-title":"IEEE Transactions on Multimedia"},{"issue":"9","key":"9320_CR13","doi-asserted-by":"crossref","first-page":"1424","DOI":"10.1109\/TASLP.2014.2335055","volume":"22","author":"IV McLoughlin","year":"2014","unstructured":"McLoughlin, I. V. (2014). Super-audible voice activity detection. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 22(9), 1424\u20131433.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"9320_CR14","doi-asserted-by":"crossref","unstructured":"Musfir, M., Krishnan, K. R., & Murthy, H. (2014). Analysis of fricatives, stop consonants and nasals in the automatic segmentation of speech using the group delay algorithm. In Twentieth National Conference on Communications (NCC) (pp. 1\u20136).","DOI":"10.1109\/NCC.2014.6811364"},{"key":"9320_CR15","doi-asserted-by":"crossref","unstructured":"Obin, N., Lamare, F., & Roebel, A. (2013). Syll-O-Matic: an adaptive time-frequency representation for the automatic segmentation of speech into syllables. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), (pp. 6699\u20136703).","DOI":"10.1109\/ICASSP.2013.6638958"},{"key":"9320_CR16","doi-asserted-by":"crossref","first-page":"155","DOI":"10.1016\/j.specom.2013.09.012","volume":"57","author":"A Origlia","year":"2014","unstructured":"Origlia, A., Cutugno, F., & Galat\u00e0, V. (2014). Continuous emotion recognition with phonetic syllables. Speech Communication, 57, 155\u2013169.","journal-title":"Speech Communication"},{"issue":"3","key":"9320_CR17","doi-asserted-by":"crossref","first-page":"305","DOI":"10.1007\/s10772-015-9271-y","volume":"18","author":"SP Panda","year":"2015","unstructured":"Panda, S. P., & Nayak, A. K. (2015). An efficient model for text-to-speech synthesis in Indian languages. International Journal of Speech Technology, 18(3), 305\u2013315.","journal-title":"International Journal of Speech Technology"},{"issue":"3\u20134","key":"9320_CR18","doi-asserted-by":"crossref","first-page":"170","DOI":"10.1504\/IJGUC.2015.070676","volume":"6","author":"SP Panda","year":"2015","unstructured":"Panda, S. P., Nayak, A. K., & Patnaik, S. (2015). Text-to-speech synthesis with an Indian language perspective. International Journal of Grid and Utility Computing, 6(3\u20134), 170\u2013178.","journal-title":"International Journal of Grid and Utility Computing"},{"issue":"3","key":"9320_CR19","doi-asserted-by":"crossref","first-page":"429","DOI":"10.1016\/j.specom.2003.12.002","volume":"42","author":"VK Prasad","year":"2004","unstructured":"Prasad, V. K., Nagarajan, T., & Murthy, H. A. (2004). Automatic segmentation of continuous speech using minimum phase group delay functions. Speech Communication, 42(3), 429\u2013446.","journal-title":"Speech Communication"},{"issue":"4","key":"9320_CR20","doi-asserted-by":"crossref","first-page":"556","DOI":"10.1109\/TASL.2008.2010884","volume":"17","author":"S Prasanna","year":"2009","unstructured":"Prasanna, S., Reddy, B. V. S., & Krishnamoorthy, P. (2009). Vowel onset point detection using source, spectral peaks, and modulation spectrum energies. IEEE Transactions on Audio, Speech, and Language Processing, 17(4), 556\u2013565.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9320_CR21","doi-asserted-by":"crossref","first-page":"835","DOI":"10.1109\/PGEC.1963.263565","volume":"6","author":"T Sakai","year":"1963","unstructured":"Sakai, T., & Doshita, S. (1963). The automatic speech recognition system for conversational sound. IEEE Transactions on Electronic Computers, 6, 835\u2013846.","journal-title":"IEEE Transactions on Electronic Computers"},{"key":"9320_CR22","unstructured":"Shastri, L., Chang, S., & Greenberg, S. (1999). Syllable detection and segmentation using temporal flow neural networks. In International Congress of Phonetic Sciences (pp. 1721\u20131724)."},{"issue":"3","key":"9320_CR23","doi-asserted-by":"crossref","first-page":"427","DOI":"10.1016\/S0167-6393(02)00012-2","volume":"38","author":"J Sirigos","year":"2002","unstructured":"Sirigos, J., Fakotakis, N., & Kokkinakis, G. (2002). A hybrid syllable recognition system based on vowel spotting. Speech Communication, 38(3), 427\u2013440.","journal-title":"Speech Communication"},{"issue":"2","key":"9320_CR24","doi-asserted-by":"crossref","first-page":"282","DOI":"10.1109\/78.124939","volume":"40","author":"TV Sreenivas","year":"1992","unstructured":"Sreenivas, T. V., & Niederjohn, R. J. (1992). Zero-crossing based spectral analysis and SVD spectral analysis for formant frequency estimation in noise. IEEE Transactions on Signal Processing, 40(2), 282\u2013293.","journal-title":"IEEE Transactions on Signal Processing"},{"issue":"1","key":"9320_CR25","doi-asserted-by":"crossref","first-page":"49","DOI":"10.1016\/S0167-6393(00)00023-6","volume":"32","author":"HM Wang","year":"2000","unstructured":"Wang, H. M. (2000). Experiments in syllable-based retrieval of broadcast news speech in Mandarin Chinese. Speech Communication, 32(1), 49\u201360.","journal-title":"Speech Communication"},{"issue":"11","key":"9320_CR26","doi-asserted-by":"crossref","first-page":"1660","DOI":"10.1109\/TASLP.2014.2344855","volume":"22","author":"G Wang","year":"2014","unstructured":"Wang, G., & Sim, K. C. (2014). Regression-based context-dependent modeling of deep neural networks for speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 22(11), 1660\u20131669.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"9320_CR27","unstructured":"Zhao, X., & Shaughnessy, D. O. (2008). A new hybrid approach for automatic speech signal segmentation using silence signal detection, energy convex hull, and spectral variation. In Canadian Conference on Electrical and Computer Engineering (pp. 145\u2013148)."},{"key":"9320_CR28","unstructured":"Ziolko, B., Manandhar, S., Wilson, R. C., & Ziolko, M. (2006). Wavelet method of speech segmentation. In 14th European Signal Processing Conference (pp. 1\u20135)."}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-015-9320-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-015-9320-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-015-9320-6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,5,30]],"date-time":"2019-05-30T20:02:48Z","timestamp":1559246568000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-015-9320-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,11,21]]},"references-count":28,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2016,3]]}},"alternative-id":["9320"],"URL":"https:\/\/doi.org\/10.1007\/s10772-015-9320-6","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,11,21]]}}}