{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:27:22Z","timestamp":1740122842161,"version":"3.37.3"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s10772-024-10117-5","type":"journal-article","created":{"date-parts":[[2024,6,25]],"date-time":"2024-06-25T18:08:15Z","timestamp":1719338895000},"page":"425-436","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Temporal feature-based approaches for enhancing phoneme boundary detection and masking in speech"],"prefix":"10.1007","volume":"27","author":[{"given":"Shaik Mulla","family":"Shabber","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0524-8251","authenticated-orcid":false,"given":"Mohan","family":"Bansal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,6,25]]},"reference":[{"issue":"6","key":"10117_CR1","first-page":"102","volume":"12","author":"MJ Anwar","year":"2006","unstructured":"Anwar, M. J., Awais, M., Masud, S., et al. (2006). Automatic Arabic speech segmentation system. International Journal of Information Technology, 12(6), 102\u2013111.","journal-title":"International Journal of Information Technology"},{"issue":"3","key":"10117_CR2","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1109\/TASSP.1976.1162800","volume":"24","author":"B Atal","year":"1976","unstructured":"Atal, B., & Rabiner, L. (1976). A pattern recognition approach to voiced-unvoiced-silence classification with applications to speech recognition. IEEE Transactions on Acoustics, Speech, and Signal Processing, 24(3), 201\u2013212.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"10117_CR3","first-page":"279","volume-title":"Voiced\/unvoiced decision for speech signals based on zero-crossing rate and energy","author":"R Bachu","year":"2010","unstructured":"Bachu, R., Kopparthi, S., Adapa, B., et al. (2010). Voiced\/unvoiced decision for speech signals based on zero-crossing rate and energy (pp. 279\u2013282). Springer."},{"key":"10117_CR4","doi-asserted-by":"crossref","unstructured":"Ball, M. J., & Rahilly, J. (2014). Phonetics: The science of speech. Routledge.","DOI":"10.4324\/9780203767252"},{"key":"10117_CR5","doi-asserted-by":"publisher","first-page":"783","DOI":"10.1007\/s10772-018-9542-5","volume":"21","author":"M Bansal","year":"2018","unstructured":"Bansal, M., & Sircar, P. (2018). Low bit-rate speech coding based on multicomponent AFM signal model. International Journal of Speech Technology, 21, 783\u2013795.","journal-title":"International Journal of Speech Technology"},{"key":"10117_CR6","doi-asserted-by":"publisher","first-page":"4079","DOI":"10.1007\/s00034-019-01040-1","volume":"38","author":"M Bansal","year":"2019","unstructured":"Bansal, M., & Sircar, P. (2019a). A novel AFM signal model for parametric representation of speech phonemes. Circuits, Systems, and Signal Processing, 38, 4079\u20134095.","journal-title":"Circuits, Systems, and Signal Processing"},{"key":"10117_CR7","doi-asserted-by":"crossref","unstructured":"Bansal, M., & Sircar, P. (2019b). Phoneme based model for gender identification and adult-child classification. In 2019 13th international conference on signal processing and communication systems (ICSPCS) (pp. 1\u20137). IEEE.","DOI":"10.1109\/ICSPCS47537.2019.9008704"},{"key":"10117_CR8","doi-asserted-by":"crossref","unstructured":"Bansal, M., & Sircar, P. (2022). Phoneme classification using modulating features. In 2022 IEEE region 10 symposium (TENSYMP) (pp. 1\u20135). IEEE.","DOI":"10.1109\/TENSYMP54529.2022.9864425"},{"key":"10117_CR9","doi-asserted-by":"crossref","unstructured":"Benesty, J., Sondhi, M. M., & Huang, Y., et al. (2008). Springer handbook of speech processing. Springer.","DOI":"10.1007\/978-3-540-49127-9"},{"issue":"10\u201311","key":"10117_CR10","doi-asserted-by":"publisher","first-page":"763","DOI":"10.1016\/j.specom.2007.02.006","volume":"49","author":"M Benzeghiba","year":"2007","unstructured":"Benzeghiba, M., De Mori, R., Deroo, O., et al. (2007). Automatic speech recognition and speech variability: A review. Speech Communication, 49(10\u201311), 763\u2013786.","journal-title":"Speech Communication"},{"issue":"10","key":"10117_CR11","doi-asserted-by":"publisher","first-page":"5169","DOI":"10.1007\/s00034-020-01408-8","volume":"39","author":"S Bhati","year":"2020","unstructured":"Bhati, S., Nayak, S., & Kodukula, S. R. M. (2020). Unsupervised speech signal-to-symbol transformation for language identification. Circuits, Systems, and Signal Processing, 39(10), 5169\u20135197.","journal-title":"Circuits, Systems, and Signal Processing"},{"issue":"1","key":"10117_CR12","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1109\/TASLP.2015.2456421","volume":"24","author":"S Brognaux","year":"2015","unstructured":"Brognaux, S., & Drugman, T. (2015). HMM-based speech segmentation: Improvements of fully automatic approaches. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 24(1), 5\u201315.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"issue":"4","key":"10117_CR13","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1016\/0167-6393(93)90083-W","volume":"12","author":"F Brugnara","year":"1993","unstructured":"Brugnara, F., Falavigna, D., & Omologo, M. (1993). Automatic segmentation and labeling of speech based on hidden Markov models. Speech Communication, 12(4), 357\u2013370.","journal-title":"Speech Communication"},{"issue":"3","key":"10117_CR14","doi-asserted-by":"publisher","first-page":"565","DOI":"10.1044\/jshr.1403.565","volume":"14","author":"RO Coleman","year":"1971","unstructured":"Coleman, R. O. (1971). Male and female voice quality and its relationship to vowel formant frequencies. Journal of Speech and Hearing Research, 14(3), 565\u2013577.","journal-title":"Journal of Speech and Hearing Research"},{"key":"10117_CR15","doi-asserted-by":"crossref","unstructured":"Dusan, S., & Rabiner, L. (2006). On the relation between maximum spectral transition positions and phone boundaries. In Ninth international conference on spoken language processing.","DOI":"10.21437\/Interspeech.2006-230"},{"key":"10117_CR16","doi-asserted-by":"crossref","unstructured":"Dutoit, T. (1997). An introduction to text-to-speech synthesis. Springer.","DOI":"10.1007\/978-94-011-5730-8"},{"key":"10117_CR17","first-page":"261","volume-title":"International School on Neural Networks, Initiated by IIASS and EMFCSC","author":"A Esposito","year":"2004","unstructured":"Esposito, A., & Aversano, G. (2004). Text independent methods for speech segmentation. In\u00a0International School on Neural Networks, initiated by IIASS and EMFCSC (pp. 261\u2013290). Springer."},{"key":"10117_CR18","doi-asserted-by":"crossref","unstructured":"Farnetani, E., & Recasens, D. (2010). Coarticulation and connected speech processes. In The handbook of phonetic sciences (pp. 316\u2013352). Blackwell.","DOI":"10.1002\/9781444317251.ch9"},{"key":"10117_CR19","unstructured":"Franke, J., Mueller, M., Hamlaoui, F., et al. (2016). Phoneme boundary detection using deep bidirectional lstms. In Speech communication; 12. ITG Symposium, VDE (pp. 1\u20135)."},{"key":"10117_CR20","doi-asserted-by":"crossref","unstructured":"Honda, K. (2008). Physiological processes of speech production. In Springer handbook of speech processing (pp. 7\u201326) Springer.","DOI":"10.1007\/978-3-540-49127-9_2"},{"issue":"1","key":"10117_CR21","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1016\/j.specom.2007.07.001","volume":"50","author":"S Jarifi","year":"2008","unstructured":"Jarifi, S., Pastor, D., & Rosec, O. (2008). A fusion approach for automatic speech segmentation of large corpora with application to speech synthesis. Speech Communication, 50(1), 67\u201380.","journal-title":"Speech Communication"},{"key":"10117_CR22","doi-asserted-by":"crossref","unstructured":"Kaiser, J. F. (1993). Some useful properties of Teager\u2019s energy operators. In 1993 IEEE international conference on acoustics, speech, and signal processing (pp. 149\u2013152). IEEE.","DOI":"10.1109\/ICASSP.1993.319457"},{"key":"10117_CR23","doi-asserted-by":"crossref","unstructured":"Kalinli, O. (2013). Combination of auditory attention features with phone posteriors for better automatic phoneme segmentation. In INTERSPEECH (pp. 2302\u20132305).","DOI":"10.21437\/Interspeech.2013-539"},{"key":"10117_CR24","doi-asserted-by":"crossref","unstructured":"Karpagavalli, S., & Chandra, E. (2015). Phoneme and word based model for Tamil speech recognition using GMM-HMM. In 2015 international conference on advanced computing and communication systems (pp. 1\u20135). IEEE.","DOI":"10.1109\/ICACCS.2015.7324119"},{"issue":"4","key":"10117_CR25","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1016\/j.specom.2008.10.002","volume":"51","author":"J Keshet","year":"2009","unstructured":"Keshet, J., Grangier, D., & Bengio, S. (2009). Discriminative keyword spotting. Speech Communication, 51(4), 317\u2013329.","journal-title":"Speech Communication"},{"issue":"500","key":"10117_CR26","doi-asserted-by":"publisher","first-page":"1590","DOI":"10.1080\/01621459.2012.737745","volume":"107","author":"R Killick","year":"2012","unstructured":"Killick, R., Fearnhead, P., & Eckley, I. A. (2012). Optimal detection of changepoints with a linear computational cost. Journal of the American Statistical Association, 107(500), 1590\u20131598.","journal-title":"Journal of the American Statistical Association"},{"issue":"1","key":"10117_CR27","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1007\/s10772-020-09672-4","volume":"23","author":"A Koduru","year":"2020","unstructured":"Koduru, A., Valiveti, H. B., & Budati, A. K. (2020). Feature extraction algorithms to improve the speech emotion recognition rate. International Journal of Speech Technology, 23(1), 45\u201355.","journal-title":"International Journal of Speech Technology"},{"key":"10117_CR28","doi-asserted-by":"crossref","unstructured":"Kreuk, F., Sheena, Y., Keshet, J., et al. (2020). Phoneme boundary detection using learnable segmental features. In ICASSP 2020\u20132020 IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 8089\u20138093). IEEE.","DOI":"10.1109\/ICASSP40776.2020.9053053"},{"issue":"3","key":"10117_CR29","doi-asserted-by":"publisher","first-page":"855","DOI":"10.1007\/s10044-016-0591-6","volume":"20","author":"FA Laleye","year":"2017","unstructured":"Laleye, F. A., Ezin, E. C., & Motamed, C. (2017). Fuzzy-based algorithm for Fongbe continuous speech segmentation. Pattern Analysis and Applications, 20(3), 855\u2013864.","journal-title":"Pattern Analysis and Applications"},{"key":"10117_CR30","doi-asserted-by":"crossref","unstructured":"Lee, C. M., Yildirim, S., Bulut, M., et al. (2004). Emotion recognition based on phoneme classes. In Interspeech (pp. 889\u2013892).","DOI":"10.21437\/Interspeech.2004-322"},{"key":"10117_CR31","doi-asserted-by":"publisher","DOI":"10.1007\/s10772-024-10099-4","author":"HA Mait","year":"2024","unstructured":"Mait, H. A., & Aboutabit, N. (2024). Unsupervised phoneme segmentation of continuous Arabic speech. International Journal of Speech Technology. https:\/\/doi.org\/10.1007\/s10772-024-10099-4","journal-title":"International Journal of Speech Technology"},{"issue":"10","key":"10117_CR32","doi-asserted-by":"publisher","first-page":"1065","DOI":"10.1016\/j.specom.2012.05.002","volume":"54","author":"MH Moattar","year":"2012","unstructured":"Moattar, M. H., & Homayounpour, M. M. (2012). A review on speaker diarization systems and approaches. Speech Communication, 54(10), 1065\u20131103.","journal-title":"Speech Communication"},{"issue":"2","key":"10117_CR33","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1016\/j.csl.2009.04.004","volume":"24","author":"I Mporas","year":"2010","unstructured":"Mporas, I., Ganchev, T., & Fakotakis, N. (2010). Speech segmentation using regression fusion of boundary predictions. Computer Speech & Language, 24(2), 273\u2013288.","journal-title":"Computer Speech & Language"},{"issue":"4","key":"10117_CR34","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1007\/s10772-011-9110-8","volume":"14","author":"HA Patil","year":"2011","unstructured":"Patil, H. A., & Viswanath, S. (2011). Effectiveness of Teager energy operator for epoch detection from speech signals. International Journal of Speech Technology, 14(4), 321\u2013337.","journal-title":"International Journal of Speech Technology"},{"key":"10117_CR35","unstructured":"Peperkamp12, S., Pettinato, M., & Dupoux, E. (2003). Allophonic variation and the acquisition of phoneme categories. In Proceedings of the 27th annual Boston University conference on language development. Cascadilla Press."},{"key":"10117_CR36","unstructured":"Rabiner, L. R. (1978). Digital processing of speech signals, Prentice Hall google schola, 2, 601\u2013604."},{"key":"10117_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2024.103069","volume":"159","author":"K Radha","year":"2024","unstructured":"Radha, K., Bansal, M., & Pachori, R. B. (2024). Automatic speaker and age identification of children from raw speech using sincNet over ERB scale. Speech Communication, 159, 103069.","journal-title":"Speech Communication"},{"key":"10117_CR38","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.specom.2019.01.003","volume":"107","author":"PB Ramteke","year":"2019","unstructured":"Ramteke, P. B., & Koolagudi, S. G. (2019). Phoneme boundary detection from speech: A rule based approach. Speech Communication, 107, 1\u201317.","journal-title":"Speech Communication"},{"key":"10117_CR39","doi-asserted-by":"crossref","unstructured":"Ravi, K. K., & Krothapalli, S. R. (2021). Phoneme segmentation-based unsupervised pattern discovery and clustering of speech signals. In Circuits, systems, and signal processing (pp. 1\u201330).","DOI":"10.1007\/s00034-021-01876-6"},{"key":"10117_CR40","unstructured":"Rogers, M., Silverman, K., Naik, D., et al. (2013). Systems and methods for concatenation of words in text to speech synthesis. US Patent 8,396,714."},{"key":"10117_CR41","doi-asserted-by":"crossref","unstructured":"Rybach, D., Gollan, C., Schluter, R., et al. (2009). Audio segmentation for speech recognition using segment features. In 2009 IEEE international conference on acoustics, speech and signal processing (pp. 4197\u20134200). IEEE.","DOI":"10.1109\/ICASSP.2009.4960554"},{"key":"10117_CR42","doi-asserted-by":"publisher","first-page":"1346297","DOI":"10.3389\/fnhum.2024.1346297","volume":"18","author":"SM Shabber","year":"2024","unstructured":"Shabber, S. M., & Sumesh, E. P. (2024). AFM signal model for dysarthric speech classification using speech biomarkers. Frontiers in Human Neuroscience, 18, 1346297.","journal-title":"Frontiers in Human Neuroscience"},{"key":"10117_CR43","doi-asserted-by":"crossref","unstructured":"Shabber, S. M., Bansal, M., & Radha, K. (2023). Machine learning-assisted diagnosis of speech disorders: A review of dysarthric speech. In 2023 international conference on electrical, electronics, communication and computers (ELEXCOM) (pp. 1\u20136). IEEE.","DOI":"10.1109\/ELEXCOM58812.2023.10370116"},{"key":"10117_CR44","doi-asserted-by":"crossref","unstructured":"Shabber, S. M., Bansal, M., & Radha, K. (2023b). A review and classification of amyotrophic lateral sclerosis with speech as a biomarker. In 2023 14th international conference on computing communication and networking technologies (ICCCNT) (pp. 1\u20137). IEEE.","DOI":"10.1109\/ICCCNT56998.2023.10308048"},{"issue":"4","key":"10117_CR45","doi-asserted-by":"publisher","first-page":"427","DOI":"10.1016\/j.ipm.2009.03.002","volume":"45","author":"M Sokolova","year":"2009","unstructured":"Sokolova, M., & Lapalme, G. (2009). A systematic analysis of performance measures for classification tasks. Information Processing & Management, 45(4), 427\u2013437.","journal-title":"Information Processing & Management"},{"key":"10117_CR46","doi-asserted-by":"crossref","unstructured":"Svendsen, T., & Soong, F. (1987). On the automatic segmentation of speech signals. In ICASSP\u201987. IEEE international conference on acoustics, speech, and signal processing (pp. 77\u201380). IEEE.","DOI":"10.1109\/ICASSP.1987.1169628"},{"issue":"6","key":"10117_CR47","doi-asserted-by":"publisher","first-page":"617","DOI":"10.1109\/TSA.2003.813579","volume":"11","author":"DT Toledano","year":"2003","unstructured":"Toledano, D. T., G\u00f3mez, L. A. H., & Grande, L. V. (2003). Automatic phonetic segmentation. IEEE Transactions on Speech and Audio Processing, 11(6), 617\u2013625.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"10117_CR48","doi-asserted-by":"publisher","DOI":"10.1016\/j.bspc.2020.102350","volume":"65","author":"M Vashkevich","year":"2021","unstructured":"Vashkevich, M., & Rushkevich, Y. (2021). Classification of als patients based on acoustic analysis of sustained vowel phonations. Biomedical Signal Processing and Control, 65, 102350.","journal-title":"Biomedical Signal Processing and Control"},{"key":"10117_CR49","unstructured":"Wang, A., et al. (2003). An industrial strength audio search algorithm, In Ismir, 2003, (pp. 7\u201313)."},{"issue":"5","key":"10117_CR50","doi-asserted-by":"publisher","first-page":"1785","DOI":"10.1007\/s11760-022-02389-8","volume":"17","author":"P Warule","year":"2023","unstructured":"Warule, P., Mishra, S. P., & Deb, S. (2023). Significance of voiced and unvoiced speech segments for the detection of common cold. Signal, Image and Video Processing, 17(5), 1785\u20131792.","journal-title":"Signal, Image and Video Processing"},{"key":"10117_CR51","doi-asserted-by":"publisher","first-page":"3202","DOI":"10.1109\/TASLP.2021.3120632","volume":"29","author":"R Yang","year":"2021","unstructured":"Yang, R., Cheng, G., Miao, H., et al. (2021). Keyword search using attention-based end-to-end asr and frame-synchronous phoneme alignments. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 29, 3202\u20133215.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"issue":"4","key":"10117_CR52","doi-asserted-by":"publisher","first-page":"2614","DOI":"10.1121\/1.4964509","volume":"140","author":"Z Zhang","year":"2016","unstructured":"Zhang, Z. (2016). Mechanics of human voice production and control. The Journal of the Acoustical Society of America, 140(4), 2614\u20132635.","journal-title":"The Journal of the Acoustical Society of America"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10117-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10117-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10117-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T16:08:31Z","timestamp":1721664511000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10117-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6]]},"references-count":52,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["10117"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10117-5","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2024,6]]},"assertion":[{"value":"15 August 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 June 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 June 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}