{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:27:52Z","timestamp":1740122872960,"version":"3.37.3"},"reference-count":70,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2018,3,9]],"date-time":"2018-03-09T00:00:00Z","timestamp":1520553600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2018,6]]},"DOI":"10.1007\/s10772-018-9498-5","type":"journal-article","created":{"date-parts":[[2018,3,9]],"date-time":"2018-03-09T13:53:13Z","timestamp":1520603593000},"page":"233-250","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Manner of articulation based Bengali phoneme classification"],"prefix":"10.1007","volume":"21","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0456-161X","authenticated-orcid":false,"given":"Tanmay","family":"Bhowmik","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shyamal Kumar Das","family":"Mandal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,3,9]]},"reference":[{"key":"9498_CR1","unstructured":"Ali, A. A., Van Der Speigel, J., & Mueller, P. (2000). Auditory-based speech processing based on the average localized synchrony detection. In Proceedings of IEEE International Conference on Acoustics, Speech, and Signal Processing ICASSP\u201900 (Vol.\u00a03, pp.\u00a01623\u20131626). IEEE."},{"key":"9498_CR2","unstructured":"Ali, A. A., Van der Spiegel, J., & Mueller, P. (1998). An acoustic-phonetic featurebased system for the automatic recognition of fricative consonants. In Proceedings of the 1998 IEEE International Conference on Acoustics, Speech and Signal Processing (Vol.\u00a02, pp.\u00a0961\u2013964). IEEE."},{"key":"9498_CR3","unstructured":"Ali, A. A., Van der Spiegel, J., Mueller, P., Haentjens, G., & Berman, J. (1999). An acoustic-phonetic feature-based system for automatic phoneme recognition in continuous speech. In Proceedings of the 1999 IEEE International Symposium on Circuits and Systems (Vol.\u00a03, pp. 118\u2013121). IEEE."},{"issue":"5","key":"9498_CR4","doi-asserted-by":"publisher","first-page":"2217","DOI":"10.1121\/1.1357814","volume":"109","author":"AMA Ali","year":"2001","unstructured":"Ali, A. M. A., Van der Spiegel, J., & Mueller, P. (2001). Acoustic-phonetic features for the automatic classification of fricatives. The Journal of the Acoustical Society of America, 109(5), 2217\u20132235.","journal-title":"The Journal of the Acoustical Society of America"},{"issue":"5","key":"9498_CR5","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1109\/TSA.2002.800556","volume":"10","author":"AMA Ali","year":"2002","unstructured":"Ali, A. M. A., Van der Spiegel, J., & Mueller, P. (2002). Robust auditory-based speech processing using the average localized synchrony detection. IEEE Transactions on Speech and Audio Processing, 10(5), 279\u2013292.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"issue":"1","key":"9498_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1561\/2200000006","volume":"2","author":"Y Bengio","year":"2009","unstructured":"Bengio, Y. (2009). Learning deep architectures for AI. Foundations and trends\u00ae in Machine Learning, 2(1), 1\u2013127.","journal-title":"Foundations and trends\u00ae in Machine Learning"},{"key":"9498_CR7","volume-title":"Bengali phonetic reader","author":"K Bhattacharya","year":"1988","unstructured":"Bhattacharya, K. (1988). Bengali phonetic reader (Vol.\u00a028). Mysuru: Central Institute of Indian Languages."},{"key":"9498_CR100","unstructured":"Bhowmik, T. (2017). Prosodic and phonological feature based speech recognition system for Bengali (Doctoral dissertation, IIT Kharagpur)."},{"key":"9498_CR8","unstructured":"Bitar, N., & Espy-Wilson, C. Y. (1995a). A signal representation of speech based on phonetic features. In Proceedings of 5th Annual Dual Use Technologies and Applications Conference (pp.\u00a0310\u2013315)."},{"key":"9498_CR9","doi-asserted-by":"crossref","unstructured":"Bitar, N. N., & Espy-Wilson, C. Y. (1995b). Speech parameterization based on phonetic features: Application to speech recognition. In Fourth European Conference on Speech Communication and Technology (pp.\u00a01411\u20131414).","DOI":"10.21437\/Eurospeech.1995-227"},{"key":"9498_CR10","unstructured":"Bitar, N. N., Espy-Wilson, C. Y. (1996). A knowledge-based signal representation for speech recognition. In Proceedings of IEEE International Conference on Acoustics, Speech, and Processing (pp.\u00a029\u201332). IEEE."},{"key":"9498_CR11","volume-title":"The origin and development of the Bengali language","author":"S Chatterji","year":"1926","unstructured":"Chatterji, S. (1926). The origin and development of the Bengali language. Calcutta: Calcutta University Press."},{"key":"9498_CR12","doi-asserted-by":"crossref","unstructured":"Dahl, G. E., Yu, D., Deng, L., & Acero, A. (2011). Large vocabulary continuous speech recognition with context dependent dbn-hmms. In IEEE International Conference on Acoustics, Speech And Signal Processing (ICASSP) (pp.\u00a04688\u20134691). IEEE.","DOI":"10.1109\/ICASSP.2011.5947401"},{"key":"9498_CR13","unstructured":"Das Mandal, S. (2007). Role of shape parameters in speech recognition: A study on standard colloquial Bengali (SCB). PhD thesis."},{"issue":"4","key":"9498_CR14","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","volume":"28","author":"S Davis","year":"1980","unstructured":"Davis, S., & Mermelstein, P. (1980). Comparison of parametric representations for monosyllabic word recognition in continuously spoken sentences. IEEE Transactions on Acoustics, Speech, and Signal Processing, 28(4), 357\u2013366.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"9498_CR15","unstructured":"Dekel, O., Keshet, J., & Singer, Y. (2004). An online algorithm for hierarchical phoneme classification. In International Workshop on Machine Learning for Multimodal Interaction (pp.\u00a0146\u2013158). Berlin: Springer."},{"key":"9498_CR16","doi-asserted-by":"crossref","unstructured":"Deng, L., Abdel-Hamid, O., & Yu, D. (2013). A deep convolutional neural network using heterogeneous pooling for trading acoustic invariance with phonetic confusion. In IEEE International Conference on Acoustics, Speech and Signal Processing (pp. 6669\u20136673). IEEE.","DOI":"10.1109\/ICASSP.2013.6638952"},{"key":"9498_CR17","volume-title":"Deep learning for signal and information processing","author":"L Deng","year":"2013","unstructured":"Deng, L., & Yu, D. (2013). Deep learning for signal and information processing. Redmond, WA: Microsoft Research Monograph."},{"key":"9498_CR18","doi-asserted-by":"crossref","unstructured":"Dusan, S. (2005). Estimation of speaker\u2019s height and vocal tract length from speech signal. In Ninth European Conference on Speech Communication and Technology.","DOI":"10.21437\/Interspeech.2005-625"},{"issue":"438","key":"9498_CR19","first-page":"548","volume":"92","author":"B Efron","year":"1997","unstructured":"Efron, B., & Tibshirani, R. (1997). Improvements on cross-validation: The 632 + bootstrap method. Journal of the American Statistical Association, 92(438), 548\u2013560.","journal-title":"Journal of the American Statistical Association"},{"issue":"8","key":"9498_CR20","doi-asserted-by":"publisher","first-page":"861","DOI":"10.1016\/j.patrec.2005.10.010","volume":"27","author":"T Fawcett","year":"2006","unstructured":"Fawcett, T. (2006). An introduction to roc analysis. Pattern Recognition Letters, 27(8), 861\u2013874.","journal-title":"Pattern Recognition Letters"},{"key":"9498_CR21","doi-asserted-by":"crossref","unstructured":"Feng, X., Zhang, Y., & Glass, J. (2014). Speech feature denoising and dereverberation via deep autoencoders for noisy reverberant speech recognition. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 1759\u20131763). IEEE.","DOI":"10.1109\/ICASSP.2014.6853900"},{"key":"9498_CR22","unstructured":"Frankel, J., & King, S. (2005). A hybrid ann\/dbn approach to articulatory feature recognition. In Proceedings of Eurospeech. Lisbon: CD-ROM."},{"key":"9498_CR23","doi-asserted-by":"publisher","DOI":"10.6028\/NIST.IR.4930","volume-title":"TIMIT: Acoustic-phonetic continuous speech corpus","author":"J Garofolo","year":"1993","unstructured":"Garofolo, J., Consortium, L. D., et al. (1993). TIMIT: Acoustic-phonetic continuous speech corpus. Philadelphia, PA: Linguistic Data Consortium."},{"key":"9498_CR24","unstructured":"Glorot, X., & Bengio, Y. (2010). Understanding the difficulty of training deep feedforward neural networks. In Proceedings of the Thirteenth International Conference on Artificial Intelligence and Statistics (Vol.\u00a09, pp.\u00a0249\u2013256)."},{"issue":"S1","key":"9498_CR25","doi-asserted-by":"publisher","first-page":"S11","DOI":"10.1121\/1.2003140","volume":"60","author":"H Goldberg","year":"1976","unstructured":"Goldberg, H., & Reddy, D. (1976). Feature extraction segmentation and labeling in the harpy and hearsay-ii systems. The Journal of the Acoustical Society of America, 60(S1), S11\u2013S11.","journal-title":"The Journal of the Acoustical Society of America"},{"issue":"5","key":"9498_CR26","doi-asserted-by":"publisher","first-page":"602","DOI":"10.1016\/j.neunet.2005.06.042","volume":"18","author":"A Graves","year":"2005","unstructured":"Graves, A., & Schmidhuber, J. (2005). Framewise phoneme classification with bidirectional lstm and other neural network architectures. Neural Networks, 18(5), 602\u2013610.","journal-title":"Neural Networks"},{"key":"9498_CR27","unstructured":"Harrington, J. (1987). Acoustic cues for automatic recognition of English consonants. In Speech Technology: A Survey. (pp.\u00a019\u201374). Edinburgh: Edinburgh University Press"},{"key":"9498_CR28","volume-title":"English sound structure","author":"J Harris","year":"1994","unstructured":"Harris, J. (1994). English sound structure. Oxford: Wiley."},{"issue":"1","key":"9498_CR29","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1007\/BF00133326","volume":"9","author":"B Hayes","year":"1991","unstructured":"Hayes, B., & Lahiri, A. (1991). Bengali intonational phonology. Natural Language & Linguistic Theory, 9(1), 47\u201396.","journal-title":"Natural Language & Linguistic Theory"},{"issue":"6","key":"9498_CR30","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G., Mohamed, A.-R., Jaitly, N., et al. (2012). Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups. IEEE Signal Processing Magazine, 29(6), 82\u201397.","journal-title":"IEEE Signal Processing Magazine"},{"issue":"7","key":"9498_CR31","doi-asserted-by":"publisher","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"G Hinton","year":"2006","unstructured":"Hinton, G., Osindero, S., & Teh, Y.-W. (2006). A fast learning algorithm for deep belief nets. Neural Computation, 18(7), 1527\u20131554.","journal-title":"Neural Computation"},{"issue":"5786","key":"9498_CR32","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1126\/science.1127647","volume":"313","author":"GE Hinton","year":"2006","unstructured":"Hinton, G. E., & Salakhutdinov, R. R. (2006). Reducing the dimensionality of data with neural networks. Science, 313(5786), 504\u2013507.","journal-title":"Science"},{"key":"9498_CR33","unstructured":"Hou, J. (2009). On the use of frame and segment-based methods for the detection and classification of speech sounds and features. PhD thesis, Rutgers University Graduate School, New Brunswick."},{"issue":"5","key":"9498_CR34","doi-asserted-by":"publisher","first-page":"1062","DOI":"10.1109\/78.134469","volume":"40","author":"X Huang","year":"1992","unstructured":"Huang, X. (1992). Phoneme classification using semicontinuous hidden markov models. IEEE Transactions on Signal Processing, 40(5), 1062\u20131067.","journal-title":"IEEE Transactions on Signal Processing"},{"issue":"4","key":"9498_CR35","doi-asserted-by":"publisher","first-page":"333","DOI":"10.1006\/csla.2000.0148","volume":"14","author":"S King","year":"2000","unstructured":"King, S., & Taylor, P. (2000). Detection of phonological features in continuous speech using neural networks. Computer Speech & Language, 14(4), 333\u2013353.","journal-title":"Computer Speech & Language"},{"key":"9498_CR36","unstructured":"King, S., Taylor, P., Frankel, J., & Richmond, K. (2000). Speech recognition via phonetically featured syllables. University of the Saarland."},{"key":"9498_CR37","unstructured":"Lahiri, A. (1999). Speech recognition with phonological features. In Proceedings of the XIVth International Congress of Phonetic Sciences (Vol.\u00a099, pp.\u00a0715\u2013718)."},{"key":"9498_CR38","doi-asserted-by":"crossref","unstructured":"Larochelle, H., Erhan, D., Courville, A., Bergstra, J., & Bengio, Y. (2007). An empirical evaluation of deep architectures on problems with many factors of variation. In Proceedings of the 24th International Conference on Machine learning (pp.\u00a0473\u2013480). ACM.","DOI":"10.1145\/1273496.1273556"},{"key":"9498_CR39","doi-asserted-by":"crossref","unstructured":"Lee, C.-H., Clements, M., Dusan, S., Fosler-Lussier, E., Johnson, K., Juang, B.-H., & Rabiner, L. (2007). An overview on automatic speech attribute transcription (ASAT). In INTERSPEECH (pp.\u00a01825\u20131828) Antwerp.","DOI":"10.21437\/Interspeech.2007-509"},{"key":"9498_CR40","volume-title":"Ethnologue: Languages of the world","author":"MP Lewis","year":"2016","unstructured":"Lewis, M. P., Simons, G. F., & Fennig, C. D. (2016). Ethnologue: Languages of the world (Vol.\u00a019). Dallas, TX: SIL International Dallas."},{"key":"9498_CR41","doi-asserted-by":"crossref","unstructured":"Mandal, S., Chandra, S., Lata, S., & Datta, A. (2011). Places and manner of articulation of Bangla consonants: An epg based study. In INTERSPEECH (pp.\u00a03149\u20133152) Florence.","DOI":"10.21437\/Interspeech.2011-788"},{"key":"9498_CR42","first-page":"49","volume":"6","author":"SD Mandal","year":"2005","unstructured":"Mandal, S. D., Saha, A., & Datta, A. (2005). Annotated speech corpora development in Indian languages. Vishwa Bharat, 6, 49\u201364.","journal-title":"Vishwa Bharat"},{"key":"9498_CR43","volume-title":"MATLAB version 8.5.0.197613 (R2015b)","author":"MATLAB","year":"2015","unstructured":"MATLAB. (2015). MATLAB version 8.5.0.197613 (R2015b). Natick: The Mathworks, Inc.."},{"key":"9498_CR44","doi-asserted-by":"crossref","unstructured":"Meyer, B. T., W\u00e4chter, M., Brand, T., & Kollmeier, B. (2007). Phoneme confusions in human and automatic speech recognition. In INTERSPEECH (pp.\u00a01485\u20131488).","DOI":"10.21437\/Interspeech.2007-430"},{"issue":"1","key":"9498_CR45","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1109\/TASL.2011.2109382","volume":"20","author":"A-R Mohamed","year":"2012","unstructured":"Mohamed, A.-R., Dahl, G. E., & Hinton, G. (2012). Acoustic modeling using deep belief networks. IEEE Transactions on Audio, Speech, and Language Processing, 20(1), 14\u201322.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9498_CR46","doi-asserted-by":"crossref","unstructured":"Mohamed, A.-R., Yu, D., & Deng, L. (2010). Investigation of full sequence training of deep belief networks for speech recognition. In INTERSPEECH (Vol.\u00a010, pp.\u00a02846\u20132849).","DOI":"10.21437\/Interspeech.2010-304"},{"key":"9498_CR47","doi-asserted-by":"crossref","unstructured":"Morales, S. O. C., & Cox, S. J. (2007). Modelling confusion matrices to improve speech recognition accuracy, with an application to dysarthric speech. In INTERSPEECH (pp.\u00a01565\u20131568).","DOI":"10.21437\/Interspeech.2007-126"},{"key":"9498_CR48","doi-asserted-by":"crossref","unstructured":"Moreau, N., Kim, H.-G., & Sikora, T. (2004). Phonetic confusion based document expansion for spoken document retrieval. In INTERSPEECH.","DOI":"10.21437\/Interspeech.2004-44"},{"key":"9498_CR49","unstructured":"Online census data (2016). Retrieved July 20, 2016, from http:\/\/censusindia.gov.in\/Census_Data_2001\/ Census_Data_Online\/Language\/Statement3.htm ."},{"key":"9498_CR50","unstructured":"Palm, R. B. (2012). Prediction as a candidate for learning deep hierarchical models of data. Master\u2019s thesis."},{"key":"9498_CR51","unstructured":"Reetz, H. (1999). Converting speech signals to phonological features. In Proceedings of the XIVth International Congress of Phonetic Sciences (Vol.\u00a099, pp.\u00a01733\u20131736)."},{"key":"9498_CR52","doi-asserted-by":"crossref","unstructured":"Renals, S., & Rohwer, R. (1989). Phoneme classification experiments using radial basis functions. In Proceedings of the IEEE International Joint Conference on Neural Networks (IJCNN\u00e2\u02d8A\u00b4Z89) (Vol.\u00a01, pp.\u00a0461\u2013467).","DOI":"10.1109\/IJCNN.1989.118620"},{"key":"9498_CR53","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1038\/323533a0","volume":"323","author":"D Rumelhart","year":"1986","unstructured":"Rumelhart, D., Hinton, G., & Williams, R. (1986). Learning representations by backpropagating errors. Nature, 323, 533\u2013536.","journal-title":"Nature"},{"key":"9498_CR54","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., & Yu, D. (2011). Conversational speech transcription using context-dependent deep neural networks. In INTERSPEECH (pp.\u00a0437\u2013440). Florence.","DOI":"10.21437\/Interspeech.2011-169"},{"issue":"11","key":"9498_CR55","doi-asserted-by":"publisher","first-page":"1139","DOI":"10.1016\/j.specom.2009.05.004","volume":"51","author":"S Siniscalchi","year":"2009","unstructured":"Siniscalchi, S., & Lee, C.-H. (2009). A study on integrating acoustic-phonetic information into lattice rescoring for automatic speech recognition. Speech Communication, 51(11), 1139\u20131153.","journal-title":"Speech Communication"},{"issue":"3","key":"9498_CR56","doi-asserted-by":"publisher","first-page":"875","DOI":"10.1109\/TASL.2011.2167610","volume":"20","author":"S Siniscalchi","year":"2012","unstructured":"Siniscalchi, S., Lyu, D.-C., Svendsen, T., Lee, C.-H. (2012). Experiments on cross-language attribute detection and phone recognition with minimal targetspecific training data. IEEE Transactions on Audio, Speech, and Language Processing, 20(3), 875\u2013887.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9498_CR57","doi-asserted-by":"crossref","unstructured":"Siniscalchi, S., Svendsen, T., & Lee, C.-H. (2007). Towards bottom-up continuous phone recognition. In IEEE Workshop on Automatic Speech Recognition & Understanding (ASRU) (pp.\u00a0566\u2013569). IEEE.","DOI":"10.1109\/ASRU.2007.4430174"},{"key":"9498_CR58","doi-asserted-by":"publisher","first-page":"148","DOI":"10.1016\/j.neucom.2012.11.008","volume":"106","author":"S Siniscalchi","year":"2013","unstructured":"Siniscalchi, S., Yu, D., Deng, L., & Lee, C.-H. (2013). Exploiting deep neural networks for detection based speech recognition. Neurocomputing, 106, 148\u2013157.","journal-title":"Neurocomputing"},{"key":"9498_CR59","doi-asserted-by":"crossref","unstructured":"Siniscalchi, S. M., & Reed, J., Svendsen, T., & Lee, C.-H. (2009). Exploring universal attribute characterization of spoken languages for spoken language recognition. In INTERSPEECH (pp.\u00a0168\u2013171). Brighton.","DOI":"10.21437\/Interspeech.2009-67"},{"key":"9498_CR60","doi-asserted-by":"crossref","unstructured":"Siniscalchi, S. M., Svendsen, T., & Lee, C.-H. (2011). A bottom-up stepwise knowledge integration approach to large vocabulary continuous speech recognition using weighted finite state machines. In INTERSPEECH (pp.\u00a0901\u2013904). Florence.","DOI":"10.21437\/Interspeech.2011-351"},{"key":"9498_CR61","doi-asserted-by":"crossref","unstructured":"Srinivasan, S., & Petkovic, D. (2000). Phonetic confusion matrix based spoken document retrieval. In Proceedings of the 23rd Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (pp. 81\u201387). ACM.","DOI":"10.1145\/345508.345552"},{"key":"9498_CR62","doi-asserted-by":"crossref","unstructured":"Vincent, P., Larochelle, H., Bengio, Y., & Manzagol, P.-A. (2008). Extracting and composing robust features with denoising autoencoders. In Proceedings of the 25th International Conference on Machine Learning (pp. 1096\u20131103). ACM.","DOI":"10.1145\/1390156.1390294"},{"key":"9498_CR63","first-page":"3371","volume":"11","author":"P Vincent","year":"2010","unstructured":"Vincent, P., Larochelle, H., Lajoie, I., Bengio, Y., Manzagol, P. A. (2010). Stacked denoising autoencoders: Learning useful representations in a deep network with a local denoising criterion. The Journal of Machine Learning Research, 11, 3371\u20133408.","journal-title":"The Journal of Machine Learning Research"},{"key":"9498_CR64","doi-asserted-by":"crossref","unstructured":"Xu, D., Wang, Y., & Metze, F. (2014). EM-based phoneme confusion matrix generation for low-resource spoken term detection. IEEE Spoken Language Technology Workshop (SLT) (pp.\u00a0424\u2013429). IEEE.","DOI":"10.1109\/SLT.2014.7078612"},{"key":"9498_CR65","volume-title":"Automatic speech recognition: A deep learning approach","author":"D Yu","year":"2014","unstructured":"Yu, D., & Deng, L. (2014). Automatic speech recognition: A deep learning approach. London: Springer."},{"key":"9498_CR66","unstructured":"Yu, D., Deng, L., & Dahl, G. (2010). Roles of pre-training and fine tuning in context dependent dbn-hmms for real world speech recognition. In Proceedings of NIPS Workshop on Deep Learning and Unsupervised Feature Learning."},{"key":"9498_CR67","doi-asserted-by":"crossref","unstructured":"Yu, D., Siniscalchi, S., Deng, L., & Lee, C.-H. (2012). Boosting attribute and phone estimation accuracies with deep neural networks for detection based speech recognition. In ICASSP (pp. 4169\u20134172). IEEE.","DOI":"10.1109\/ICASSP.2012.6288837"},{"issue":"3","key":"9498_CR68","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1016\/j.specom.2005.03.011","volume":"47","author":"A \u017dgank","year":"2005","unstructured":"\u017dgank, A., Horvat, B., & Ka\u010di\u010d Z. (2005). Data driven generation of phonetic broad classes, based on phoneme confusion matrix similarity. Speech Communication, 47(3), 379\u2013393.","journal-title":"Speech Communication"},{"key":"9498_CR69","unstructured":"Zhang, P., Shao, J., Han, J., Liu, Z., & Yan, Y. (2006). Keyword spotting based on phoneme confusion matrix. Proceedings of ICSLP (Vol.\u00a02, pp.\u00a0408\u2013419)."}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-018-9498-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-9498-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-9498-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,15]],"date-time":"2022-08-15T19:15:42Z","timestamp":1660590942000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-018-9498-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,3,9]]},"references-count":70,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2018,6]]}},"alternative-id":["9498"],"URL":"https:\/\/doi.org\/10.1007\/s10772-018-9498-5","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2018,3,9]]},"assertion":[{"value":"29 April 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 March 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 March 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}