{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T03:37:54Z","timestamp":1761709074849,"version":"3.37.3"},"reference-count":27,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2015,4,21]],"date-time":"2015-04-21T00:00:00Z","timestamp":1429574400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2016,2]]},"DOI":"10.1007\/s11265-015-1001-9","type":"journal-article","created":{"date-parts":[[2015,4,20]],"date-time":"2015-04-20T05:34:00Z","timestamp":1429508040000},"page":"187-196","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Ensemble Acoustic Modeling for CD-DNN-HMM Using Random Forests of Phonetic Decision Trees"],"prefix":"10.1007","volume":"82","author":[{"given":"Tuo","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5511-3692","authenticated-orcid":false,"given":"Yunxin","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,4,21]]},"reference":[{"key":"1001_CR1","doi-asserted-by":"crossref","unstructured":"Young, S.J., Odell, J.J., & Woodland, P.C. (1994). Tree-based state tying for high accuracy modeling. In Proc. ARPA Human Lang. Tech. Workshop (pp. 307\u2013312).","DOI":"10.3115\/1075812.1075885"},{"issue":"1","key":"1001_CR2","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"GE Dahl","year":"2012","unstructured":"Dahl, G.E., Yu, D., Deng, L., & Acero, A. (2012). Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Transactions on Audio, Speech and Language Processing, 20(1), 30\u201342.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"key":"1001_CR3","doi-asserted-by":"crossref","unstructured":"Deng, L., Yu, D., & Platt, J. (2012). Scalable stacking and learning for building deep architectures. In Proc. ICASSP (pp. 2133\u20132136).","DOI":"10.1109\/ICASSP.2012.6288333"},{"key":"1001_CR4","first-page":"1305","volume":"3","author":"G Cook","year":"1996","unstructured":"Cook, G., & Robinson, T. (1996). Boosting the performance of connectionist large vocabulary speech recognition. ICSLP, 3, 1305\u20131308.","journal-title":"ICSLP"},{"key":"1001_CR5","doi-asserted-by":"crossref","first-page":"1959","DOI":"10.21437\/Eurospeech.1997-520","volume":"3","author":"G Cook","year":"1997","unstructured":"Cook, G., Waterhouse, S., & Robinson, A. (1997). Ensemble methods for connectionist acoustic modelling. Proc. Eurospeech, 3, 1959\u20131962.","journal-title":"Proc. Eurospeech"},{"key":"1001_CR6","doi-asserted-by":"crossref","unstructured":"Schwenk, H. (1999). Using boosting to improve a hybrid HMM\/neural network speech recognizer. In Proc. ICASSP (pp. 1009\u20131012).","DOI":"10.1109\/ICASSP.1999.759874"},{"key":"1001_CR7","unstructured":"Kazemi, A., Sobhanmanesh, F., & Boostani, R. (2011). Boosting small MLPs with entropy combination improves phoneme posteriors enstimation. In Proc. International Symposium on AISP (pp. 11\u201314)."},{"key":"1001_CR8","doi-asserted-by":"crossref","unstructured":"Qian, Y., & Liu, J. (2012). Cross-lingual and ensemble MLPs strategies for low-resource speech recognition. In Proc. Interspeech (pp. 354\u2013358).","DOI":"10.21437\/Interspeech.2012-11"},{"issue":"3","key":"1001_CR9","doi-asserted-by":"crossref","first-page":"498","DOI":"10.1109\/TASL.2012.2227729","volume":"21","author":"X Chen","year":"2013","unstructured":"Chen, X., & Zhao, Y. (2013). Building acoustic model ensembles by data sampling with enhanced trainings and features. IEEE Transactions on Audio, Speech and Language Processing, 21(3), 498\u2013507.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"issue":"3","key":"1001_CR10","doi-asserted-by":"crossref","first-page":"519","DOI":"10.1109\/TASL.2007.913036","volume":"16","author":"J Xue","year":"2008","unstructured":"Xue, J., & Zhao, Y. (2008). Random forests of phonetic decision trees for acoustic modeling in conversational speech recognition. IEEE Transactions on Audio, Speech and Language Processing, 16(3), 519\u2013528.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"key":"1001_CR11","unstructured":"Siohan, O., Ramabhadran, B., & Kingsbury, B. (2005). Constructing ensembles of ASR systems using randomized decision trees. In Proc. ICASSP (pp. I-197-I-200)."},{"issue":"1","key":"1001_CR12","doi-asserted-by":"crossref","first-page":"5","DOI":"10.1023\/A:1010933404324","volume":"45","author":"L Breiman","year":"2001","unstructured":"Breiman, L. (2001). Random forests. Machine Learning, 45(1), 5\u201332.","journal-title":"Machine Learning"},{"issue":"2","key":"1001_CR13","doi-asserted-by":"crossref","first-page":"341","DOI":"10.1016\/0031-3203(95)00085-2","volume":"29","author":"K Tumer","year":"1996","unstructured":"Tumer, K., & Ghosh, J. (1996). Analysis of decision boundaries in linearly combined neural classifiers. Pattern Recognition, 29(2), 341\u2013348.","journal-title":"Pattern Recognition"},{"key":"1001_CR14","unstructured":"Krogh, A., & Vedelsby, J. (1995). Neural network ensembles, cross validation, and active learning. In G. Tesauro, D. S. Touretzky, & T. K. Leen (Eds.), Advances in neural information processing systems (pp. 231\u2013238)."},{"issue":"3","key":"1001_CR15","first-page":"711","volume":"22","author":"K Audhkhasi","year":"2014","unstructured":"Audhkhasi, K., Zavou, A.M., Georgiou, P.G., & Narayanan, S.S. (2014). Theoretical analysis of diversity in an ensemble of automatic speech recognition systems. IEEE Transactions on ASLP, 22(3), 711\u2013726.","journal-title":"IEEE Transactions on ASLP"},{"key":"1001_CR16","unstructured":"Zhao, Y., Xue, J., & Chen, X. (2014). Ensemble learning approaches in speech recognition. In T. Ogunfunmi, R. Togneri, & M. Narasimha (Eds.), Speech and audio processing for coding, enhancement and recognition: Springer."},{"key":"1001_CR17","doi-asserted-by":"crossref","unstructured":"Fiscus, J.G. (1997). A post-processing system to yield reduced word error rates: recognizer output voting error reduction (ROVER). In Proc. IEEE ASRU (pp. 347\u2013352).","DOI":"10.1109\/ASRU.1997.659110"},{"key":"1001_CR18","doi-asserted-by":"crossref","unstructured":"Shinozaki, T., & Furui, S. (2004). Spontaneous speech recognition using a massively parallel decoder. In Proc. ICSLP (pp. 1705\u20131708).","DOI":"10.21437\/Interspeech.2004-185"},{"key":"1001_CR19","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Zhang, X., Hu, R.-S., Xue, J., Li, X., Che, L., Hu, R., & Schopp, L. (2006). An automatic captioning system for telemedicine. In Proc. ICASSP (pp. I-957-I-960).","DOI":"10.1109\/ICASSP.2006.1660181"},{"key":"1001_CR20","doi-asserted-by":"crossref","unstructured":"Zhao, T., Zhao, Y., & Chen, X. (2014). Building an ensemble of CD-DNN-HMM acoustic model using random forests of phonetic decision trees. In Proc. ISCSLP (pp. 98\u2013102).","DOI":"10.1109\/ISCSLP.2014.6936680"},{"key":"1001_CR21","doi-asserted-by":"crossref","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"GE Hinton","year":"2006","unstructured":"Hinton, G.E., Osindero, S., & Teh, Y. (2006). A fast learning algorithm for deep belief nets. Neural Computation, 18, 1527\u20131554.","journal-title":"Neural Computation"},{"key":"1001_CR22","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., & Yu, D. (2011). Feature engineering in context-dependent deep neural networks for conversational speech transcription. In Proc. IEEE ASRU (pp. 24\u201329).","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"1001_CR23","unstructured":"(2009). The hidden Markov model toolkit (HTK). CUED Machine Intelligence Lab. accessed 28 June 2013. http:\/\/htk.eng.cam.ac.uk\/ftp\/software\/HTK-3.4.1.tar.gz ."},{"key":"1001_CR24","doi-asserted-by":"crossref","unstructured":"Vesely, K., Burget, L., & Grezl, F. (2010). Parallel training of neural networks for speech recognition. In Proc. International Conf Text, Speech and Dialog (pp. 439\u2013446).","DOI":"10.1007\/978-3-642-15760-8_56"},{"issue":"11","key":"1001_CR25","first-page":"1641","volume":"37","author":"K Lee","year":"1989","unstructured":"Lee, K., & Hon, H. (1989). Speaker-independent phone recognition using hidden Markov models. IEEE Transactions on Audio, Speech and Language Processing, 37(11), 1641\u20131648.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"issue":"3","key":"1001_CR26","doi-asserted-by":"crossref","first-page":"332","DOI":"10.1109\/TITB.2006.885549","volume":"11","author":"X Zhang","year":"2007","unstructured":"Zhang, X., Zhao, Y., & Schopp, L. (2007). A novel method of language modeling for automatic captioning in telemedicine. IEEE Transactions on Information Technology in Biomedicine, 11(3), 332\u2013337.","journal-title":"IEEE Transactions on Information Technology in Biomedicine"},{"key":"1001_CR27","doi-asserted-by":"crossref","unstructured":"Sun, X., & Zhao, Y. (2014). Integrated exemplar-based template matching and statistical modeling for continuous speech recognition. In Proc. EURASIP Journal on Audio, Speech and Music (Vol. 4, p. 16).","DOI":"10.1186\/1687-4722-2014-4"}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-015-1001-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-015-1001-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-015-1001-9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,9]],"date-time":"2023-08-09T16:16:54Z","timestamp":1691597814000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-015-1001-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,4,21]]},"references-count":27,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2016,2]]}},"alternative-id":["1001"],"URL":"https:\/\/doi.org\/10.1007\/s11265-015-1001-9","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"type":"print","value":"1939-8018"},{"type":"electronic","value":"1939-8115"}],"subject":[],"published":{"date-parts":[[2015,4,21]]}}}