{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:26:42Z","timestamp":1740122802822,"version":"3.37.3"},"reference-count":23,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2016,8,4]],"date-time":"2016-08-04T00:00:00Z","timestamp":1470268800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100005713","name":"Technische Universit\u00e4t M\u00fcnchen (DE)","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005713","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007601","name":"Horizon 2020","doi-asserted-by":"publisher","award":["644632","645378"],"award-info":[{"award-number":["644632","645378"]}],"id":[{"id":"10.13039\/501100007601","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2016,12]]},"DOI":"10.1007\/s10772-016-9357-1","type":"journal-article","created":{"date-parts":[[2016,8,4]],"date-time":"2016-08-04T16:19:52Z","timestamp":1470327592000},"page":"669-675","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Stream fusion for multi-stream automatic speech recognition"],"prefix":"10.1007","volume":"19","author":[{"given":"Hesam","family":"Sagha","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feipeng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ehsan","family":"Variani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jos\u00e9 del R.","family":"Mill\u00e1n","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ricardo","family":"Chavarriaga","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bj\u00f6rn","family":"Schuller","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,8,4]]},"reference":[{"issue":"4","key":"9357_CR1","doi-asserted-by":"crossref","first-page":"567","DOI":"10.1109\/89.326615","volume":"2","author":"J Allen","year":"1994","unstructured":"Allen, J. (1994). How do humans process and recognize speech? IEEE Transactions on Speech and Audio Processing, 2(4), 567\u2013577.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"9357_CR2","doi-asserted-by":"crossref","unstructured":"Bourlard, H. & Dupont, S. (1997). Subband-based speech recognition. In 22nd International Conference on Acoustics, Speech, and Signal Processing (ICASSP), vol\u00a02 (pp 1251\u20131254). Munich, Germany","DOI":"10.1109\/ICASSP.1997.596172"},{"key":"9357_CR3","unstructured":"Bourlard, H., Dupont , S., Ris, C. (1997). Multi-stream speech recognition. Tech. Rep. IDIAP-RR 96-07, IDIAP"},{"key":"9357_CR4","volume-title":"Speech and hearing in communication","author":"H Fletcher","year":"1953","unstructured":"Fletcher, H. (1953). Speech and hearing in communication. New York: Krieger."},{"key":"9357_CR5","unstructured":"Furui, S. (1992). Towards robust speech recognition under adverse conditions. In ESCA Workshop on Speech Processing in Adverse Conditions (pp. 31\u201341)"},{"issue":"5","key":"9357_CR6","doi-asserted-by":"crossref","first-page":"436","DOI":"10.1121\/1.4758826","volume":"132","author":"S Ganapathy","year":"2012","unstructured":"Ganapathy, S., & Hermansky, H. (2012). Temporal resolution analysis in frequency domain linear prediction. The Journal of the Acoustical Society of America, 132(5), 436\u2013442.","journal-title":"The Journal of the Acoustical Society of America"},{"key":"9357_CR7","unstructured":"Garofolo, J. S., et\u00a0al. (1988). Getting started with the darpa timit cd-rom: An acoustic phonetic continuous speech database. National Institute of Standards and Technology (NIST), Gaithersburgh, MD, p. 107"},{"key":"9357_CR8","doi-asserted-by":"crossref","unstructured":"Geiger, J. T., Zhang, Z., Weninger, F., Schuller, B., Rigoll, G. (2014). Robust speech recognition using long short-term memory recurrent neural networks for hybrid acoustic modelling. In: Proceedings of 15th Annual Conference of the International Speech Communication Association (INTERSPEECH), ISCA, Singapore, Singapore","DOI":"10.21437\/Interspeech.2014-151"},{"key":"9357_CR9","doi-asserted-by":"crossref","unstructured":"Giacinto, G., Roli, F. (2000). Dynamic classifier selection. In Multiple Classifier Systems (pp. 177\u2013189). Springer","DOI":"10.1007\/3-540-45014-9_17"},{"issue":"5","key":"9357_CR10","doi-asserted-by":"crossref","first-page":"1076","DOI":"10.1109\/JPROC.2012.2236871","volume":"101","author":"H Hermansky","year":"2013","unstructured":"Hermansky, H. (2013). Multistream recognition of speech: Dealing with unknown unknowns. IEEE Proceedings, 101(5), 1076\u20131088.","journal-title":"IEEE Proceedings"},{"issue":"4","key":"9357_CR11","doi-asserted-by":"crossref","first-page":"578","DOI":"10.1109\/89.326616","volume":"2","author":"H Hermansky","year":"1994","unstructured":"Hermansky, H., & Morgan, N. (1994). Rasta processing of speech. IEEE Transactions on Speech and Audio Processing, 2(4), 578\u2013589.","journal-title":"IEEE Transactions on Speech and Audio Processing"},{"key":"9357_CR12","doi-asserted-by":"crossref","unstructured":"Hermansky, H., Tibrewala, S., Pavel, M. (1996). Towards ASR on partially corrupted speech. In Fourth International Conference on Spoken Language (ICSLP), vol\u00a01 (pp. 462\u2013465). IEEE, Philadelphia, PA, USA","DOI":"10.21437\/ICSLP.1996-123"},{"key":"9357_CR13","doi-asserted-by":"crossref","unstructured":"Hermansky, H., Variani, E., Peddinti, V. (2013). Mean temporal distance: Predicting ASR error from temporal properties of speech signal. In 38th International Conference on Acoustics, Speech, and Signal Processing (ICASSP). IEEE, Vancouver, Canada","DOI":"10.1109\/ICASSP.2013.6639105"},{"issue":"7","key":"9357_CR14","doi-asserted-by":"crossref","first-page":"867","DOI":"10.1016\/j.specom.2012.02.005","volume":"54","author":"S Ikbal","year":"2012","unstructured":"Ikbal, S., Misra, H., Hermansky, H., & Magimai-Doss, M. (2012). Phase autocorrelation (PAC) features for noise robust speech recognition. Speech Communication, 54(7), 867\u2013880.","journal-title":"Speech Communication"},{"key":"9357_CR15","doi-asserted-by":"crossref","unstructured":"Mallidi, S. H., & Hermansky, H. (2016). Novel neural network based fusion for multistream ASR. In 41st International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 5680\u20135684). Shanghai, China: IEEE.","DOI":"10.1109\/ICASSP.2016.7472765"},{"key":"9357_CR16","doi-asserted-by":"crossref","unstructured":"Mallidi, S. H., Ogawa, T., & Hermansky, H. (2015). Uncertainty estimation of dnn classifiers. In IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU) (pp. 283\u2013288). USA: Arizona.","DOI":"10.1109\/ASRU.2015.7404806"},{"key":"9357_CR17","doi-asserted-by":"crossref","unstructured":"Mesgarani, N., Thomas, S., Hermansky, H. (2011). Adaptive stream fusion in multistream recognition of speech. In 12th Annual Conference of the International Speech Communication Association (InterSpeech). Portland, Oregon","DOI":"10.21437\/Interspeech.2011-618"},{"issue":"1","key":"9357_CR18","doi-asserted-by":"crossref","first-page":"14","DOI":"10.1109\/TASL.2011.2109382","volume":"20","author":"A Mohamed","year":"2012","unstructured":"Mohamed, A., Dahl, G., & Hinton, G. (2012). Acoustic modeling using deep belief networks. IEEE Transactions on Audio Speech and Language Processing, 20(1), 14\u201322.","journal-title":"IEEE Transactions on Audio Speech and Language Processing"},{"key":"9357_CR19","unstructured":"Sharma, S. R. (1999). Multi-stream approach to robust speech recognition. PhD thesis"},{"key":"9357_CR20","doi-asserted-by":"crossref","unstructured":"Tibrewala, S., Hermansky, H. (1997). Sub-band based recognition of noisy speech. In 22nd IEEE International Conference on Acoustics, Speech, and Signal Processing (ICASSP), vol\u00a02 (pp. 1255\u20131258). Munich, Germany,","DOI":"10.1109\/ICASSP.1997.596173"},{"key":"9357_CR21","doi-asserted-by":"crossref","unstructured":"Variani, E., Li, F., Hermansky, H. (2013). Multi-stream recognition of noisy speech with performance monitoring. In 14th Annual Conference of the International Speech Communication Association (InterSpeech). Lyon, France","DOI":"10.21437\/Interspeech.2013-273"},{"issue":"4","key":"9357_CR22","doi-asserted-by":"crossref","first-page":"888","DOI":"10.1016\/j.csl.2014.01.001","volume":"28","author":"F Weninger","year":"2014","unstructured":"Weninger, F., Geiger, J., W\u00f6llmer, M., Schuller, B., & Rigoll, G. (2014). Feature enhancement by deep LSTM networks for ASR in reverberant multisource environments. Computer Speech and Language, 28(4), 888\u2013902.","journal-title":"Computer Speech and Language"},{"issue":"3","key":"9357_CR23","doi-asserted-by":"crossref","first-page":"780","DOI":"10.1016\/j.csl.2012.05.002","volume":"27","author":"M W\u00f6llmer","year":"2013","unstructured":"W\u00f6llmer, M., Weninger, F., Geiger, J., Schuller, B., & Rigoll, G. (2013). Noise robust ASR in reverberated multisource environments applying convolutive NMF and long short-term memory. Computer Speech and Language, 27(3), 780\u2013797.","journal-title":"Computer Speech and Language"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-016-9357-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-016-9357-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-016-9357-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,19]],"date-time":"2023-08-19T14:02:06Z","timestamp":1692453726000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-016-9357-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,8,4]]},"references-count":23,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2016,12]]}},"alternative-id":["9357"],"URL":"https:\/\/doi.org\/10.1007\/s10772-016-9357-1","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2016,8,4]]}}}