{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T20:43:03Z","timestamp":1725914583819},"publisher-location":"Cham","reference-count":31,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_16","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"355-368","source":"Crossref","is-referenced-by-count":4,"title":["Distant Speech Recognition Experiments Using the AMI Corpus"],"prefix":"10.1007","author":[{"given":"Steve","family":"Renals","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pawel","family":"Swietojanski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"16_CR1","doi-asserted-by":"crossref","unstructured":"Abdel-Hamid, O., Mohamed, A.R., Hui, J., Penn, G.: Applying convolutional neural networks concepts to hybrid NN\u2013HMM model for speech recognition. In: Proceedings of the IEEE ICASSP, pp.\u00a04277\u20134280 (2012)","DOI":"10.1109\/ICASSP.2012.6288864"},{"key":"16_CR2","volume-title":"Exploring convolutional neural network structures and optimization techniques for speech recognition","author":"O. Abdel-Hamid","year":"2013","unstructured":"Abdel-Hamid, O., Deng, L., Yu, D.: Exploring convolutional neural network structures and optimization techniques for speech recognition. In: Proceedings of the ICSA Interspeech (2013)"},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Adcock, J., Gotoh, Y., Mashao, D., Silverman, H.: Microphone-array speech recognition via incremental MAP training. In: Proceedings of the IEEE ICASSP, pp.\u00a0897\u2013900 (1996)","DOI":"10.1109\/ICASSP.1996.543266"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Anastasakos, T., McDonough, J., Schwartz, R., Makhoul, J.: A compact model for speaker-adaptive training. In: Proceedings of the ICSLP, pp.\u00a01137\u20131140 (1996)","DOI":"10.1109\/ICSLP.1996.607807"},{"key":"16_CR5","doi-asserted-by":"crossref","first-page":"2011","DOI":"10.1109\/TASL.2007.902460","volume":"15","author":"X. Anguera","year":"2007","unstructured":"Anguera, X., Wooters, C., Hernando, J.: Acoustic beamforming for speaker diarization of meetings. IEEE Trans. Audio Speech Lang. Process. 15, 2011\u20132021 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"16_CR6","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1017\/CBO9781139136310.002","volume-title":"Multimodal Signal Processing: Human Interactions in Meetings, chap. 2","author":"J. Carletta","year":"2012","unstructured":"Carletta, J., Lincoln, M.: Data collection. In: Renals, S., Bourlard, H., Carletta, J., Popescu-Belis, A. (eds.) Multimodal Signal Processing: Human Interactions in Meetings, chap. 2, pp. 11\u201327. Cambridge University Press, Cambridge (2012)"},{"key":"16_CR7","doi-asserted-by":"crossref","unstructured":"Carletta, J., Ashby, S., Bourban, S., Flynn, M., Guillemot, M., Hain, T., Kadlec, J., Karaiskos, V., Kraaij, W., Kronenthal, M., Lathoud, G., Lincoln, M., Lisowska, A., McCowan, I., Post, W., Reidsma, D., Wellner, P.: The AMI meeting corpus: a pre-announcement. In: Proceedings of the Machine Learning for Multimodal Interaction (MLMI), pp.\u00a028\u201339 (2005)","DOI":"10.1007\/11677482_3"},{"key":"16_CR8","doi-asserted-by":"crossref","first-page":"313","DOI":"10.1007\/s10579-006-9001-9","volume":"39","author":"J. Carletta","year":"2005","unstructured":"Carletta, J., Evert, S., Heid, U., Kilgour, J.: The NITE XML toolkit: data model and query language. Lang. Resour. Eval. 39, 313\u2013334 (2005)","journal-title":"Lang. Resour. Eval."},{"key":"16_CR9","volume-title":"Multiple dimension Levenshtein edit distance calculations for evaluating ASR systems during simultaneous speech","author":"J. Fiscus","year":"2006","unstructured":"Fiscus, J., Ajot, J., Radde, N., Laprun, C.: Multiple dimension Levenshtein edit distance calculations for evaluating ASR systems during simultaneous speech. In: Proceedings of the LREC (2006)"},{"issue":"3","key":"16_CR10","doi-asserted-by":"crossref","first-page":"272","DOI":"10.1109\/89.759034","volume":"7","author":"M. Gales","year":"1999","unstructured":"Gales, M.: Semi-tied covariance matrices for hidden Markov models. IEEE Trans. Speech Audio Process. 7(3), 272\u2013281 (1999)","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"16_CR11","volume-title":"Multilingual training of deep neural networks","author":"A. Ghoshal","year":"2013","unstructured":"Ghoshal, A., Swietojanski, P., Renals, S.: Multilingual training of deep neural networks. In: Proceedings of the IEEE ICASSP (2013)"},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Grezl, F., Karafiat, M., Kontar, S., Cernocky, J.: Probabilistic and bottle-neck features for LVCSR of meetings. In: Proceedings of IEEE ICASSP, pp.\u00a0IV-757\u2013IV-760 (2007)","DOI":"10.1109\/ICASSP.2007.367023"},{"key":"16_CR13","unstructured":"Haeb-Umbach, R., Ney, H.: Linear discriminant analysis for improved large vocabulary continuous speech recognition. In: Proceedings of the IEEE ICASSP, pp.\u00a013\u201316 (1992). http:\/\/dl.acm.org\/citation.cfm?id=1895550.1895555"},{"key":"16_CR14","doi-asserted-by":"crossref","first-page":"486","DOI":"10.1109\/TASL.2011.2163395","volume":"20","author":"T. Hain","year":"2012","unstructured":"Hain, T., Burget, L., Dines, J., Garner, P., Gr\u00e9zl, F., El Hannani, A., Karaf\u00edat, M., Lincoln, M., Wan, V.: Transcribing meetings with the AMIDA systems. IEEE Trans. Audio Speech Lang. Process. 20, 486\u2013498 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"16_CR15","doi-asserted-by":"crossref","unstructured":"Janin, A., Baron, D., Edwards, J., Ellis, D., Gelbart, D., Morgan, N., Peskin, B., Pfau, T., Shriberg, E., Stolcke, A., Wooters, C.: The ICSI meeting corpus. In: Proceedings of the IEEE ICASSP, pp.\u00a0I-364\u2013I-367 (2003)","DOI":"10.1109\/ICASSP.2003.1198793"},{"key":"16_CR16","doi-asserted-by":"crossref","first-page":"23","DOI":"10.1016\/0893-6080(90)90044-L","volume":"3","author":"K. Lang","year":"1990","unstructured":"Lang, K., Waibel, A., Hinton, G.: A time-delay neural network architecture for isolated word recognition. Neural Netw. 3, 23\u201343 (1990)","journal-title":"Neural Netw."},{"key":"16_CR17","doi-asserted-by":"crossref","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y. LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86, 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"16_CR18","volume-title":"The multi-channel Wall Street Journal audio visual corpus (MC-WSJ-AV): specification and initial experiments","author":"M. Lincoln","year":"2005","unstructured":"Lincoln, M., McCowan, I., Vepa, J., Maganti, H.: The multi-channel Wall Street Journal audio visual corpus (MC-WSJ-AV): specification and initial experiments. In: Proceedings of the IEEE ASRU (2005)"},{"key":"16_CR19","doi-asserted-by":"crossref","unstructured":"Marino, D., Hain, T.: An analysis of automatic speech recognition with multiple microphones. In: Proceedings of the Interspeech, pp.\u00a01281\u20131284 (2011)","DOI":"10.21437\/Interspeech.2011-106"},{"key":"16_CR20","doi-asserted-by":"crossref","unstructured":"Omologo, M., Matassoni, M., Svaizer, P., Giuliani, D.: Microphone array based speech recognition with different talker-array positions. In: Proceedings of the IEEE ICASSP, pp.\u00a0227\u2013230 (1997)","DOI":"10.1109\/ICASSP.1997.599610"},{"key":"16_CR21","doi-asserted-by":"crossref","unstructured":"Povey, D., Woodland, P.: Minimum phone error and I-smoothing for improved discriminative training. In: Proceedings of the IEEE ICASSP, pp.\u00a0105\u2013108 (2002)","DOI":"10.1109\/ICASSP.2002.1005687"},{"key":"16_CR22","volume-title":"Neural networks for distant speech recognition","author":"S. Renals","year":"2014","unstructured":"Renals, S., Swietojanski, P.: Neural networks for distant speech recognition. In: Proceedings of the HSCMA (2014)"},{"key":"16_CR23","volume-title":"Improvements to deep convolutional neural networks for LVCSR","author":"T. Sainath","year":"2013","unstructured":"Sainath, T., Kingsbury, B., Mohamed, A., Dahl, G., Saon, G., Soltau, H., Beran, T., Aravkin, A., Ramabhadran, B.: Improvements to deep convolutional neural networks for LVCSR. In: Proceedings of the IEEE ASRU (2013)"},{"key":"16_CR24","doi-asserted-by":"crossref","first-page":"2109","DOI":"10.1109\/TASL.2006.872614","volume":"14","author":"M. Seltzer","year":"2006","unstructured":"Seltzer, M., Stern, R.: Subband likelihood-maximizing beamforming for speech recognition in reverberant environments. IEEE Trans. Audio Speech Lang. Process. 14, 2109\u20132121 (2006)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"16_CR25","doi-asserted-by":"crossref","first-page":"489","DOI":"10.1109\/TSA.2004.832988","volume":"12","author":"M. Seltzer","year":"2004","unstructured":"Seltzer, M., Raj, B., Stern, R.: Likelihood-maximizing beamforming for robust hands-free speech recognition. IEEE Trans. Speech Audio Process. 12, 489\u2013498 (2004)","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"16_CR26","first-page":"373","volume-title":"Multimodal Technologies for Perception of Humans. Lecture Notes in Computer Science","author":"A. Stolcke","year":"2008","unstructured":"Stolcke, A., Anguera, X., Boakye, K., Cetin, O., Janin, A., Magimai-Doss, M., Wooters, C., Zheng, J.: The SRI-ICSI spring 2007 meeting and lecture recognition system. In: Stiefelhagen, R., Bowers, R., Fiscus, J. (eds.) Multimodal Technologies for Perception of Humans. Lecture Notes in Computer Science, vol.\u00a04625, pp.\u00a0373\u2013389. Springer, New York (2008)"},{"key":"16_CR27","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707744","author":"P. Swietojanski","year":"2013","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Hybrid acoustic models for distant and multichannel large vocabulary speech recognition. In: Proceedings of the IEEE ASRU (2013). doi:10.1109\/ASRU.2013.6707744","journal-title":"In: Proceedings of the IEEE ASRU"},{"key":"16_CR28","doi-asserted-by":"crossref","first-page":"1120","DOI":"10.1109\/LSP.2014.2325781","volume":"21","author":"P. Swietojanski","year":"2014","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Convolutional neural networks for distant speech recognition. IEEE Signal Process. Lett. 21, 1120\u20131124 (2014)","journal-title":"IEEE Signal Process. Lett."},{"key":"16_CR29","doi-asserted-by":"crossref","first-page":"433","DOI":"10.1016\/0167-6393(90)90019-6","volume":"9","author":"D. Compernolle Van","year":"1990","unstructured":"Van Compernolle, D., Ma, W., Xie, F., Van Diest, M.: Speech recognition in noisy environments with the aid of microphone arrays. Speech Commun. 9, 433\u2013442 (1990)","journal-title":"Speech Commun."},{"key":"16_CR30","doi-asserted-by":"crossref","DOI":"10.1002\/9780470714089","volume-title":"Distant Speech Recognition","author":"M. W\u00f6lfel","year":"2009","unstructured":"W\u00f6lfel, M., McDonough, J.: Distant Speech Recognition. Wiley, Chichester (2009)"},{"key":"16_CR31","doi-asserted-by":"crossref","unstructured":"Zwyssig, E., Lincoln, M., Renals, S.: A digital microphone array for distant speech recognition. In: Proceedings of the IEEE ICASSP, pp.\u00a05106\u20135109 (2010). doi:10.1109\/ICASSP.2010.5495040","DOI":"10.1109\/ICASSP.2010.5495040"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T22:20:27Z","timestamp":1659738027000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_16","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}