{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T15:14:10Z","timestamp":1774624450728,"version":"3.50.1"},"publisher-location":"Berlin, Heidelberg","reference-count":19,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"value":"9783540685845","type":"print"},{"value":"9783540685852","type":"electronic"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.1007\/978-3-540-68585-2_40","type":"book-chapter","created":{"date-parts":[[2008,8,12]],"date-time":"2008-08-12T16:07:43Z","timestamp":1218557263000},"page":"429-441","source":"Crossref","is-referenced-by-count":6,"title":["The IBM Rich Transcription 2007 Speech-to-Text Systems for Lecture Meetings"],"prefix":"10.1007","author":[{"given":"Jing","family":"Huang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Etienne","family":"Marcheret","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Karthik","family":"Visweswariah","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vit","family":"Libal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gerasimos","family":"Potamianos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"40_CR1","unstructured":"Computers in the Human Interaction Loop, http:\/\/chil.server.de"},{"key":"40_CR2","unstructured":"Augmented Multi-party Interaction, http:\/\/www.amiproject.org"},{"key":"40_CR3","unstructured":"The NIST SmartSpace Laboratory, http:\/\/www.nist.gov\/smartspace"},{"key":"40_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"309","DOI":"10.1007\/11965152_28","volume-title":"Machine Learning for Multimodal Interaction","author":"J.G. Fiscus","year":"2006","unstructured":"Fiscus, J.G., Ajot, J., Michel, M., Garofolo, J.S.: The Rich Transcription 2006 Spring meeting recognition evaluation. In: Renals, S., Bengio, S., Fiscus, J.G. (eds.) MLMI 2006. LNCS, vol.\u00a04299, pp. 309\u2013322. Springer, Heidelberg (2006)"},{"key":"40_CR5","unstructured":"Huang, J., Marcheret, E., Visweswariah, K., Potamianos, G.: The IBM RT07 evaluation system for speaker diarization in CHIL seminars (same volume) (2007)"},{"key":"40_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"432","DOI":"10.1007\/11965152_38","volume-title":"Machine Learning for Multimodal Interaction","author":"J. Huang","year":"2006","unstructured":"Huang, J., Westphal, M., Chen, S., et al.: The IBM Rich Transcription Spring 2006 speech-to-text system for lecture meetings. In: Renals, S., Bengio, S., Fiscus, J.G. (eds.) MLMI 2006. LNCS, vol.\u00a04299, pp. 432\u2013443. Springer, Heidelberg (2006)"},{"key":"40_CR7","unstructured":"Fiscus, J.G.: A post-processing system to yield reduced word error rates: Recogniser output voting error reduction (ROVER). In: Proc. Automatic Speech Recognition Underst. Works, Santa Barbara, CA, pp. 347\u2013352 (1997)"},{"key":"40_CR8","doi-asserted-by":"crossref","unstructured":"Siohan, O., Ramabhadran, B., Kingsbury, B.: Constructing ensembles of ASR systems using randomized decision trees. In: Proc. Int. Conf. Acoustics Speech Signal Process, Philadelphia, vol.\u00a01, pp. 197\u2013200 (2005)","DOI":"10.1109\/ICASSP.2005.1415084"},{"key":"40_CR9","unstructured":"The LDC Corpus Catalog, Linguistic Data Consortium, University of Pennsylvania, Philadelphia, PA, http:\/\/www.ldc.upenn.edu\/Catalog"},{"key":"40_CR10","doi-asserted-by":"crossref","unstructured":"Lamel, L.F., Schiel, F., Fourcin, A., Mariani, J., Tillmann, H.: The translanguage English database (TED). In: Proc. Int. Conf. Spoken Language Process, Yokohama, Japan (1994)","DOI":"10.21437\/ICSLP.1994-451"},{"key":"40_CR11","doi-asserted-by":"crossref","unstructured":"Boakye, K., Stolcke, A.: Improved speech activity detection using cross-channel features for recognition of multiparty meetings. In: Proc. Int. Conf. Spoken Language Process, Pittsburgh, pp. 1962\u20131965 (2006)","DOI":"10.21437\/Interspeech.2006-538"},{"key":"40_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"323","DOI":"10.1007\/11965152_29","volume-title":"Machine Learning for Multimodal Interaction","author":"E. Marcheret","year":"2006","unstructured":"Marcheret, E., Potamianos, G., Visweswariah, K., Huang, J.: The IBM RT06s evaluation system for speech activity detection in CHIL seminars. In: Renals, S., Bengio, S., Fiscus, J.G. (eds.) MLMI 2006. LNCS, vol.\u00a04299, pp. 323\u2013335. Springer, Heidelberg (2006)"},{"key":"40_CR13","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"M.J.F. Gales","year":"1998","unstructured":"Gales, M.J.F.: Maximum likelihood linear transformations for HMM-based speech recognition. Computer Speech and Language\u00a012, 75\u201398 (1998)","journal-title":"Computer Speech and Language"},{"key":"40_CR14","doi-asserted-by":"crossref","unstructured":"Saon, G., Zweig, G., Padmanabhan, M.: Linear feature space projections for speaker adaptation. In: Proc. Int. Conf. Acoustics Speech Signal Process, Salt Lake City, UT, pp. 325\u2013328 (2001)","DOI":"10.1109\/ICASSP.2001.940833"},{"key":"40_CR15","doi-asserted-by":"crossref","unstructured":"Povey, D., Kingsbury, B., Mangu, L., Saon, G., Soltau, H., Zweig, G.: fMPE: Discriminatively trained features for speech recognition. In: Proc. Int. Conf. Acoustics Speech Signal Process, Philadelphia, vol.\u00a01, pp. 961\u2013964 (2005)","DOI":"10.1109\/ICASSP.2005.1415275"},{"key":"40_CR16","doi-asserted-by":"crossref","unstructured":"Povey, D., Woodland, P.C.: Minimum phone error and I-smoothing for improved discriminative training. In: Proc. Int. Conf. Acoustics Speech Signal Process, Orlando, FL, pp. 105\u2013108 (2002)","DOI":"10.1109\/ICASSP.2002.5743665"},{"key":"40_CR17","doi-asserted-by":"crossref","unstructured":"Zheng, J., Stolcke, A.: Improved discriminative training using phone lattices. In: Proc. Eurospeech, Lisbon, Portugal, pp. 2125\u20132128 (2005)","DOI":"10.21437\/Interspeech.2005-691"},{"key":"40_CR18","doi-asserted-by":"publisher","first-page":"359","DOI":"10.1006\/csla.1999.0128","volume":"13","author":"S.F. Chen","year":"1999","unstructured":"Chen, S.F., Goodman, J.: An empirical study of smoothing techniques for language modeling. Computer Speech and Language\u00a013, 359\u2013393 (1999)","journal-title":"Computer Speech and Language"},{"key":"40_CR19","unstructured":"Stolcke, A.: Entropy-based pruning of backoff language models. In: Proc. DARPA Broadcast News Transcr. Underst. Works, Lansdowne, VA, pp. 270\u2013274 (1998)"}],"container-title":["Lecture Notes in Computer Science","Multimodal Technologies for Perception of Humans"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-540-68585-2_40.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,31]],"date-time":"2025-01-31T11:56:08Z","timestamp":1738324568000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-540-68585-2_40"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[null]]},"ISBN":["9783540685845","9783540685852"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-540-68585-2_40","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[]}}