{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,4]],"date-time":"2024-09-04T19:34:32Z","timestamp":1725478472709},"publisher-location":"Berlin, Heidelberg","reference-count":31,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540692676"},{"type":"electronic","value":"9783540692683"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2006]]},"DOI":"10.1007\/11965152_36","type":"book-chapter","created":{"date-parts":[[2007,1,23]],"date-time":"2007-01-23T08:48:58Z","timestamp":1169542138000},"page":"407-418","source":"Crossref","is-referenced-by-count":3,"title":["The ISL RT-06S Speech-to-Text System"],"prefix":"10.1007","author":[{"given":"Christian","family":"F\u00fcgen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shajith","family":"Ikbal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Florian","family":"Kraft","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kenichi","family":"Kumatani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kornel","family":"Laskowski","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John W.","family":"McDonough","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mari","family":"Ostendorf","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sebastian","family":"St\u00fcker","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Matthias","family":"W\u00f6lfel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"36_CR1","unstructured":"F\u00fcgen, C., Kolss, M., Bernreuther, D., Paulik, M., St\u00fcker, S., Vogel, S., Waibel, A.: Open Domain Speech Recognition & Translation: Lectures and Speeches. In: ICASSP (2006)"},{"key":"36_CR2","doi-asserted-by":"crossref","unstructured":"W\u00f6lfel, M., McDonough, J.: Combining Multi-Source Far Distance Speech Recognition Strategies: Beamforming, Blind Channel and Confusion Network Combination. In: INTERSPEECH (2005)","DOI":"10.21437\/Interspeech.2005-270"},{"key":"36_CR3","unstructured":"Metze, F., Jin, Q., F\u00fcgen, C., Laskowski, K., Pan, Y., Schultz, T.: Issues in Meeting Transcription \u2013 The ISL Meeting Transcription System. In: ICSLP (2004)"},{"key":"36_CR4","doi-asserted-by":"crossref","unstructured":"W\u00f6lfel, M., McDonough, J.: Minimum Variance Distortionless Response Spectral Estimation Review and Refinements. IEEE Signal Processing Magazine (September 2005)","DOI":"10.1109\/MSP.2005.1511829"},{"key":"36_CR5","doi-asserted-by":"crossref","unstructured":"St\u00fcker, S., F\u00fcgen, C., Burger, S., W\u00f6lfel, M.: Cross-System Adaptation and Combination for Continuous Speech Recognition: The Influence of Phoneme Set and Acoustic Front-End. In: INTERSPEECH (2006)","DOI":"10.21437\/Interspeech.2006-199"},{"key":"36_CR6","doi-asserted-by":"crossref","unstructured":"Jin, Q., Schultz, T.: Speaker Segmentation and Clustering in Meetings. In: ICSLP (2004)","DOI":"10.21437\/Interspeech.2004-249"},{"key":"36_CR7","unstructured":"St\u00fcker, S., F\u00fcgen, C., Hsiao, R., Ikbal, S., Jin, Q., Kraft, F., Paulik, M., Raab, M.W.M., Tam, Y.-C.: The ISL TC-STAR Spring 2006 ASR Evaluation Systems. In: TC-Star Workshop on Speech-to-Speech Translation (2006)"},{"issue":"4","key":"36_CR8","doi-asserted-by":"publisher","first-page":"561","DOI":"10.1109\/PROC.1975.9792","volume":"63","author":"J. Makhoul","year":"1975","unstructured":"Makhoul, J.: Linear Prediction: A Tutorial Review. Proc. of the IEEE\u00a063(4), 561\u2013580 (1975)","journal-title":"Proc. of the IEEE"},{"key":"36_CR9","doi-asserted-by":"crossref","unstructured":"F\u00fcgen, C., W\u00f6lfel, M., McDonough, J.W., Ikbal, S., Kraft, F., Laskowski, K., Ostendorf, M., St\u00fcker, S., Kumatani, K.: Advances in Lecture Recognition: The ISL RT-06S Evaluation System. In: INTERSPEECH (2006)","DOI":"10.21437\/Interspeech.2006-370"},{"key":"36_CR10","doi-asserted-by":"crossref","unstructured":"Pfau, T., Ellis, D.P.W., Stolcke, A.: Multispeaker Speech Activity Detection for the ICSI Meeting Recorder. In: Proc. ASRU (2001)","DOI":"10.1109\/ASRU.2001.1034599"},{"key":"36_CR11","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1109\/TSA.2004.838531","volume":"13","author":"S.N. Wrigley","year":"2005","unstructured":"Wrigley, S.N., Brown, G.J., Wan, V., Renals, S.: Speech and Crosstalk Detection in Multichannel Audio. IEEE Trans. on Speech and Audio Processing\u00a013, 84\u201391 (2005)","journal-title":"IEEE Trans. on Speech and Audio Processing"},{"key":"36_CR12","doi-asserted-by":"crossref","unstructured":"Laskowski, K., Schultz, T.: Unsupervised Learning of Overlapped Speech Model Parameters for Multichannel Speech Activity Detection in Meetings. In: Proc. ICASSP (2006)","DOI":"10.1109\/ICASSP.2006.1660190"},{"key":"36_CR13","unstructured":"\u00c7etin, \u00d6., Shriberg, E.: Speaker Overlaps and ASR Errors in Meetings: Effects Before, During, and After the Overlap. In: Proc. ICASSP (2006)"},{"key":"36_CR14","unstructured":"Soltau, H., Metze, F., F\u00fcgen, C., Waibel, A.: A One Pass-Decoder Based on Polymorphic Linguistic Context Assignment. In: ASRU (2001)"},{"key":"36_CR15","doi-asserted-by":"crossref","unstructured":"Gales, M.J.F.: Semi-tied covariance matrices. In: ICASSP (1998)","DOI":"10.1109\/ICASSP.1998.675350"},{"key":"36_CR16","doi-asserted-by":"crossref","unstructured":"McDonough, J., Schaaf, T., Waibel, A.: On Maximum Mutual Information Speaker-Adapted Training. In: ICASSP (2002)","DOI":"10.1109\/ICASSP.2002.1005811"},{"key":"36_CR17","doi-asserted-by":"crossref","unstructured":"Fisher, W.M.: A Statistical Text-to-Phone Function Using Ngrams and Rules. In: ICASSP (1999)","DOI":"10.1109\/ICASSP.1999.759750"},{"key":"36_CR18","doi-asserted-by":"crossref","unstructured":"Stolcke, A.: SRILM \u2013 An Extensible Language Modeling Toolkit. In: ICSLP (2002)","DOI":"10.21437\/ICSLP.2002-303"},{"key":"36_CR19","unstructured":"Chen, S.F., Goodman, J.: An Empirical Study of Smoothing Techniques for Language Modeling. Computer Science Group, Harvard University, Tech. Rep. TR-10-98 (1998)"},{"key":"36_CR20","doi-asserted-by":"crossref","unstructured":"Bulyko, I., Ostendorf, M., Stolcke, A.: Getting more Mileage from Web Text Sources for Conversational Speech Language Modeling using Class-Dependent Mixtures. In: Proc. HLT-NAACL (2003)","DOI":"10.3115\/1073483.1073486"},{"key":"36_CR21","unstructured":"\u00c7etin, \u00d6., Stolcke, A.: Language Modeling in the ICSI-SRI Spring 2005 Meeting Speech Recognition Evaluation System. International Computer Science Institute, Berkeley, CA, USA, Tech. Rep. TR-05-006 (2005)"},{"key":"36_CR22","doi-asserted-by":"crossref","unstructured":"Venkataraman, A., Wang, W.: Techniques for Effective Vocabulary Selection. In: Proc. Eurospeech (2003)","DOI":"10.21437\/Eurospeech.2003-110"},{"key":"36_CR23","unstructured":"Black, A.W., Taylor, P.A.: The Festival Speech Synthesis System: System documentation. Human Communciation Research Centre, University of Edinburgh, Edinburgh, Scotland, United Kongdom, Tech. Rep. HCRC\/TR-83 (1997)"},{"key":"36_CR24","unstructured":"Zhan, P., Westphal, M.: Speaker Normalization Based on Frequency Warping. In: ICASSP (1997)"},{"key":"36_CR25","unstructured":"Gales, M.J.F.: Maximum Likelihood Linear Transformations for HMM-based Speech Recognition. Cambridge University, Cambridge, United Kingdom, Tech. Rep. (1997)"},{"key":"36_CR26","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1006\/csla.1995.0010","volume":"9","author":"C.J. Leggetter","year":"1995","unstructured":"Leggetter, C.J., Woodland, P.C.: Maximum Likelihood Linear Regression for Speaker Adaptation of Continuous Density Hidden Markov Models. Computer Speech and Language\u00a09, 171\u2013185 (1995)","journal-title":"Computer Speech and Language"},{"key":"36_CR27","unstructured":"Yu, H., Tam, Y.-C., Schaaf, T., St\u00fcker, S., Jin, Q., Noamany, M., Schultz, T.: The ISL RT04 Mandarin Broadcast News Evaluation System. In: EARS Rich Transcription Workshop (2004)"},{"key":"36_CR28","doi-asserted-by":"crossref","unstructured":"Lamel, L., Gauvain, J.-L.: Alternate Phone Models for Conversational Speech. In: ICASSP (2005)","DOI":"10.1109\/ICASSP.2005.1415286"},{"key":"36_CR29","doi-asserted-by":"crossref","unstructured":"Mangu, L., Brill, E., Stolcke, A.: Finding Consensus among Words: Lattice-based Word Error Minimization. In: EUROSPEECH (1999)","DOI":"10.21437\/Eurospeech.1999-127"},{"key":"36_CR30","doi-asserted-by":"crossref","unstructured":"W\u00f6lfel, M., F\u00fcgen, C., Ikbal, S., McDonough, J.W.: Multi-Source Far-Distance Microphone Selection and Combination for Automatic Transcription of Lectures. In: INTERSPEECH (2006)","DOI":"10.21437\/Interspeech.2006-122"},{"key":"36_CR31","unstructured":"CHIL \u2013 Computers in the Human Interaction Loop, http:\/\/chil.server.de"}],"container-title":["Lecture Notes in Computer Science","Machine Learning for Multimodal Interaction"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/11965152_36.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,10]],"date-time":"2023-05-10T09:50:22Z","timestamp":1683712222000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/11965152_36"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2006]]},"ISBN":["9783540692676","9783540692683"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/11965152_36","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2006]]}}}