{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T04:14:43Z","timestamp":1750997683499,"version":"3.41.0"},"publisher-location":"Cham","reference-count":39,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_10","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"245-260","source":"Crossref","is-referenced-by-count":3,"title":["Training Data Augmentation and Data Selection"],"prefix":"10.1007","author":[{"given":"Martin","family":"Karafi\u00e1t","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Karel","family":"Vesel\u00fd","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kate\u0159ina","family":"\u017dmol\u00edkov\u00e1","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marc","family":"Delcroix","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luk\u00e1\u0161","family":"Burget","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jan","family":"\u201cHonza\u201d\u010cernock\u00fd","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Igor","family":"Sz\u0151ke","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"10_CR1","unstructured":"Ager, M., Cvetkovic, Z., Sollich, P., Bin, Y.: Towards robust phoneme classification: augmentation of PLP models with acoustic waveforms. In: 16th European Signal Processing Conference, 2008, pp.\u00a01\u20135 (2008)"},{"key":"10_CR2","volume-title":"Unsupervised discovery and training of maximally dissimilar cluster models","author":"F. Beaufays","year":"2010","unstructured":"Beaufays, F., Vanhoucke, V., Strope, B.: Unsupervised discovery and training of maximally dissimilar cluster models. In: Proceedings of Interspeech (2010)"},{"issue":"3","key":"10_CR3","doi-asserted-by":"publisher","first-page":"413","DOI":"10.1109\/89.294355","volume":"2","author":"J.R. Bellegarda","year":"1994","unstructured":"Bellegarda, J.R., de\u00a0Souza, P.V., Nadas, A., Nahamoo, D., Picheny, M.A., Bahl, L.R.: The metamorphic algorithm: a speaker mapping approach to data augmentation. IEEE Trans. Speech Audio Process. 2(3), 413\u2013420 (1994). doi: 10.1109\/89.294355","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"10_CR4","doi-asserted-by":"publisher","unstructured":"Bellegarda, J., de\u00a0Souza, P., Nahamoo, D., Padmanabhan, M., Picheny, M., Bahl, L.: Experiments using data augmentation for speaker adaptation. In: International Conference on Acoustics, Speech, and Signal Processing, 1995, ICASSP-95, vol.\u00a01, pp.\u00a0692\u2013695 (1995). doi: 10.1109\/ICASSP.1995.479788","DOI":"10.1109\/ICASSP.1995.479788"},{"issue":"9","key":"10_CR5","doi-asserted-by":"publisher","first-page":"1469","DOI":"10.1109\/TASLP.2015.2438544","volume":"23","author":"X. Cui","year":"2015","unstructured":"Cui, X., Goel, V., Kingsbury, B.: Data augmentation for deep neural network acoustic modeling. IEEE\/ACM Trans. Audio Speech Lang. Process. 23(9), 1469\u20131477 (2015). doi: 10.1109\/TASLP.2015.2438544","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10_CR6","doi-asserted-by":"publisher","unstructured":"Dehak, N., Kenny, P., Dehak, R., Dumouchel, P., Ouellet, P.: Front-end factor analysis for speaker verification. IEEE Trans. Audio Speech Lang. Process. 19(4), 788\u2013798 (2011). doi: 10.1109\/TASL.2010.2064307 . http:\/\/dx.doi.org\/10.1109\/TASL.2010.2064307","DOI":"10.1109\/TASL.2010.2064307"},{"key":"10_CR7","unstructured":"Delcroix, M., Yoshioka, T., Ogawa, A., Kubo, Y., Fujimoto, M., Ito, N., Kinoshita, K., Espi, M., Hori, T., Nakatani, T., Nakamura, A.: Linear prediction-based dereverberation with advanced speech enhancement and recognition technologies for the REVERB challenge. In: Proceedings of REVERB\u201914 (2014)"},{"key":"10_CR8","doi-asserted-by":"crossref","unstructured":"Delcroix, M., Yoshioka, T., Ogawa, A., Kubo, Y., Fujimoto, M., Ito, N., Kinoshita, K., Espi, M., Araki, S., Hori, T., Nakatani, T.: Strategies for distant speech recognition in reverberant environments. EURASIP J. Adv. Signal Process. 2015, Article ID 60, 15 pp. (2015)","DOI":"10.1186\/s13634-015-0245-7"},{"key":"10_CR9","doi-asserted-by":"crossref","unstructured":"Egorova, E., Vesel\u00fd, K., Karafi\u00e1t, M., Janda, M., \u010cernock\u00fd, J.: Manual and semi-automatic approaches to building a multilingual phoneme set. In: Proceedings of ICASSP 2013, pp.\u00a07324\u20137328. IEEE Signal Processing Society, Piscataway (2013). http:\/\/www.fit.vutbr.cz\/research\/view_pub.php?id=10323","DOI":"10.1109\/ICASSP.2013.6639085"},{"key":"10_CR10","volume-title":"Model-Based Techniques for Noise Robust Speech Recognition","author":"M.J.F. Gales","year":"1995","unstructured":"Gales, M.J.F., College, C.: Model-Based Techniques for Noise Robust Speech Recognition. University of Cambridge, Cambridge (1995)"},{"key":"10_CR11","volume-title":"Adaptive Filter Theory","author":"S. Haykin","year":"1996","unstructured":"Haykin, S.: Adaptive Filter Theory, 3rd edn. Prentice-Hall, Upper Saddle River, NJ (1996)","edition":"3"},{"key":"10_CR12","unstructured":"Hinton, G., Bengio, Y.: Visualizing data using t-SNE. In: Cost-Sensitive Machine Learning for Information Retrieval 33 (2008)"},{"key":"10_CR13","unstructured":"Hu, Y., Loizou, P.C.: Subjective comparison of speech enhancement algorithms. In: Proceedings of IEEE International Conference on Speech and Signal Processing, pp.\u00a0153\u2013156 (2006)"},{"key":"10_CR14","unstructured":"Jaitly, N., Hinton, G.E.: Vocal tract length perturbation (VTLP) improves speech recognition. In: Proceedings of the 30th International Conference on Machine Learning, Atlanta, GA (2013)"},{"key":"10_CR15","doi-asserted-by":"publisher","unstructured":"Kalinli, O., Seltzer, M.L., Acero, A.: Noise adaptive training using a vector Taylor series approach for noise robust automatic speech recognition. In: Proceedings of the 2009 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP\u201909, pp.\u00a03825\u20133828. IEEE Computer Society, Washington (2009) doi: 10.1109\/ICASSP.2009.4960461 . http:\/\/dx.doi.org\/10.1109\/ICASSP.2009.4960461","DOI":"10.1109\/ICASSP.2009.4960461"},{"key":"10_CR16","unstructured":"Karafi\u00e1t, M., Burget, L., Mat\u011bjka, P., Glembek, O., \u010cernock\u00fd, J.: iVector-based discriminative adaptation for automatic speech recognition. In: Proceedings of ASRU 2011, pp.\u00a0152\u2013157. IEEE Signal Processing Society, Piscataway (2011). http:\/\/www.fit.vutbr.cz\/research\/view_pub.php?id=9762"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Karafi\u00e1t, M., Vesel\u00fd, K., Sz\u0151ke, I., Burget, L., Gr\u00e9zl, F., Hannemann, M., \u010cernock\u00fd, J.: BUT ASR system for BABEL surprise evaluation 2014. In: Proceedings of 2014 Spoken Language Technology Workshop, pp.\u00a0501\u2013506. IEEE Signal Processing Society, Piscataway (2014). http:\/\/www.fit.vutbr.cz\/research\/view_pub.php?id=10799","DOI":"10.1109\/SLT.2014.7078625"},{"key":"10_CR18","unstructured":"Karafi\u00e1t, M., Gr\u00e9zl, F., Burget, L., Sz\u0151ke, I., \u010cernock\u00fd, J.: Three ways to adapt a CTS recognizer to unseen reverberated speech in BUT system for the ASpIRE challenge. In: Proceedings of Interspeech 2015, pp.\u00a02454\u20132458. International Speech Communication Association, Grenoble (2015). http:\/\/www.fit.vutbr.cz\/research\/view_pub.php?id=10972"},{"issue":"4","key":"10_CR19","doi-asserted-by":"crossref","first-page":"534","DOI":"10.1109\/TASL.2008.2009015","volume":"17","author":"K. Kinoshita","year":"2009","unstructured":"Kinoshita, K., Delcroix, M., Nakatani, T., Miyoshi, M.: Suppression of late reverberation effect on speech signal using long-term multiple-step linear prediction. IEEE Trans. Audio Speech Lang. Process. 17(4), 534\u2013545 (2009)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Ko, T., Peddinti, V., Povey, D., Khudanpur, S.: Audio augmentation for speech recognition. In: INTERSPEECH, pp.\u00a03586\u20133589. ISCA, Grenoble (2015)","DOI":"10.21437\/Interspeech.2015-711"},{"key":"10_CR21","doi-asserted-by":"crossref","unstructured":"Nakatani, T., Yoshioka, T., Kinoshita, K., Miyoshi, M., Juang, B.H.: Blind speech dereverberation with multi-channel linear prediction based on short time Fourier transform representation. In: Proceedings of ICASSP\u201908, pp.\u00a085\u201388 (2008)","DOI":"10.1109\/ICASSP.2008.4517552"},{"key":"10_CR22","doi-asserted-by":"crossref","unstructured":"Ogata, K., Tachibana, M., Yamagishi, J., Kobayashi, T.: Acoustic model training based on linear transformation and MAP modification for HSMM-based speech synthesis. In: INTERSPEECH, pp.\u00a01328\u20131331 (2006)","DOI":"10.21437\/Interspeech.2006-389"},{"key":"10_CR23","unstructured":"Ragni, A., Knill, K.M., Rath, S.P., Gales, M.J.F.: Data augmentation for low resource languages. In: INTERSPEECH 2014, 15th Annual Conference of the International Speech Communication Association, Singapore, September 14\u201318, 2014, pp. 810\u2013814 (2014)"},{"key":"10_CR24","doi-asserted-by":"crossref","unstructured":"Saon, G., Soltau, H., Nahamoo, D., Picheny, M.: Speaker adaptation of neural network acoustic models using i-vectors. In: 2013 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp.\u00a055\u201359. IEEE, New York (2013)","DOI":"10.1109\/ASRU.2013.6707705"},{"key":"10_CR25","doi-asserted-by":"crossref","unstructured":"Siohan, O., Bacchiani, M.: iVector-based acoustic data selection. In: Proceedings of INTERSPEECH, pp.\u00a0657\u2013661 (2013)","DOI":"10.21437\/Interspeech.2013-188"},{"key":"10_CR26","doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Hybrid acoustic models for distant and multichannel large vocabulary speech recognition. In: 2013 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU). IEEE, New York (2013)","DOI":"10.1109\/ASRU.2013.6707744"},{"key":"10_CR27","unstructured":"Tokuda, K., Zen, H., Black, A.: An HMM-based approach to multilingual speech synthesis. In: Text to Speech Synthesis: New Paradigms and Advances, pp. 135\u2013153. Prentice Hall, Upper Saddle River (2004)"},{"key":"10_CR28","unstructured":"Vesel\u00fd, K., Ghoshal, A., Burget, L., Povey, D.: Sequence-discriminative training of deep neural networks. In: Proceedings of INTERSPEECH 2013, pp.\u00a02345\u20132349. International Speech Communication Association, Grenoble (2013). http:\/\/www.fit.vutbr.cz\/research\/view_pub.php?id=10422"},{"key":"10_CR29","volume-title":"Sequence summarizing neural network for speaker adaptation","author":"K. Vesel\u00fd","year":"2016","unstructured":"Vesel\u00fd, K., Watanabe, S., \u017dmol\u00edkov\u00e1, K., Karafi\u00e1t, M., Burget, L., \u010cernock\u00fd, J.: Sequence summarizing neural network for speaker adaptation. In: Proceedings of ICASSP (2016)"},{"issue":"7","key":"10_CR30","doi-asserted-by":"crossref","first-page":"2149","DOI":"10.1109\/TASL.2012.2198059","volume":"20","author":"Y. Wang","year":"2012","unstructured":"Wang, Y., Gales, M.J.F.: Speaker and noise factorization for robust speech recognition. IEEE Trans. Audio Speech Lang. Process. 20(7), 2149\u20132158 (2012). http:\/\/dblp.uni-trier.de\/db\/journals\/taslp\/taslp20.html#WangG12","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10_CR31","doi-asserted-by":"crossref","unstructured":"Wei, K., Liu, Y., Kirchhoff, K., Bartels, C., Bilmes, J.: Submodular subset selection for large-scale speech training data. In: Proceedings of ICASSP, pp.\u00a03311\u20133315 (2014)","DOI":"10.1109\/ICASSP.2014.6854213"},{"key":"10_CR32","doi-asserted-by":"crossref","unstructured":"Wu, Y., Zhang, R., Rudnicky, A.: Data selection for speech recognition. In: Proceedings of ASRU, pp.\u00a0562\u2013565 (2007)","DOI":"10.1109\/ASRU.2007.4430173"},{"issue":"1","key":"10_CR33","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1109\/LSP.2013.2291240","volume":"21","author":"Y. Xu","year":"2014","unstructured":"Xu, Y., Du, J., Dai, L.R., Lee, C.H.: An experimental study on speech enhancement based on deep neural networks. IEEE Signal Process Lett. 21(1), 65\u201368 (2014)","journal-title":"IEEE Signal Process Lett."},{"key":"10_CR34","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Masuko, T., Tokuda, K., Kobayashi, T., Kitamura, T.: Speaker interpolation in HMM-based speech synthesis system. In: Eurospeech, pp.\u00a02523\u20132526 (1997)","DOI":"10.21437\/Eurospeech.1997-655"},{"issue":"10","key":"10_CR35","doi-asserted-by":"crossref","first-page":"2707","DOI":"10.1109\/TASL.2012.2210879","volume":"20","author":"T. Yoshioka","year":"2012","unstructured":"Yoshioka, T., Nakatani, T.: Generalization of multi-channel linear prediction methods for blind MIMO impulse response shortening. IEEE Trans. Audio Speech Lang. Process. 20(10), 2707\u20132720 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"1","key":"10_CR36","doi-asserted-by":"crossref","first-page":"69","DOI":"10.1109\/TASL.2010.2045183","volume":"19","author":"T. Yoshioka","year":"2011","unstructured":"Yoshioka, T., Nakatani, T., Miyoshi, M., Okuno, H.G.: Blind separation and dereverberation of speech mixtures by joint optimization. IEEE Trans. Audio Speech Lang. Process. 19(1), 69\u201384 (2011)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10_CR37","doi-asserted-by":"crossref","unstructured":"Yoshioka, T., Chen, X., Gales, M.J.F.: Impact of single-microphone dereverberation on DNN-based meeting transcription systems. In: Proceedings of ICASSP\u201914 (2014)","DOI":"10.1109\/ICASSP.2014.6854660"},{"key":"10_CR38","doi-asserted-by":"crossref","unstructured":"Yoshioka, T., Ito, N., Delcroix, M., Ogawa, A., Kinoshita, K., Fujimoto, M., Yu, C., Fabian, W.J., Espi, M., Higuchi, T., Araki, S., Nakatani, T.: The NTT CHiME-3 system: advances in speech enhancement and recognition for mobile multi-microphone devices. In: Proceedings of ASRU\u201915, pp.\u00a0436\u2013443 (2015)","DOI":"10.1109\/ASRU.2015.7404828"},{"key":"10_CR39","doi-asserted-by":"crossref","unstructured":"Zavaliagkos, G., Siu, M.-H., Colthurst, T., Billa, J.: Using untranscribed training data to improve performance. In: The 5th International Conference on Spoken Language Processing, Incorporating the 7th Australian International Speech Science and Technology Conference, Sydney, Australia, 30 November\u20134 December 1998 (1998)","DOI":"10.21437\/ICSLP.1998-679"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T20:15:57Z","timestamp":1750968957000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_10","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}