{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T07:06:29Z","timestamp":1725865589025},"publisher-location":"Cham","reference-count":40,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319459240"},{"type":"electronic","value":"9783319459257"}],"license":[{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016]]},"DOI":"10.1007\/978-3-319-45925-7_10","type":"book-chapter","created":{"date-parts":[[2016,9,20]],"date-time":"2016-09-20T10:11:34Z","timestamp":1474366294000},"page":"120-132","source":"Crossref","is-referenced-by-count":0,"title":["A New Perspective on Combining GMM and DNN Frameworks for Speaker Adaptation"],"prefix":"10.1007","author":[{"given":"Natalia","family":"Tomashenko","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuri","family":"Khokhlov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yannick","family":"Est\u00e8ve","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,9,21]]},"reference":[{"issue":"2","key":"10_CR1","doi-asserted-by":"crossref","first-page":"75","DOI":"10.1006\/csla.1998.0043","volume":"12","author":"MJ Gales","year":"1998","unstructured":"Gales, M.J.: Maximum likelihood linear transformations for HMM-based speech recognition. Comput. Speech Lang. 12(2), 75\u201398 (1998)","journal-title":"Comput. Speech Lang."},{"key":"10_CR2","doi-asserted-by":"crossref","first-page":"291","DOI":"10.1109\/89.279278","volume":"2","author":"J-L Gauvain","year":"1994","unstructured":"Gauvain, J.-L., Lee, C.-H.: Maximum a posteriori estimation for multivariate Gaussian mixture observations of Markov chains. IEEE Trans. Speech Audio Process. 2, 291\u2013298 (1994)","journal-title":"IEEE Trans. Speech Audio Process."},{"doi-asserted-by":"crossref","unstructured":"Gemello, R., Mana, F., Scanzio, S., Laface, P., De Mori, R.: Adaptation of hybrid ANN\/HMM models using linear hidden transformations and conservative training. In: Proceedings of ICASSP (2006)","key":"10_CR3","DOI":"10.1109\/ICASSP.2006.1660239"},{"doi-asserted-by":"crossref","unstructured":"Li, B., Sim, K.C.: Comparison of discriminative input and output transformations for speaker adaptation in the hybrid NN\/HMM systems, pp. 526\u2013529 (2010)","key":"10_CR4","DOI":"10.21437\/Interspeech.2010-214"},{"doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., Yu, D.: Feature engineering in context-dependent deep neural networks for conversational speech transcription. In: Proceedings of ASRU, pp. 24\u201329. IEEE (2011)","key":"10_CR5","DOI":"10.1109\/ASRU.2011.6163899"},{"doi-asserted-by":"crossref","unstructured":"Yao, K., Yu, D., Seide, F., Su, H., Deng, L., Gong, Y.: Adaptation of context-dependent deep neural networks for automatic speech recognition. In: Proceedings of SLT, pp. 366\u2013369. IEEE (2012)","key":"10_CR6","DOI":"10.1109\/SLT.2012.6424251"},{"doi-asserted-by":"crossref","unstructured":"Liao, H.: Speaker adaptation of context dependent deep neural networks. In: Proceedings of ICASSP, pp. 7947\u20137951. IEEE (2013)","key":"10_CR7","DOI":"10.1109\/ICASSP.2013.6639212"},{"doi-asserted-by":"crossref","unstructured":"Yu, D., Yao, K., Su, H., Li, G., Seide, F.: KL-divergence regularized deep neural network adaptation for improved large vocabulary speech recognition. In: Proceedings of ICASSP, pp. 7893\u20137897 (2013)","key":"10_CR8","DOI":"10.1109\/ICASSP.2013.6639201"},{"doi-asserted-by":"crossref","unstructured":"Albesano, D., Gemello, R., Laface, P., Mana, F., Scanzio, S.: Adaptation of artificial neural networks avoiding catastrophic forgetting. In: Proceedings of IJCNN 2006, pp. 1554\u20131561. IEEE (2006)","key":"10_CR9","DOI":"10.1109\/IJCNN.2006.246618"},{"doi-asserted-by":"crossref","unstructured":"Ochiai, T., Matsuda, S., Lu, X., Hori, C., Katagiri, S.: Speaker adaptive training using deep neural networks. In: Proceedings of ICASSP, pp. 6349\u20136353. IEEE (2014)","key":"10_CR10","DOI":"10.1109\/ICASSP.2014.6854826"},{"issue":"10","key":"10_CR11","doi-asserted-by":"crossref","first-page":"2152","DOI":"10.1109\/TASL.2013.2270370","volume":"21","author":"SM Siniscalchi","year":"2013","unstructured":"Siniscalchi, S.M., Li, J., Lee, C.-H.: Hermitian polynomial for speaker adaptation of connectionist speech recognition systems. IEEE Trans. Audio Speech Lang. Process. 21(10), 2152\u20132161 (2013)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Renals, S.: Learning hidden unit contributions for unsupervised speaker adaptation of neural network acoustic models. In: Proceedings of SLT, pp. 171\u2013176. IEEE (2014)","key":"10_CR12","DOI":"10.1109\/SLT.2014.7078569"},{"doi-asserted-by":"crossref","unstructured":"Huang, Z., Li, J., Siniscalchi, S.M., Chen, I.-F., Weng, C., Lee, C.-H.: Feature space maximum a posteriori linear regression for adaptation of deep neural networks. In: Proceedings of INTERSPEECH, pp. 2992\u20132996 (2014)","key":"10_CR13","DOI":"10.21437\/Interspeech.2014-500"},{"doi-asserted-by":"crossref","unstructured":"Huang, Z., Siniscalchi, S.M., Chen, I.-F., Li, J., Wu, J., Lee, C.-H.: Maximum a posteriori adaptation of network parameters in deep models. In: Proceedings of INTERSPEECH (2015)","key":"10_CR14","DOI":"10.21437\/Interspeech.2015-285"},{"doi-asserted-by":"crossref","unstructured":"Li, S., Lu, X., Akita, Y., Kawahara, T.: Ensemble speaker modeling using speaker adaptive training deep neural network for speaker adaptation. In: Proceedings of INTERSPEECH (2015)","key":"10_CR15","DOI":"10.21437\/Interspeech.2015-608"},{"doi-asserted-by":"crossref","unstructured":"Huang, Z., Li, J., Siniscalchi, S.M., Chen, I.-F., Wu, J., Lee, C.-H.: Rapid adaptation for deep neural networks through multi-task learning. In: Proceedings of INTERSPEECH (2015)","key":"10_CR16","DOI":"10.21437\/Interspeech.2015-719"},{"doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Bell, P., Renals, S.: Structured output layer with auxiliary targets for context-dependent acoustic modelling. In: Proceedings of INTERSPEECH (2015)","key":"10_CR17","DOI":"10.21437\/Interspeech.2015-715"},{"doi-asserted-by":"crossref","unstructured":"Karanasou, P., Wang, Y., Gales, M.J., Woodland, P.C.: Adaptation of deep neural network acoustic models using factorised i-vectors. In: Proceedings of INTERSPEECH, pp. 2180\u20132184 (2014)","key":"10_CR18","DOI":"10.21437\/Interspeech.2014-488"},{"doi-asserted-by":"crossref","unstructured":"Gupta, V., Kenny, P., Ouellet, P., Stafylakis, T.: I-vector-based speaker adaptation of deep neural networks for french broadcast audio transcription. In: Proceedings of ICASSP, pp. 6334\u20136338. IEEE (2014)","key":"10_CR19","DOI":"10.1109\/ICASSP.2014.6854823"},{"doi-asserted-by":"crossref","unstructured":"Senior, A., Lopez-Moreno, I.: Improving DNN speaker independence with i-vector inputs. In: Proceedings of ICASSP, pp. 225\u2013229 (2014)","key":"10_CR20","DOI":"10.1109\/ICASSP.2014.6853591"},{"issue":"12","key":"10_CR21","doi-asserted-by":"crossref","first-page":"1713","DOI":"10.1109\/TASLP.2014.2346313","volume":"22","author":"S Xue","year":"2014","unstructured":"Xue, S., Abdel-Hamid, O., Jiang, H., Dai, L., Liu, Q.: Fast adaptation of deep neural network based on discriminant codes for speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(12), 1713\u20131725 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"doi-asserted-by":"crossref","unstructured":"Li, J., Huang, J.-T., Gong, Y.: Factorized adaptation for deep neural network. In: Proceedings of ICASSP, pp. 5537\u20135541. IEEE (2014)","key":"10_CR22","DOI":"10.1109\/ICASSP.2014.6854662"},{"doi-asserted-by":"crossref","unstructured":"Rath, S.P., Povey, D., Vesel\u1ef3, K., Cernock\u1ef3, J.: Improved feature processing for deep neural networks. In: Proceedings of INTERSPEECH, pp. 109\u2013113 (2013)","key":"10_CR23","DOI":"10.21437\/Interspeech.2013-48"},{"doi-asserted-by":"crossref","unstructured":"Kanagawa, H., Tachioka, Y., Watanabe, S., Ishii, J.: Feature-space structural MAPLR with regression tree-based multiple transformation matrices for DNN (2015)","key":"10_CR24","DOI":"10.1109\/APSIPA.2015.7415425"},{"doi-asserted-by":"crossref","unstructured":"Lei, X., Lin, H., Heigold, G.: Deep neural networks with auxiliary Gaussian mixture models for real-time speech recognition. In: Proceedings of ICASSP, pp. 7634\u20137638. IEEE (2013)","key":"10_CR25","DOI":"10.1109\/ICASSP.2013.6639148"},{"doi-asserted-by":"crossref","unstructured":"Liu, S., Sim, K.C.: On combining DNN and GMM with unsupervised speaker adaptation for robust automatic speech recognition. In: Proceedings of ICASSP, pp. 195\u2013199. IEEE (2014)","key":"10_CR26","DOI":"10.1109\/ICASSP.2014.6853585"},{"doi-asserted-by":"crossref","unstructured":"Murali Karthick, B., Kolhar, P., Umesh, S.: Speaker adaptation of convolutional neural network using speaker specific subspace vectors of SGMM (2015)","key":"10_CR27","DOI":"10.21437\/Interspeech.2015-289"},{"doi-asserted-by":"crossref","unstructured":"Parthasarathi, S.H.K., Hoffmeister, B., Matsoukas, S., Mandal, A., Strom, N., Garimella, S.: fMLLR based feature-space speaker adaptation of DNN acoustic models. In: Proceedings of INTERSPEECH (2015)","key":"10_CR28","DOI":"10.21437\/Interspeech.2015-720"},{"doi-asserted-by":"crossref","unstructured":"Tomashenko, N., Khokhlov, Y.: Speaker adaptation of context dependent deep neural networks based on MAP-adaptation and GMM-derived feature processing. In: Proceedings of INTERSPEECH, pp. 2997\u20133001 (2014)","key":"10_CR29","DOI":"10.21437\/Interspeech.2014-501"},{"key":"10_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"304","DOI":"10.1007\/978-3-319-43958-7_36","volume-title":"Speech and Computer","author":"N Tomashenko","year":"2016","unstructured":"Tomashenko, N., Khokhlov, Y., Larcher, A., Est\u00e8ve, Y.: Exploring GMM-derived features for unsupervised adaptation of deep neural network acoustic models. In: Ronzhin, A., Potapova, R., N\u00e9meth, G. (eds.) SPECOM 2016. LNCS, vol. 9811, pp. 304\u2013311. Springer, Heidelberg (2016). doi: 10.1007\/978-3-319-43958-7_36"},{"doi-asserted-by":"crossref","unstructured":"Tomashenko, N., Khokhlov, Y.: GMM-derived features for effective unsupervised adaptation of deep neural network acoustic models. In: Proceedings of INTERSPEECH, pp. 2882\u20132886 (2015)","key":"10_CR31","DOI":"10.1007\/978-3-319-43958-7_36"},{"unstructured":"Tomashenko, N., Khokhlov, Y., Larcher, A., Est\u00e8ve, Y.: Exploration de param\u00e8tres acoustiques d\u00e9riv\u00e9s de GMM pour l\u2019adaptation non supervis\u00e9e de mod\u00e8les acoustiques \u00e0 base de r\u00e9seaux de neurones profonds. In: Proceedings of 31\u00e9me Journ\u00e9es d\u2019\u00c9tudes sur la Parole (JEP), pp. 337\u2013345 (2016)","key":"10_CR32"},{"doi-asserted-by":"crossref","unstructured":"Tomashenko, N., Khokhlov, Y., Esteve, Y.: On the use of Gaussian mixture model framework to improve speaker adaptation of deep neural network acoustic models. In: Proceedings of INTERSPEECH (2016)","key":"10_CR33","DOI":"10.21437\/Interspeech.2016-1230"},{"doi-asserted-by":"crossref","unstructured":"Pinto, J.P., Hermansky, H.: Combining evidence from a generative and a discriminative model in phoneme recognition. Technical report, IDIAP (2008)","key":"10_CR34","DOI":"10.21437\/Interspeech.2008-132"},{"doi-asserted-by":"crossref","unstructured":"Gr\u00e9zl, F., Karafi\u00e1t, M., Vesely, K.: Adaptation of multilingual stacked bottle-neck neural network structure for new language. In: Proceedings of ICASSP, pp. 7654\u20137658 (2014)","key":"10_CR35","DOI":"10.1109\/ICASSP.2014.6855089"},{"doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Revisiting hybrid and GMM-HMM system combination techniques. In: Proceedings of ICASSP, pp. 6744\u20136748. IEEE (2013)","key":"10_CR36","DOI":"10.1109\/ICASSP.2013.6638967"},{"doi-asserted-by":"crossref","unstructured":"Fiscus, J.G.: A post-processing system to yield reduced word error rates: recognizer output voting error reduction (ROVER). In: Proceedings of ASRU, pp. 347\u2013354. IEEE (1997)","key":"10_CR37","DOI":"10.1109\/ASRU.1997.659110"},{"unstructured":"Evermann, G., Woodland, P.: Posterior probability decoding, confidence estimation and system combination. In: Proceedings of Speech Transcription Workshop, Baltimore, vol. 27 (2000)","key":"10_CR38"},{"unstructured":"Rousseau, A., Del\u00e9glise, P., Est\u00e8ve, Y.: Enhancing the TED-LIUM corpus with selected data for language modeling and more TED talks. In: Proceedings of LREC, pp. 3935\u20133939 (2014)","key":"10_CR39"},{"unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., et al.: The Kaldi speech recognition toolkit. In: Proceedings of ASRU (2011)","key":"10_CR40"}],"container-title":["Lecture Notes in Computer Science","Statistical Language and Speech Processing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-45925-7_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,8]],"date-time":"2022-07-08T17:46:07Z","timestamp":1657302367000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-45925-7_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"ISBN":["9783319459240","9783319459257"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-45925-7_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2016]]}}}