{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,27]],"date-time":"2025-07-27T07:50:26Z","timestamp":1753602626309,"version":"3.37.3"},"publisher-location":"Cham","reference-count":42,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319664286"},{"type":"electronic","value":"9783319664293"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-66429-3_47","type":"book-chapter","created":{"date-parts":[[2017,8,12]],"date-time":"2017-08-12T02:02:55Z","timestamp":1502503375000},"page":"473-482","source":"Crossref","is-referenced-by-count":7,"title":["Language Adaptive Multilingual CTC Speech Recognition"],"prefix":"10.1007","author":[{"given":"Markus","family":"M\u00fcller","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sebastian","family":"St\u00fcker","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alex","family":"Waibel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,8,13]]},"reference":[{"key":"47_CR1","unstructured":"PyTorch. http:\/\/pytorch.org . Accessed 13 Apr 2017"},{"key":"47_CR2","unstructured":"warp-ctc. https:\/\/github.com\/baidu-research\/warp-ctc . Accessed 13 Apr 2017"},{"key":"47_CR3","unstructured":"Woszczyna, M., et al.: JANUS 93: towards spontaneous speech translation. In: International Conference on Acoustics, Speech, and Signal Processing 1994, Adelaide, Australia (1994)"},{"key":"47_CR4","unstructured":"Amodei, D., Anubhai, R., Battenberg, E., Case, C., Casper, J., Catanzaro, B., Chen, J., Chrzanowski, M., Coates, A., Diamos, G., et al.: Deep speech 2: end-to-end speech recognition in english and mandarin. arXiv preprint (2015). arXiv:1512.02595"},{"issue":"1","key":"47_CR5","doi-asserted-by":"crossref","first-page":"41","DOI":"10.1023\/A:1007379606734","volume":"28","author":"R Caruana","year":"1997","unstructured":"Caruana, R.: Multitask learning. Mach. Learn. 28(1), 41\u201375 (1997)","journal-title":"Mach. Learn."},{"key":"47_CR6","doi-asserted-by":"crossref","unstructured":"Chen, D., Mak, B., Leung, C.C., Sivadas, S.: Joint acoustic modeling of triphones and trigraphemes by multi-task learning deep neural networks for low-resource speech recognition. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5592\u20135596. IEEE (2014)","DOI":"10.1109\/ICASSP.2014.6854673"},{"key":"47_CR7","unstructured":"Collobert, R., Kavukcuoglu, K., Farabet, C.: Torch7: a Matlab-like environment for machine learning. In: BigLearn, NIPS Workshop (2011)"},{"key":"47_CR8","doi-asserted-by":"crossref","unstructured":"Ghoshal, A., Swietojanski, P., Renals, S.: Multilingual training of deep-neural networks. In: Proceedings of the ICASSP, Vancouver, Canada (2013)","DOI":"10.1109\/ICASSP.2013.6639084"},{"key":"47_CR9","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., Schmidhuber, J.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proceedings of the 23rd International Conference On Machine Learning, pp. 369\u2013376. ACM (2006)","DOI":"10.1145\/1143844.1143891"},{"key":"47_CR10","doi-asserted-by":"crossref","unstructured":"Gretter, R.: Euronews: a multilingual benchmark for ASR and LID. In: Fifteenth Annual Conference of the International Speech Communication Association (2014)","DOI":"10.21437\/Interspeech.2014-381"},{"key":"47_CR11","doi-asserted-by":"crossref","unstructured":"Gr\u00e9zl, F., Karafi\u00e1t, M., Vesely, K.: Adaptation of multilingual stacked bottle-neck neural network structure for new language. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7654\u20137658. IEEE (2014)","DOI":"10.1109\/ICASSP.2014.6855089"},{"key":"47_CR12","doi-asserted-by":"crossref","unstructured":"Heigold, G., Vanhoucke, V., Senior, A., Nguyen, P., Ranzato, M., Devin, M., Dean, J.: Multilingual acoustic models using distributed deep neural networks. In: Proceedings of the ICASSP, Vancouver, Canada, May 2013","DOI":"10.1109\/ICASSP.2013.6639348"},{"issue":"7","key":"47_CR13","doi-asserted-by":"crossref","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"GE Hinton","year":"2006","unstructured":"Hinton, G.E., Osindero, S., Teh, Y.W.: A fast learning algorithm for deep belief nets. Neural Comput. 18(7), 1527\u20131554 (2006)","journal-title":"Neural Comput."},{"issue":"8","key":"47_CR14","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"47_CR15","doi-asserted-by":"crossref","unstructured":"Huang, H., Sim, K.C.: An investigation of augmenting speaker representations to improve speaker normalisation for DNN-based speech recognition. In: ICASSP, pp. 4610\u20134613. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178844"},{"key":"47_CR16","unstructured":"Kim, S., Hori, T., Watanabe, S.: Joint ctc-attention based end-to-end speech recognition using multi-task learning. arXiv preprint (2016). arXiv:1609.06773"},{"key":"47_CR17","unstructured":"Laskowski, K., Heldner, M., Edlund, J.: The fundamental frequency variation spectrum. In: Proceedings of the 21st Swedish Phonetics Conference (Fonetik 2008), pp. 29\u201332, Gothenburg, Sweden, June 2008"},{"key":"47_CR18","unstructured":"Lu, L., Kong, L., Dyer, C., Smith, N.A.: Multi-task learning with ctc and segmental crf for speech recognition. arXiv preprint (2017). arXiv:1702.06378"},{"key":"47_CR19","doi-asserted-by":"crossref","unstructured":"Metze, F., Sheikh, Z., Waibel, A., Gehring, J., Kilgour, K., Nguyen, Q.B., Nguyen, V.H., et al.: Models of Tone for tonal and non-tonal languages. In: 2013 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 261\u2013266. IEEE (2013)","DOI":"10.1109\/ASRU.2013.6707740"},{"key":"47_CR20","doi-asserted-by":"crossref","unstructured":"Miao, Y., Gowayyed, M., Metze, F.: EESEN: end-to-end speech recognition using deep RNN models and WFST-based decoding. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp. 167\u2013174. IEEE (2015)","DOI":"10.1109\/ASRU.2015.7404790"},{"key":"47_CR21","doi-asserted-by":"crossref","unstructured":"Miao, Y., Zhang, H., Metze, F.: Towards speaker adaptive training of deep neural network acoustic models (2014)","DOI":"10.1109\/SLT.2014.7078568"},{"key":"47_CR22","doi-asserted-by":"crossref","unstructured":"Mohan, A., Rose, R.: Multi-lingual speech recognition with low-rank multi-task deep neural networks. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4994\u20134998. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178921"},{"key":"47_CR23","doi-asserted-by":"crossref","unstructured":"M\u00fcller, M., St\u00fcker, S., Waibel, A.: Language Adaptive DNNs for improved low resource speech recognition. In: Interspeech (2016)","DOI":"10.21437\/Interspeech.2016-1143"},{"key":"47_CR24","unstructured":"M\u00fcller, M., St\u00fcker, S., Waibel, A.: Language feature vectors for resource constraint speech recognition. In: ITG Symposium, Proceedings of Speech Communication, vol. 12. VDE (2016)"},{"key":"47_CR25","unstructured":"M\u00fcller, M., Waibel, A.: Using language adaptive deep neural networks for improved multilingual speech recognition. In: IWSLT (2015)"},{"key":"47_CR26","unstructured":"Sak, H., Rao, K.: Multi-accent speech recognition with hierarchical grapheme based models (2017)"},{"key":"47_CR27","unstructured":"Saon, G., Kurata, G., Sercu, T., Audhkhasi, K., Thomas, S., Dimitriadis, D., Cui, X., Ramabhadran, B., Picheny, M., Lim, L.L., et al.: English conversational telephone speech recognition by humans and machines. arXiv preprint (2017). arXiv:1703.02136"},{"key":"47_CR28","doi-asserted-by":"crossref","unstructured":"Saon, G., Soltau, H., Nahamoo, D., Picheny, M.: Speaker adaptation of neural network acoustic models using i-Vectors. In: ASRU, pp. 55\u201359. IEEE (2013)","DOI":"10.1109\/ASRU.2013.6707705"},{"key":"47_CR29","doi-asserted-by":"crossref","unstructured":"Scanzio, S., Laface, P., Fissore, L., Gemello, R., Mana, F.: On the use of a multilingual neural network front-end. In: Proceedings of the Interspeech, pp. 2711\u20132714 (2008)","DOI":"10.21437\/Interspeech.2008-672"},{"issue":"4","key":"47_CR30","doi-asserted-by":"crossref","first-page":"365","DOI":"10.1023\/A:1025708916924","volume":"6","author":"M Schr\u00f6der","year":"2003","unstructured":"Schr\u00f6der, M., Trouvain, J.: The German text-to-speech synthesis system MARY: a tool for research, development and teaching. Int. J. Speech Technol. 6(4), 365\u2013377 (2003)","journal-title":"Int. J. Speech Technol."},{"key":"47_CR31","unstructured":"Schubert, K.: Grundfrequenzverfolgung und deren Anwendung in der Spracherkennung. Master\u2019s thesis, Universit\u00e4t Karlsruhe (TH), Germany (1999) (in German)"},{"key":"47_CR32","doi-asserted-by":"crossref","unstructured":"Schultz, T., Waibel, A.: Fast bootstrapping of lvcsr systems with multilingual phoneme sets. In: Eurospeech (1997)","DOI":"10.21437\/Eurospeech.1997-141"},{"issue":"1","key":"47_CR33","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1016\/S0167-6393(00)00094-7","volume":"35","author":"T Schultz","year":"2001","unstructured":"Schultz, T., Waibel, A.: Language-independent and language-adaptive acoustic modeling for speech recognition. Speech Commun. 35(1), 31\u201351 (2001)","journal-title":"Speech Commun."},{"key":"47_CR34","unstructured":"Soltau, H., Liao, H., Sak, H.: Neural speech recognizer: acoustic-to-word lstm model for large vocabulary speech recognition. arXiv preprint (2016). arXiv:1610.09975"},{"key":"47_CR35","doi-asserted-by":"crossref","unstructured":"Soltau, H., Metze, F., Fugen, C., Waibel, A.: A One-pass decoder based on polymorphic linguistic context assignment. In: IEEE Workshop on Automatic Speech Recognition and Understanding, ASRU 2001, pp. 214\u2013217. IEEE (2001)","DOI":"10.1109\/ASRU.2001.1034625"},{"key":"47_CR36","unstructured":"St\u00fcker, S.: Acoustic modelling for under-resourced languages. Ph.D. thesis, Karlsruhe University, Dissertation (2009)"},{"key":"47_CR37","unstructured":"Sutskever, I., Martens, J., Dahl, G., Hinton, G.: On the importance of initialization and momentum in deep learning. In: Proceedings of the 30th International Conference on Machine Learning (ICML-2013), pp. 1139\u20131147 (2013)"},{"key":"47_CR38","doi-asserted-by":"crossref","unstructured":"Swietojanski, P., Ghoshal, A., Renals, S.: Unsupervised cross-lingual knowledge transfer in DNN-based LVCSR. In: SLT, pp. 246\u2013251. IEEE (2012)","DOI":"10.1109\/SLT.2012.6424230"},{"key":"47_CR39","doi-asserted-by":"crossref","unstructured":"Vesely, K., Karafiat, M., Grezl, F., Janda, M., Egorova, E.: The language-independent bottleneck features. In: Proceedings of the Spoken Language Technology Workshop (SLT), pp. 336\u2013341. IEEE (2012)","DOI":"10.1109\/SLT.2012.6424246"},{"key":"47_CR40","unstructured":"Waibel, A., Hanazawa, T., Hinton, G., Shikano, K.: Phoneme recognition using time-delay neural networks. In: ATR Interpreting Telephony Research Laboratories, 30 October 1987"},{"key":"47_CR41","doi-asserted-by":"crossref","unstructured":"Wheatley, B., Kondo, K., Anderson, W., Muthusamy, Y.: An evaluation of cross-language adaptation for rapid hmm development in a new language. In: 1994 IEEE International Conference on Acoustics, Speech, and Signal Processing, ICASSP-1994, vol. 1, pp. I-237. IEEE (1994)","DOI":"10.1109\/ICASSP.1994.389311"},{"key":"47_CR42","unstructured":"Xiong, W., Droppo, J., Huang, X., Seide, F., Seltzer, M., Stolcke, A., Yu, D., Zweig, G.: Achieving human parity in conversational speech recognition. arXiv preprint (2016). arXiv:1610.05256"}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-66429-3_47","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,25]],"date-time":"2023-08-25T00:51:45Z","timestamp":1692924705000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-66429-3_47"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319664286","9783319664293"],"references-count":42,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-66429-3_47","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2017]]}}}