{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T20:43:34Z","timestamp":1725914614210},"publisher-location":"Cham","reference-count":38,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_17","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"369-382","source":"Crossref","is-referenced-by-count":0,"title":["Toolkits for Robust Speech Processing"],"prefix":"10.1007","author":[{"given":"Shinji","family":"Watanabe","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takaaki","family":"Hori","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yajie","family":"Miao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marc","family":"Delcroix","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Florian","family":"Metze","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John R.","family":"Hershey","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"unstructured":"Abadi, M., Agarwal, A., Barham, P., Brevdo, E., Chen, Z., Citro, C., Corrado, G.S., Davis, A., Dean, J., Devin, M., et\u00a0al.: TensorFlow: large-scale machine learning on heterogeneous distributed systems (2016). arXiv preprint arXiv:1603.04467. https:\/\/www.tensorflow.org\/","key":"17_CR1"},{"unstructured":"Agarwal, A., Akchurin, E., Basoglu, C., Chen, G., Cyphers, S., Droppo, J., Eversole, A., Guenter, B., Hillebrand, M., Hoens, T.R., et\u00a0al.: An introduction to computational networks and the computational network toolkit. Microsoft Technical Report MSR-TR-2014-112 (2014). https:\/\/github.com\/Microsoft\/CNTK","key":"17_CR2"},{"doi-asserted-by":"crossref","unstructured":"Allauzen, C., Riley, M., Schalkwyk, J., Skut, W., Mohri, M.: OpenFst: a general and efficient weighted finite-state transducer library. In: International Conference on Implementation and Application of Automata, pp.\u00a011\u201323. Springer, New York (2007). http:\/\/www.openfst.org\/","key":"17_CR3","DOI":"10.1007\/978-3-540-76336-9_3"},{"unstructured":"Amodei, D., Anubhai, R., Battenberg, E., Case, C., Casper, J., Catanzaro, B., Chen, J., Chrzanowski, M., Coates, A., Diamos, G., et\u00a0al.: Deep speech 2: end-to-end speech recognition in English and Mandarin (2015). arXiv preprint arXiv:1512.02595. https:\/\/github.com\/baidu-research\/warp-ctc","key":"17_CR4"},{"doi-asserted-by":"crossref","unstructured":"Anguera, X., Wooters, C., Hernando, J.: Acoustic beamforming for speaker diarization of meetings. IEEE Trans. Audio Speech Lang. Process. 15(7), 2011\u20132022 (2007). http:\/\/www.xavieranguera.com\/beamformit\/","key":"17_CR5","DOI":"10.1109\/TASL.2007.902460"},{"unstructured":"Bahdanau, D., Chorowski, J., Serdyuk, D., Brakel, P., Bengio, Y.: End-to-end attention-based large vocabulary speech recognition. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a04945\u20134949 (2016). https:\/\/github.com\/rizar\/attention-lvcsr","key":"17_CR6"},{"doi-asserted-by":"crossref","unstructured":"Bergstra, J., Breuleux, O., Bastien, F., Lamblin, P., Pascanu, R., Desjardins, G., Turian, J., Warde-Farley, D., Bengio, Y.: Theano: A CPU and GPU math compiler in Python. In: Proceedings of the 9th Python in Science Conference, pp.\u00a01\u20137 (2010). http:\/\/deeplearning.net\/software\/theano\/","key":"17_CR7","DOI":"10.25080\/Majora-92bf1922-003"},{"unstructured":"Chen, T., Li, M., Li, Y., Lin, M., Wang, N., Wang, M., Xiao, T., Xu, B., Zhang, C., Zhang, Z.: Mxnet: a flexible and efficient machine learning library for heterogeneous distributed systems. In: Proceedings of Workshop on Machine Learning Systems (LearningSys) in 29th Annual Conference on Neural Information Processing Systems (NIPS) (2015). http:\/\/mxnet-mli.readthedocs.io\/en\/latest\/","key":"17_CR8"},{"doi-asserted-by":"crossref","unstructured":"Chen, X., Liu, X., Qian, Y., Gales, M., Woodland, P.: CUED-RNNLM: an open-source toolkit for efficient training and evaluation of recurrent neural network language models. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a06000\u20136004. IEEE, New York (2016). http:\/\/mi.eng.cam.ac.uk\/projects\/cued-rnnlm\/","key":"17_CR9","DOI":"10.1109\/ICASSP.2016.7472829"},{"unstructured":"Collobert, R., Kavukcuoglu, K., Farabet, C.: Torch7: a MATLAB-like environment for machine learning. In: BigLearn, NIPS Workshop, EPFL-CONF-192376 (2011). http:\/\/torch.ch\/","key":"17_CR10"},{"unstructured":"Degottex, G., Kane, J., Drugman, T., Raitio, T., Scherer, S.: COVAREP: a collaborative voice analysis repository for speech technologies. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a0960\u2013964. IEEE, New York (2014). http:\/\/covarep.github.io\/covarep\/","key":"17_CR11"},{"unstructured":"ELRA: ELDA Portal. http:\/\/www.elra.info\/en\/","key":"17_CR12"},{"unstructured":"Federico, M., Bertoldi, N., Cettolo, M.: IRSTLM: An open source toolkit for handling large scale language models. In: Interspeech, pp.\u00a01618\u20131621 (2008). http:\/\/hlt-mt.fbk.eu\/technologies\/irstlm","key":"17_CR13"},{"key":"17_CR14","first-page":"1764","volume":"14","author":"A. Graves","year":"2014","unstructured":"Graves, A., Jaitly, N.: Towards end-to-end speech recognition with recurrent neural networks. In: ICML, vol.\u00a014, pp.\u00a01764\u20131772 (2014)","journal-title":"In: ICML"},{"unstructured":"Grondin, F., L\u00e9tourneau, D., Ferland, F., Rousseau, V., Michaud, F.: The ManyEars open framework. Auton. Robot. 34(3), 217\u2013232 (2013). https:\/\/sourceforge.net\/projects\/manyears\/","key":"17_CR15"},{"unstructured":"Heafield, K.: KenLM: faster and smaller language model queries. In: Proceedings of the Sixth Workshop on Statistical Machine Translation, pp.\u00a0187\u2013197. Association for Computational Linguistics (2011). http:\/\/kheafield.com\/code\/kenlm\/","key":"17_CR16"},{"doi-asserted-by":"crossref","unstructured":"Hsu, B.J.P., Glass, J.R.: Iterative language model estimation: efficient data structure & algorithms. In: INTERSPEECH, pp.\u00a0841\u2013844 (2008). https:\/\/github.com\/mitlm\/mitlm","key":"17_CR17","DOI":"10.21437\/Interspeech.2008-255"},{"unstructured":"Idiap Research Institute: Bob 2.4.0 documentation. https:\/\/pythonhosted.org\/bob\/","key":"17_CR18"},{"unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., Darrell, T.: Caffe: convolutional architecture for fast feature embedding. In: Proceedings of the 22nd ACM International Conference on Multimedia, pp.\u00a0675\u2013678. ACM (2014). http:\/\/caffe.berkeleyvision.org\/","key":"17_CR19"},{"unstructured":"Kumatani, K., McDonough, J., Schacht, S., Klakow, D., Garner, P.N., Li, W.: Filter bank design based on minimization of individual aliasing terms for minimum mutual information subband adaptive beamforming. In: IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01609\u20131612. IEEE (2008). http:\/\/distantspeechrecognition.sourceforge.net\/","key":"17_CR20"},{"doi-asserted-by":"crossref","unstructured":"Lee, K.F., Hon, H.W., Reddy, R.: An overview of the SPHINX speech recognition system. IEEE Trans. Acoust. Speech Signal Process. 38(1), 35\u201345 (1990). http:\/\/cmusphinx.sourceforge.net\/","key":"17_CR21","DOI":"10.1109\/29.45616"},{"unstructured":"Lee, A., Kawahara, T., Shikano, K.: Julius\u00a0\u2013 an open source real-time large vocabulary recognition engine. In: Interspeech, pp.\u00a01691\u20131694 (2001). http:\/\/julius.osdn.jp\/en_index.php","key":"17_CR22"},{"unstructured":"Linguistic Data Consortium: https:\/\/www.ldc.upenn.edu\/","key":"17_CR23"},{"doi-asserted-by":"crossref","unstructured":"Maas, A.L., Xie, Z., Jurafsky, D., Ng, A.Y.: Lexicon-free conversational speech recognition with neural networks. In: Proceedings of the North American Chapter of the Association for Computational Linguistics (NAACL) (2015). https:\/\/github.com\/amaas\/stanford-ctc","key":"17_CR24","DOI":"10.3115\/v1\/N15-1038"},{"unstructured":"Metze, F., Fosler-Lussier, E.: The speech recognition virtual kitchen: An initial prototype. In: Interspeech, pp.\u00a01872\u20131873 (2012). http:\/\/speechkitchen.org\/","key":"17_CR25"},{"doi-asserted-by":"crossref","unstructured":"Miao, Y., Gowayyed, M., Metze, F.: EESEN: end-to-end speech recognition using deep RNN models and WFST-based decoding. In: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU), pp.\u00a0167\u2013174 (2015). https:\/\/github.com\/srvk\/eesen","key":"17_CR26","DOI":"10.1109\/ASRU.2015.7404790"},{"unstructured":"Mikolov, T., Karafi\u00e1t, M., Burget, L., Cernock\u1ef3, J., Khudanpur, S.: Recurrent neural network based language model. In: Interspeech, pp.\u00a01045\u20131048 (2010). http:\/\/www.rnnlm.org\/","key":"17_CR27"},{"doi-asserted-by":"crossref","unstructured":"Nakadai, K., Takahashi, T., Okuno, H.G., Nakajima, H., Hasegawa, Y., Tsujino, H.: Design and implementation of robot audition system \u201chark\u201d open source software for listening to three simultaneous speakers. Adv. Robot. 24(5\u20136), 739\u2013761 (2010). http:\/\/www.hark.jp\/","key":"17_CR28","DOI":"10.1163\/016918610X493561"},{"doi-asserted-by":"crossref","unstructured":"Ozerov, A., Vincent, E., Bimbot, F.: A general flexible framework for the handling of prior information in audio source separation. IEEE Trans. Audio Speech Lang. Process. 20(4), 1118\u20131133 (2012). http:\/\/bass-db.gforge.inria.fr\/fasst\/","key":"17_CR29","DOI":"10.1109\/TASL.2011.2172425"},{"unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., Silovsky, J., Stemmer, G., Vesely, K.: The Kaldi speech recognition toolkit. In: IEEE 2011 Workshop on Automatic Speech Recognition and Understanding (2011). http:\/\/kaldi-asr.org\/","key":"17_CR30"},{"unstructured":"Rybach, D., Gollan, C., Heigold, G., Hoffmeister, B., L\u00f6\u00f6f, J., Schl\u00fcter, R., Ney, H.: The RWTH Aachen University open source speech recognition system. In: Interspeech, pp.\u00a02111\u20132114 (2009). https:\/\/www-i6.informatik.rwth-aachen.de\/rwth-asr\/","key":"17_CR31"},{"unstructured":"Schwenk, H.: CSLM\u00a0\u2013 a modular open-source continuous space language modeling toolkit. In: INTERSPEECH, pp.\u00a01198\u20131202 (2013). http:\/\/www-lium.univ-lemans.fr\/cslm\/","key":"17_CR32"},{"unstructured":"Stolcke, A., et al.: SRILM \u2013 an extensible language modeling toolkit. In: Interspeech, vol. 2002, pp. 901\u2013904 (2002). http:\/\/www.speech.sri.com\/projects\/srilm\/","key":"17_CR33"},{"unstructured":"Sundermeyer, M., Schl\u00fcter, R., Ney, H.: RWTHLM \u2013 the RWTH Aachen University neural network language modeling toolkit. In: INTERSPEECH, pp.\u00a02093\u20132097 (2014). https:\/\/www-i6.informatik.rwth-aachen.de\/web\/Software\/rwthlm.php","key":"17_CR34"},{"unstructured":"Tokui, S., Oono, K., Hido, S., Clayton, J.: Chainer: a next-generation open source framework for deep learning. In: Proceedings of Workshop on Machine Learning Systems (LearningSys) in 29th Annual Conference on Neural Information Processing Systems (NIPS) (2015). http:\/\/chainer.org\/","key":"17_CR35"},{"unstructured":"Weninger, F., Bergmann, J., Schuller, B.: Introducing CURRENNT \u2013 the Munich open-source CUDA RecurREnt neural network toolkit. J. Mach. Learn. Res. 16(3), 547\u2013551 (2015). https:\/\/sourceforge.net\/projects\/currennt\/","key":"17_CR36"},{"doi-asserted-by":"crossref","unstructured":"Yoshioka, T., Nakatani, T., Miyoshi, M., Okuno, H.G.: Blind separation and dereverberation of speech mixtures by joint optimization. IEEE Trans. Audio Speech Lang. Process. 19(1), 69\u201384 (2011). http:\/\/www.kecl.ntt.co.jp\/icl\/signal\/wpe\/","key":"17_CR37","DOI":"10.1109\/TASL.2010.2045183"},{"unstructured":"Young, S., Evermann, G., Gales, M., Hain, T., Kershaw, D., Liu, X., Moore, G., Odell, J., Ollason, D., Povey, D., et\u00a0al.: The HTK Book, vol.\u00a03, p.\u00a0175. Cambridge University Engineering Department (2002). http:\/\/htk.eng.cam.ac.uk\/","key":"17_CR38"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,27]],"date-time":"2023-08-27T19:50:12Z","timestamp":1693165812000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_17","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}