{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T20:43:23Z","timestamp":1725914603981},"publisher-location":"Cham","reference-count":73,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_18","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"385-399","source":"Crossref","is-referenced-by-count":2,"title":["Speech Research at Google to Enable Universal Speech Interfaces"],"prefix":"10.1007","author":[{"given":"Michiel","family":"Bacchiani","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fran\u00e7oise","family":"Beaufays","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexander","family":"Gruenstein","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pedro","family":"Moreno","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Johan","family":"Schalkwyk","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Trevor","family":"Strohman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Heiga","family":"Zen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"18_CR1","volume-title":"Vocaine the vocoder and applications in speech synthesis","author":"Y. Agiomyrgiannakis","year":"2015","unstructured":"Agiomyrgiannakis, Y.: Vocaine the vocoder and applications in speech synthesis. In: Proceedings of ICASSP (2015)"},{"key":"18_CR2","volume-title":"Discriminative features for language identification","author":"C. Alberti","year":"2011","unstructured":"Alberti, C., Bacchiani, M.: Discriminative features for language identification. In: Proceeding of Interspeech (2011)"},{"key":"18_CR3","volume-title":"An audio indexing system for election video material","author":"C. Alberti","year":"2009","unstructured":"Alberti, C., Bacchiani, M., Bezman, A., Chelba, C., Drofa, A., Liao, H., Moreno, P., Power, T., Sahuguet, A., Shugrina, M., Siohan, O.: An audio indexing system for election video material. In: Proceedings of ICASSP (2009)"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Aleksic, P., Allauzen, C., Elson, D., Kracun, A., Casado, D.M., Moreno, P.J.: Improved recognition of contact names in voice commands. In: Proceedings of ICASSP, pp.\u00a04441\u20134444 (2015)","DOI":"10.1109\/ICASSP.2015.7178957"},{"key":"18_CR5","volume-title":"Bringing contextual information to Google speech recognition","author":"P. Aleksic","year":"2015","unstructured":"Aleksic, P., Ghodsi, M., Michaely, A., Allauzen, C., Hall, K., Roark, B., Rybach, D., Moreno, P.: Bringing contextual information to Google speech recognition. In: Proceedings of Interspeech (2015)"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Allauzen, C., Riley, M.: Bayesian language model interpolation for mobile speech input. In: Proceedings of Interspeech, pp.\u00a01429\u20131432 (2011)","DOI":"10.21437\/Interspeech.2011-249"},{"key":"18_CR7","doi-asserted-by":"crossref","unstructured":"Allauzen, C., Riley, M., Schalkwyk, J., Skut, W., Mohri, M.: OpenFst: a general and efficient weighted finite-state transducer library. In: Proceedings of the 12th International Conference on Implementation and Application of Automata (CIAA) (2007)","DOI":"10.1007\/978-3-540-76336-9_3"},{"key":"18_CR8","volume-title":"On the efficient representation and execution of deep acoustic models","author":"R. Alvarez","year":"2016","unstructured":"Alvarez, R., Prabhavalkar, R., Bakhtin, A.: On the efficient representation and execution of deep acoustic models. In: Proceedings of Interspeech (2016)"},{"key":"18_CR9","volume-title":"Context dependent state tying for speech recognition using deep neural network acoustic models","author":"M. Bacchiani","year":"2014","unstructured":"Bacchiani, M., Rybach, D.: Context dependent state tying for speech recognition using deep neural network acoustic models. In: Proceedings of ICASSP (2014)"},{"key":"18_CR10","volume-title":"Deploying GOOG-411: early lessons in data, measurement, and testing","author":"M. Bacchiani","year":"2008","unstructured":"Bacchiani, M., Beaufays, F., Schalkwyk, J., Schuster, M., Strope, B.: Deploying GOOG-411: early lessons in data, measurement, and testing. In: Proceedings of ICASSP (2008)"},{"key":"18_CR11","volume-title":"The neural networks behind Google voice transcription","author":"F. Beaufays","year":"2015","unstructured":"Beaufays, F.: The neural networks behind Google voice transcription. In: Google Research blog (2015). https:\/\/research.googleblog.com\/2015\/08\/the-neural-networks-behind-google-voice.html"},{"key":"18_CR12","volume-title":"How the dream of speech recognition became a reality","author":"F. Beaufays","year":"2016","unstructured":"Beaufays, F.: How the dream of speech recognition became a reality. In: Google (2016). www.google.com\/about\/careers\/stories\/how-one-team-turned-the-dream-of-speech-recognition-into-a-reality"},{"key":"18_CR13","volume-title":"Language modeling capitalization","author":"F. Beaufays","year":"2013","unstructured":"Beaufays, F., Strope, B.: Language modeling capitalization. In: Proceedings of ICASSP (2013)"},{"key":"18_CR14","volume-title":"Connectionist Speech Recognition: A Hybrid Approach","author":"H. Bourlard","year":"1993","unstructured":"Bourlard, H., Morgan, N.: Connectionist Speech Recognition: A Hybrid Approach. Kluwer Academic, Dordrecht (1993)"},{"key":"18_CR15","volume-title":"Small-footprint keyword spotting using deep neural networks","author":"G. Chen","year":"2014","unstructured":"Chen, G., Parada, C., Heigold, G.: Small-footprint keyword spotting using deep neural networks. In: Proceedings of ICASSP (2014)"},{"key":"18_CR16","volume-title":"Query-by-example keyword spotting using long short-term memory networks","author":"G. Chen","year":"2015","unstructured":"Chen, G., Parada, C., Sainath, T.N.: Query-by-example keyword spotting using long short-term memory networks. In: Proceedings of ICASSP (2015)"},{"key":"18_CR17","volume-title":"Locally-connected and convolutional neural networks for small footprint speaker recognition","author":"Y.H. Chen","year":"2015","unstructured":"Chen, Y.H., Lopez-Moreno, I., Sainath, T., Visontai, M., Alvarez, R., Parada, C.: Locally-connected and convolutional neural networks for small footprint speaker recognition. In: Proceedings of Interspeech (2015)"},{"key":"18_CR18","unstructured":"Dean, J., Ghemawat, S.: MapReduce: simplified data processing on large Clusters. In: OSDI\u201904, Sixth Symposium on Operating System Design and Implementation (2004)"},{"key":"18_CR19","unstructured":"Dean, J., Corrado, G.S., Monga, R., Chen, K., Devin, M., Le, Q.V., Mao, M.Z., Ranzato, M., Senior, A., Tucker, P., Yang, K., Ng, A.Y.: Large scale distributed deep networks. In: Proceedings of Neural Information Processing Systems (NIPS) (2012)"},{"issue":"3","key":"18_CR20","doi-asserted-by":"crossref","first-page":"333","DOI":"10.1017\/S1351324914000175","volume":"21","author":"P. Ebden","year":"2014","unstructured":"Ebden, P., Sproat, R.: The Kestrel TTS text normalization system. J. Nat. Lang. Eng. 21(3), 333\u2013353 (2014)","journal-title":"J. Nat. Lang. Eng."},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Gonzalez-Dominguez, J., Lopez-Moreno, I., Moreno, P.J., Gonzalez-Rodriguez, J.: Frame by frame language identification in short utterances using deep neural networks. In: Neural Networks, Special Issue: Neural Network Learning in Big Data, pp.\u00a049\u201358 (2014)","DOI":"10.1016\/j.neunet.2014.08.006"},{"key":"18_CR22","volume-title":"Recent advances in Google real-time HMM-driven unit selection synthesizer","author":"X. Gonzalvo","year":"2016","unstructured":"Gonzalvo, X., Tazari, S., Chan, C.A., Becker, M., Gutkin, A., Silen, H.: Recent advances in Google real-time HMM-driven unit selection synthesizer. In: Proceeding of Interspeech (2016)"},{"key":"18_CR23","volume-title":"Restoring punctuation and capitalization in transcribed speech","author":"A. Gravano","year":"2009","unstructured":"Gravano, A., Jansche, M., Bacchiani, M.: Restoring punctuation and capitalization in transcribed speech. In: Proceedings of ICASSP (2009)"},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Graves, A.: Supervised Sequence Labelling with Recurrent Neural Networks. Studies in Computational Intelligence, vol.\u00a0385. Springer, New York (2012)","DOI":"10.1007\/978-3-642-24797-2"},{"key":"18_CR25","volume-title":"Asynchronous stochastic optimization for sequence training of deep neural networks","author":"G. Heigold","year":"2014","unstructured":"Heigold, G., McDermott, E., Vanhoucke, V., Senior, A., Bacchiani, M.: Asynchronous stochastic optimization for sequence training of deep neural networks. In: Proceedings of ICASSP (2014)"},{"key":"18_CR26","volume-title":"End-to-end text-dependent speaker verification","author":"G. Heigold","year":"2016","unstructured":"Heigold, G., Moreno, I., Bengio, S., Shazeer, N.M.: End-to-end text-dependent speaker verification. In: Proceedings of ICASSP (2016)"},{"key":"18_CR27","unstructured":"Hershey, J.R., Roux, J.L., Weninger, F.: Deep unfolding: model-based inspiration of novel deep architectures. CoRR abs\/1409.2574 (2014)"},{"issue":"6","key":"18_CR28","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G. Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G., Rahman Mohamed, A., Jaitly, N., Senior, A., Vanhoucke, V., Nguyen, P., Sainath, T., Kingsbury, B.: Deep neural networks for acoustic modeling in speech recognition. Signal Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"Signal Process. Mag."},{"key":"18_CR29","volume-title":"Speech acoustic modeling from raw multichannel waveforms","author":"Y. Hoshen","year":"2015","unstructured":"Hoshen, Y., Weiss, R.J., Wilson, K.W.: Speech acoustic modeling from raw multichannel waveforms. In: Proceedings of ICASSP (2015)"},{"key":"18_CR30","volume-title":"Building transcribed speech corpora quickly and cheaply for many languages","author":"T. Hughes","year":"2010","unstructured":"Hughes, T., Nakajima, K., Ha, L., Vasu, A., Moreno, P., LeBeau, M.: Building transcribed speech corpora quickly and cheaply for many languages. In: Proceedings of Interspeech (2010)"},{"key":"18_CR31","doi-asserted-by":"crossref","unstructured":"Kawahara, H., Agiomyrgiannakis, Y., Zen, H.: Using instantaneous frequency and aperiodicity detection to estimate f0 for high-quality speech synthesis. In: ISCA SSW9 (2016)","DOI":"10.21437\/SSW.2016-36"},{"key":"18_CR32","volume-title":"Accurate and compact large vocabulary speech recognition on mobile devices","author":"X. Lei","year":"2013","unstructured":"Lei, X., Senior, A., Gruenstein, A., Sorensen, J.: Accurate and compact large vocabulary speech recognition on mobile devices. In: Proceedings of Interspeech (2013)"},{"key":"18_CR33","volume-title":"Neural network adaptive beamforming for robust multichannel speech recognition","author":"B. Li","year":"2016","unstructured":"Li, B., Sainath, T.N., Weiss, R.J., Wilson, K.W., Bacchiani, M.: Neural network adaptive beamforming for robust multichannel speech recognition. In: Interspeech (2016)"},{"key":"18_CR34","doi-asserted-by":"crossref","unstructured":"Liao, H., McDermott, E., Senior, A.: Large scale deep neural network acoustic modeling with semi-supervised training data for YouTube video transcription. In: Proceedings of IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU) (2013)","DOI":"10.1109\/ASRU.2013.6707758"},{"key":"18_CR35","volume-title":"Personalized speech recognition on mobile devices","author":"I. McGraw","year":"2016","unstructured":"McGraw, I., Prabhavalkar, R., Alvarez, R., Arenas, M.G., Rao, K., Rybach, D., Alsharif, O., Sak, H., Gruenstein, A., Beaufays, F., Parada, C.: Personalized speech recognition on mobile devices. In: Proceedings of ICASSP (2016)"},{"key":"18_CR36","doi-asserted-by":"crossref","unstructured":"Nakkiran, P., Alvarez, R., Prabhavalkar, R., Parada, C.: Compressing deep neural networks using a rank-constrained topology. In: Proceedings of Interspeech, pp.\u00a01473\u20131477 (2015)","DOI":"10.21437\/Interspeech.2015-351"},{"key":"18_CR37","doi-asserted-by":"crossref","unstructured":"Prabhavalkar, R., Alvarez, R., Parada, C., Nakkiran, P., Sainath, T.: Automatic gain control and multi-style training for robust small-footprint keyword spotting with deep neural networks. In: Proceedings of ICASSP, pp.\u00a04704\u20134708 (2015)","DOI":"10.1109\/ICASSP.2015.7178863"},{"key":"18_CR38","volume-title":"On the compression of recurrent neural networks with an application to LVCSR acoustic modeling for embedded speech recognition","author":"R. Prabhavalkar","year":"2016","unstructured":"Prabhavalkar, R., Alsharif, O., Bruguier, A., McGraw, I.: On the compression of recurrent neural networks with an application to LVCSR acoustic modeling for embedded speech recognition. In: Proceedings of ICASSP (2016)"},{"key":"18_CR39","volume-title":"Grapheme-to-phoneme conversion using long short-term memory recurrent neural networks","author":"K. Rao","year":"2016","unstructured":"Rao, K., Peng, F., Beaufays, F.: Grapheme-to-phoneme conversion using long short-term memory recurrent neural networks. In: Proceedings of ICASSP (2016)"},{"key":"18_CR40","first-page":"693","volume-title":"Advances in Neural Information Processing Systems","author":"B. Recht","year":"2011","unstructured":"Recht, B., Re, C., Wright, S., Feng, N.: Hogwild: a lock-free approach to parallelizing stochastic gradient descent. In: Shawe-Taylor, J., Zemel, R.S., Bartlett, P.L., Pereira, F., Weinberger, K.Q. (eds.) Advances in Neural Information Processing Systems, vol.\u00a024, pp.\u00a0693\u2013701. Curran Associates, Red Hook (2011)"},{"key":"18_CR41","volume-title":"The Use of Recurrent Neural Networks in Continuous Speech Recognition","author":"T. Robinson","year":"1995","unstructured":"Robinson, T., Hochberg, M., Renals, S.: The Use of Recurrent Neural Networks in Continuous Speech Recognition. Springer, New York (1995)"},{"key":"18_CR42","volume-title":"Pronunciation learning for named-entities through crowd-sourcing","author":"A. Rutherford","year":"2014","unstructured":"Rutherford, A., Peng, F., Beaufays, F.: Pronunciation learning for named-entities through crowd-sourcing. In: Proceedings of Interspeech (2014)"},{"key":"18_CR43","volume-title":"Convolutional neural networks for small-footprint keyword spotting","author":"T. Sainath","year":"2015","unstructured":"Sainath, T., Parada, C.: Convolutional neural networks for small-footprint keyword spotting. In: Proceedings of Interspeech (2015)"},{"key":"18_CR44","volume-title":"Convolutional, long short-term memory, fully connected deep neural networks","author":"T. Sainath","year":"2015","unstructured":"Sainath, T., Vinyals, O., Senior, A., Sak, H.: Convolutional, long short-term memory, fully connected deep neural networks. In: Proceedings of ICASSP (2015)"},{"key":"18_CR45","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Weiss, R.J., Wilson, K.W., Narayanan, A., Bacchiani, M., Senior, A.: Speaker localization and microphone spacing invariant acoustic modeling from raw multichannel waveforms. In: Proceedings of IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU) (2015)","DOI":"10.1109\/ASRU.2015.7404770"},{"key":"18_CR46","volume-title":"Learning the speech front-end with raw waveform CLDNNS","author":"T.N. Sainath","year":"2015","unstructured":"Sainath, T.N., Weiss, R.J., Wilson, K.W., Senior, A., Vinyals, O.: Learning the speech front-end with raw waveform CLDNNS. In: Proceedings of Interspeech (2015)"},{"key":"18_CR47","volume-title":"Improvements to factorized neural network multichannel models","author":"T.N. Sainath","year":"2016","unstructured":"Sainath, T.N., Narayanan, A., Weiss, R.J., Wilson, K.W., Bacchiani, M., Shafran, I.: Improvements to factorized neural network multichannel models. In: Interspeech (2016)"},{"key":"18_CR48","volume-title":"Factored spatial and spectral multichannel raw waveform CLDNNS","author":"T.N. Sainath","year":"2016","unstructured":"Sainath, T.N., Weiss, R.J., Wilson, K.W., Narayanan, A., Bacchiani, M.: Factored spatial and spectral multichannel raw waveform CLDNNS. In: Proceedings of ICASSP (2016)"},{"key":"18_CR49","volume-title":"Written-domain language modeling for automatic speech recognition","author":"H. Sak","year":"2013","unstructured":"Sak, H., Sung, Y., Beaufays, F., Allauzen, C.: Written-domain language modeling for automatic speech recognition. In: Proceedings of Interspeech (2013)"},{"key":"18_CR50","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A.W., Beaufays, F.: Long short-term memory recurrent neural network architectures for large scale acoustic modeling. In: Proceedings of Interspeech, pp.\u00a0338\u2013342 (2014)","DOI":"10.21437\/Interspeech.2014-80"},{"key":"18_CR51","volume-title":"Sequence discriminative distributed training of long short-term memory recurrent neural networks","author":"H. Sak","year":"2014","unstructured":"Sak, H., Vinyals, O., Heigold, G., Senior, A., McDermott, E., Monga, R., Mao, M.: Sequence discriminative distributed training of long short-term memory recurrent neural networks. In: Proceedings of Interspeech (2014)"},{"key":"18_CR52","volume-title":"Google voice search: faster and more accurate","author":"H. Sak","year":"2015","unstructured":"Sak, H., Senior, A., Rao, K., Beaufays, F., Schalkwyk, J.: Google voice search: faster and more accurate. In: Google Research blog (2015). https:\/\/research.googleblog.com\/2015\/09\/google-voice-search-faster-and-more.html"},{"key":"18_CR53","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A.W., Rao, K., Beaufays, F.: Fast and accurate recurrent neural network acoustic models for speech recognition. CoRR abs\/1507.06947 (2015)","DOI":"10.21437\/Interspeech.2015-350"},{"key":"18_CR54","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A.W., Rao, K., Irsoy, O., Graves, A., Beaufays, F., Schalkwyk, J.: Learning acoustic frame labeling for speech recognition with recurrent neural networks. In: Proceedings of ICASSP, pp.\u00a04280\u20134284 (2015)","DOI":"10.1109\/ICASSP.2015.7178778"},{"key":"18_CR55","volume-title":"Google Search by Voice: A Case Study","author":"J. Schalkwyk","year":"2010","unstructured":"Schalkwyk, J., Beeferman, D., Beaufays, F., Byrne, B., Chelba, C., Cohen, M., Garrett, M., Strope, B.: Google Search by Voice: A Case Study. Springer, New York (2010)"},{"issue":"5","key":"18_CR56","doi-asserted-by":"crossref","first-page":"489","DOI":"10.1109\/TSA.2004.832988","volume":"12","author":"M. Seltzer","year":"2004","unstructured":"Seltzer, M., Raj, B., Stern, R.M.: Likelihood-maximizing beamforming for robust handsfree speech recognition. IEEE Trans. Audio Speech Lang. Process. 12(5), 489\u2013498 (2004)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"18_CR57","volume-title":"GMM-free DNN training","author":"A. Senior","year":"2014","unstructured":"Senior, A., Heigold, G., Bacchiani, M., Liao, H.: GMM-free DNN training. In: Proceedings of ICASSP (2014)"},{"key":"18_CR58","doi-asserted-by":"crossref","unstructured":"Senior, A.W., Sak, H., Shafran, I.: Context dependent phone models for LSTM RNN acoustic modelling. In: Proceedings of ICASSP, pp.\u00a04585\u20134589 (2015)","DOI":"10.1109\/ICASSP.2015.7178839"},{"key":"18_CR59","doi-asserted-by":"crossref","unstructured":"Shan, J., Wu, G., Hu, Z., Tang, X., Jansche, M., Moreno, P.J.: Search by voice in Mandarin Chinese. In: Proceedings of Interspeech, pp.\u00a0354\u2013357 (2010)","DOI":"10.21437\/Interspeech.2010-129"},{"key":"18_CR60","unstructured":"Shugrina, M.: Formatting time-aligned ASR transcripts for readability. In: 2010 Annual Conference of the North American Chapter of the Association for Computational Linguistics (2010)"},{"key":"18_CR61","volume-title":"Unary data structures for language models","author":"J. Sorensen","year":"2011","unstructured":"Sorensen, J., Allauzen, C.: Unary data structures for language models. In: Proceedings of Interspeech (2011)"},{"key":"18_CR62","unstructured":"Sparrowhawk. https:\/\/github.com\/google\/sparrowhawk (2016)"},{"key":"18_CR63","doi-asserted-by":"crossref","unstructured":"Tokuda, K., Zen, H.: Directly modeling speech waveforms by neural networks for statistical parametric speech synthesis. In: Proceedings of ICASSP, pp.\u00a04215\u20134219 (2015)","DOI":"10.1109\/ICASSP.2015.7178765"},{"key":"18_CR64","volume-title":"Directly modeling voiced and unvoiced components in speech waveforms by neural networks","author":"K. Tokuda","year":"2015","unstructured":"Tokuda, K., Zen, H.: Directly modeling voiced and unvoiced components in speech waveforms by neural networks. In: Proceedings of ICASSP (2015)"},{"key":"18_CR65","volume-title":"Speech recognition and deep learning","author":"V. Vanhoucke","year":"2012","unstructured":"Vanhoucke, V.: Speech recognition and deep learning. In: Google Research blog (2012). https:\/\/research.googleblog.com\/2012\/08\/speech-recognition-and-deep-learning.html"},{"key":"18_CR66","volume-title":"Deep neural networks for small footprint text-dependent speaker verification","author":"E. Variani","year":"2014","unstructured":"Variani, E., Lei, X., McDermott, E., Moreno, I.L., Gonzalez-Dominguez, J.: Deep neural networks for small footprint text-dependent speaker verification. In: Proceedings of ICASSP (2014)"},{"key":"18_CR67","volume-title":"Complex Linear Prediction (CLP): a discriminative approach to joint feature extraction and acoustic modeling","author":"E. Variani","year":"2016","unstructured":"Variani, E., Sainath, T.N., Shafran, I., Bacchiani, M.: Complex Linear Prediction (CLP): a discriminative approach to joint feature extraction and acoustic modeling. In: Proceedings of Interspeech (2016)"},{"key":"18_CR68","first-page":"328","volume":"37","author":"A. Waibel","year":"1989","unstructured":"Waibel, A., Hanazawa, T., Hinton, G., Shikano, K., Lang, K.: Phoneme recognition using time-delay neural networks. In: Proceedings of ICASSP, vol.\u00a037, pp.\u00a0328\u2013339 (1989)","journal-title":"In: Proceedings of ICASSP"},{"key":"18_CR69","volume-title":"Acoustic modeling for speech synthesis \u2013 from HMM to RNN","author":"H. Zen","year":"2015","unstructured":"Zen, H.: Acoustic modeling for speech synthesis \u2013 from HMM to RNN. Invited Talk. In: ASRU (2015)"},{"key":"18_CR70","doi-asserted-by":"crossref","unstructured":"Zen, H., Sak, H.: Unidirectional long short-term memory recurrent neural network with recurrent output layer for low-latency speech synthesis. In: Proceedings of ICASSP, pp.\u00a04470\u20134474 (2015)","DOI":"10.1109\/ICASSP.2015.7178816"},{"key":"18_CR71","doi-asserted-by":"crossref","unstructured":"Zen, H., Senior, A.: Deep mixture density networks for acoustic modeling in statistical parametric speech synthesis. In: Proceedings of ICASSP, pp.\u00a03872\u20133876 (2014)","DOI":"10.1109\/ICASSP.2014.6854321"},{"key":"18_CR72","doi-asserted-by":"crossref","unstructured":"Zen, H., Senior, A., Schuster, M.: Statistical parametric speech synthesis using deep neural networks. In: Proceedings of ICASSP, pp.\u00a07962\u20137966 (2013)","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"18_CR73","volume-title":"Fast, compact, and high quality LSTM-RNN based statistical parametric speech synthesizers for mobile devices","author":"H. Zen","year":"2016","unstructured":"Zen, H., Agiomyrgiannakis, Y., Egberts, N., Henderson, F., Szczepaniak, P.: Fast, compact, and high quality LSTM-RNN based statistical parametric speech synthesizers for mobile devices. In: Interspeech (2016)"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T22:21:16Z","timestamp":1659738076000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":73,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_18","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}