{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,9,9]],"date-time":"2024-09-09T20:43:22Z","timestamp":1725914602828},"publisher-location":"Cham","reference-count":31,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319646794"},{"type":"electronic","value":"9783319646800"}],"license":[{"start":{"date-parts":[[2017,1,1]],"date-time":"2017-01-01T00:00:00Z","timestamp":1483228800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2017]]},"DOI":"10.1007\/978-3-319-64680-0_12","type":"book-chapter","created":{"date-parts":[[2017,10,31]],"date-time":"2017-10-31T08:37:26Z","timestamp":1509439046000},"page":"281-297","source":"Crossref","is-referenced-by-count":2,"title":["Sequence-Discriminative Training of Neural Networks"],"prefix":"10.1007","author":[{"given":"Guoguo","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,7,26]]},"reference":[{"key":"12_CR1","unstructured":"Agarwal, A., Akchurin, E., Basoglu, C., Chen, G., Cyphers, S., Droppo, J., Eversole, A., Guenter, B., Hillebrand, M., Hoens, R., Huang, X., Huang, Z., Ivanov, V., Kamenev, A., Kranen, P., Kuchaiev, O., Manousek, W., May, A., Mitra, B., Nano, O., Navarro, G., Orlov, A., Parthasarathi, H., Peng, B., Padmilac, M., Reznichenko, A., Seide, F., Seltzer, M.L., Slaney, M., Stolcke, A., Wang, Y., Wang, H., Yao, K., Yu, D., Zhang, Y., Zweig, G.: An introduction to computational networks and the computational network toolkit. Technical Report MSR-TR-2014-112, Microsoft Research (2014)"},{"key":"12_CR2","first-page":"49","volume":"86","author":"L. Bahl","year":"1986","unstructured":"Bahl, L., Brown, P.F., De\u00a0Souza, P.V., Mercer, R.L.: Maximum mutual information estimation of hidden Markov model parameters for speech recognition. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), vol.\u00a086, pp.\u00a049\u201352 (1986)","journal-title":"Speech and Signal Processing (ICASSP)"},{"key":"12_CR3","doi-asserted-by":"crossref","unstructured":"Bridle, J., Dodd, L.: An Alphanet approach to optimising input transformations for continuous speech recognition. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a0277\u2013280. IEEE (1991)","DOI":"10.1109\/ICASSP.1991.150331"},{"issue":"2","key":"12_CR4","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1007\/s10579-007-9040-x","volume":"41","author":"J. Carletta","year":"2007","unstructured":"Carletta, J.: Unleashing the killer corpus: experiences in creating the multi-everything AMI Meeting Corpus. Lang. Resour. Eval. 41(2), 181\u2013190 (2007)","journal-title":"Lang. Resour. Eval."},{"issue":"7","key":"12_CR5","doi-asserted-by":"crossref","first-page":"1185","DOI":"10.1109\/TASLP.2016.2539499","volume":"24","author":"K. Chen","year":"2016","unstructured":"Chen, K., Huo, Q.: Training deep bidirectional LSTM acoustic model for LVCSR by a context-sensitive-chunk BPTT approach. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(7), 1185\u20131193 (2016)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"1","key":"12_CR6","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"G.E. Dahl","year":"2012","unstructured":"Dahl, G.E., Yu, D., Deng, L., Acero, A.: Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Trans. Audio Speech Lang. Process. 20(1), 30\u201342 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"12_CR7","unstructured":"Fiscus, J.G., Ajot, J., Radde, N., Laprun, C.: Multiple dimension Levenshtein edit distance calculations for evaluating automatic speech recognition systems during simultaneous speech. In: Proceedings of the International Conference on Language Resources and Evaluation (LERC) (2006)"},{"key":"12_CR8","volume-title":"Hypothesis spaces for minimum Bayes risk training in large vocabulary speech recognition","author":"M. Gibson","year":"2006","unstructured":"Gibson, M., Hain, T.: Hypothesis spaces for minimum Bayes risk training in large vocabulary speech recognition. In: Proceedings of INTERSPEECH (2006)"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Gopalakrishnan, P., Kanevsky, D., Nadas, A., Nahamoo, D., Picheny, M.: Decoder selection based on cross-entropies. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a020\u201323. IEEE, New York (1988)","DOI":"10.1109\/ICASSP.1988.196499"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Graves, A., Jaitly, N., Mohamed, A.R.: Hybrid speech recognition with deep bidirectional LSTM. In: Proceedings of Automatic Speech Recognition and Understanding (ASRU), pp.\u00a0273\u2013278. IEEE, New York (2013)","DOI":"10.1109\/ASRU.2013.6707742"},{"key":"12_CR11","doi-asserted-by":"crossref","unstructured":"Heigold, G., McDermott, E., Vanhoucke, V., Senior, A., Bacchiani, M.: Asynchronous stochastic optimization for sequence training of deep neural networks. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a05587\u20135591. IEEE, New York (2014)","DOI":"10.1109\/ICASSP.2014.6854672"},{"key":"12_CR12","volume-title":"A novel loss function for the overall risk criterion based discriminative training of HMM models","author":"J. Kaiser","year":"2000","unstructured":"Kaiser, J., Horvat, B., Kacic, Z.: A novel loss function for the overall risk criterion based discriminative training of HMM models. In: Proceedings of the Sixth International Conference on Spoken Language Processing (2000)"},{"key":"12_CR13","doi-asserted-by":"crossref","unstructured":"Kapadia, S., Valtchev, V., Young, S.: MMI training for continuous phoneme recognition on the TIMIT database. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), vol.\u00a02, pp.\u00a0491\u2013494. IEEE, New York (1993)","DOI":"10.1109\/ICASSP.1993.319349"},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Kingsbury, B.: Lattice-based optimization of sequence classification criteria for neural-network acoustic modeling. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a03761\u20133764. IEEE, New York (2009)","DOI":"10.1109\/ICASSP.2009.4960445"},{"key":"12_CR15","volume-title":"Scalable minimum Bayes risk training of deep neural network acoustic models using distributed Hessian-free optimization","author":"B. Kingsbury","year":"2012","unstructured":"Kingsbury, B., Sainath, T.N., Soltau, H.: Scalable minimum Bayes risk training of deep neural network acoustic models using distributed Hessian-free optimization. In: Proceedings of INTERSPEECH (2012)"},{"key":"12_CR16","unstructured":"Povey, D.: Discriminative training for large vocabulary speech recognition. Ph.D. thesis, University of Cambridge (2005)"},{"key":"12_CR17","doi-asserted-by":"crossref","unstructured":"Povey, D., Kingsbury, B.: Evaluation of proposed modifications to MPE for large scale discriminative training. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), vol.\u00a04, pp.\u00a0IV-321. IEEE, New York (2007)","DOI":"10.1109\/ICASSP.2007.366914"},{"key":"12_CR18","doi-asserted-by":"crossref","unstructured":"Povey, D., Woodland, P.C.: Minimum phone error and I-smoothing for improved discriminative training. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), vol.\u00a01, pp.\u00a0I-105. IEEE, New York (2002)","DOI":"10.1109\/ICASSP.2002.1005687"},{"key":"12_CR19","doi-asserted-by":"crossref","unstructured":"Povey, D., Kanevsky, D., Kingsbury, B., Ramabhadran, B., Saon, G., Visweswariah, K.: Boosted MMI for model and feature-space discriminative training. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a04057\u20134060. IEEE, New York (2008)","DOI":"10.1109\/ICASSP.2008.4518545"},{"key":"12_CR20","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., Hannemann, M., Motlicek, P., Qian, Y., Schwarz, P., et\u00a0al.: The Kaldi speech recognition toolkit. In: Proceedings of Automatic Speech Recognition and Understanding (ASRU), EPFL-CONF-192584. IEEE Signal Processing Society, Piscataway (2011)"},{"key":"12_CR21","unstructured":"Sak, H., Senior, A., Beaufays, F.: Long short-term memory based recurrent neural network architectures for large vocabulary speech recognition (2014). arXiv preprint arXiv:1402.1128"},{"key":"12_CR22","doi-asserted-by":"crossref","unstructured":"Sak, H., Vinyals, O., Heigold, G., Senior, A., McDermott, E., Monga, R., Mao, M.: Sequence discriminative distributed training of long short-term memory recurrent neural networks. In: Proceedings of INTERSPEECH (2014)","DOI":"10.21437\/Interspeech.2014-305"},{"key":"12_CR23","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., Yu, D.: Feature engineering in context-dependent deep neural networks for conversational speech transcription. In: Proceedings of Automatic Speech Recognition and Understanding (ASRU), pp.\u00a024\u201329. IEEE, New York (2011)","DOI":"10.1109\/ASRU.2011.6163899"},{"key":"12_CR24","doi-asserted-by":"crossref","unstructured":"Su, H., Li, G., Yu, D., Seide, F.: Error back propagation for sequence training of context-dependent deep networks for conversational speech transcription. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a06664\u20136668. IEEE, New York (2013)","DOI":"10.1109\/ICASSP.2013.6638951"},{"issue":"4","key":"12_CR25","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1016\/S0167-6393(97)00029-0","volume":"22","author":"V. Valtchev","year":"1997","unstructured":"Valtchev, V., Odell, J., Woodland, P.C., Young, S.J.: MMIE training of large vocabulary recognition systems. Speech Commun. 22(4), 303\u2013314 (1997)","journal-title":"Speech Commun."},{"key":"12_CR26","doi-asserted-by":"crossref","unstructured":"Vesel\u1ef3, K., Ghoshal, A., Burget, L., Povey, D.: Sequence-discriminative training of deep neural networks. In: Proceedings of INTERSPEECH, pp.\u00a02345\u20132349 (2013)","DOI":"10.21437\/Interspeech.2013-548"},{"key":"12_CR27","volume-title":"Sequential classification criteria for NNs in automatic speech recognition","author":"G. Wang","year":"2011","unstructured":"Wang, G., Sim, K.C.: Sequential classification criteria for NNs in automatic speech recognition. In: Proceedings of INTERSPEECH (2011)"},{"issue":"4","key":"12_CR28","doi-asserted-by":"crossref","first-page":"490","DOI":"10.1162\/neco.1990.2.4.490","volume":"2","author":"R.J. Williams","year":"1990","unstructured":"Williams, R.J., Peng, J.: An efficient gradient-based algorithm for on-line training of recurrent network trajectories. Neural Comput. 2(4), 490\u2013501 (1990)","journal-title":"Neural Comput."},{"key":"12_CR29","doi-asserted-by":"crossref","unstructured":"Yu, D., Deng, L.: Automatic Speech Recognition, pp.\u00a0137\u2013153. Springer, London (2015)","DOI":"10.1007\/978-1-4471-5779-3"},{"key":"12_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Chen, G., Yu, D., Yao, K., Khudanpur, S., Glass, J.: Highway long short-term memory RNNs for distant speech recognition. In: Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, New York (2016)","DOI":"10.1109\/ICASSP.2016.7472780"},{"key":"12_CR31","unstructured":"Zilly, J.G., Srivastava, R.K., Koutn\u00edk, J., Schmidhuber, J.: Recurrent highway networks (2016). arXiv preprint arXiv:1607.03474"}],"container-title":["New Era for Robust Speech Recognition"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-64680-0_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T22:20:18Z","timestamp":1659738018000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-64680-0_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017]]},"ISBN":["9783319646794","9783319646800"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-64680-0_12","relation":{},"subject":[],"published":{"date-parts":[[2017]]}}}