{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,14]],"date-time":"2025-05-14T20:25:28Z","timestamp":1747254328492},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2017,9,30]],"date-time":"2017-09-30T00:00:00Z","timestamp":1506729600000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2018,7]]},"DOI":"10.1007\/s11265-017-1292-0","type":"journal-article","created":{"date-parts":[[2017,9,30]],"date-time":"2017-09-30T14:12:34Z","timestamp":1506780754000},"page":"1013-1023","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Lattice Based Transcription Loss for End-to-End Speech Recognition"],"prefix":"10.1007","volume":"90","author":[{"given":"Jian","family":"Kang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei-Qiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei-Wei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jia","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael T.","family":"Johnson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,9,30]]},"reference":[{"key":"1292_CR1","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., & Yu, D (2011). Conversational speech transcription using context-dependent deep neural networks. In Twelfth annual conference of the international speech communication.","DOI":"10.21437\/Interspeech.2011-169"},{"key":"1292_CR2","doi-asserted-by":"crossref","unstructured":"Seide, F., Li, G., Chen, X., & Yu, D. (2011). Feature engineering in context-dependent deep neural networks for conversational speech transcription. In IEEE workshop on automatic speech recognition and understanding (ASRU) (pp. 24\u201329).","DOI":"10.1109\/ASRU.2011.6163899"},{"issue":"1","key":"1292_CR3","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","volume":"20","author":"GE Dahl","year":"2012","unstructured":"Dahl, G. E., Yu, D., Deng, L., & Acero, A (2012). Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition. IEEE Transactions on Audio, Speech, and Language Processing, 20(1), 30\u201342.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"issue":"8","key":"1292_CR4","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., & Schmidhuber, J (1997). Long short-term memory. Neural Computation, 9(8), 1735\u20131780.","journal-title":"Neural Computation"},{"key":"1292_CR5","first-page":"3","volume":"2","author":"T Mikolov","year":"2010","unstructured":"Mikolov, T., Karafit, M., Burget, L., Cernocky, J., & Khudanpur, S (2010). Recurrent neural network based language model. Interspeech, 2, 3.","journal-title":"Interspeech"},{"key":"1292_CR6","doi-asserted-by":"crossref","unstructured":"Graves, A., Jaitly, N., & Mohamed, A. R. (2013). Hybrid speech recognition with deep bidirectional LSTM. In IEEE workshop on automatic speech recognition and understanding (ASRU).","DOI":"10.1109\/ASRU.2013.6707742"},{"key":"1292_CR7","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A. W., & Beaufays, F (2014). Long short-term memory recurrent neural network architectures for large scale acoustic modeling. In Fiftteenth annual conference of the international speech communication associaton.","DOI":"10.21437\/Interspeech.2014-80"},{"key":"1292_CR8","doi-asserted-by":"crossref","unstructured":"Sainath, T. N., Vinyals, O., Senior, A., & Sak, H. (2015). Convolutional, long short-term memory, fully connected deep neural networks. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 4580\u20134584).","DOI":"10.1109\/ICASSP.2015.7178838"},{"key":"1292_CR9","doi-asserted-by":"crossref","unstructured":"Geiger, J. T., Zhang, Z., Weninger, F., Schuller, B., & Rigoll, G (2014). Robust speech recognition using long short-term memory recurrent neural networks for hybrid acoustic modelling. In Fifteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2014-151"},{"key":"1292_CR10","doi-asserted-by":"crossref","unstructured":"Senior, A., Sak, H., & Shafran, I. (2015). Context dependent phone models for LSTM RNN acoustic modelling. In IEEE international conference on acoustics, speech and signal processing (ICASSP).","DOI":"10.1109\/ICASSP.2015.7178839"},{"key":"1292_CR11","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A. R., & Hinton, G. (2013). Speech recognition with deep recurrent neural networks. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 6645\u20136649).","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"1292_CR12","unstructured":"Graves, A., & Jaitly, N (2014). Towards end-to-end speech recognition with recurrent neural networks. In Proceedings of the 31st international conference on machine learning (ICML-14) (pp. 1764\u20131772)."},{"key":"1292_CR13","unstructured":"Chorowski, J., Bahdanau, D., Cho, K., & Bengio, Y (2014). End-to-end continuous speech recognition using attention-based recurrent NN: first results. arXiv: 1412.1602 ."},{"key":"1292_CR14","unstructured":"Chorowski, J. K., Bahdanau, D., Serdyuk, D., Cho, K., & Bengio, Y (2015). Attention-based models for speech recognition. In Advances in neural information processing systems (pp. 577\u2013585)."},{"key":"1292_CR15","doi-asserted-by":"crossref","unstructured":"Bahdanau, D., Chorowski, J., Serdyuk, D., & Bengio, Y. (2016). End-to-end attention-based large vocabulary speech recognition. In IEEE International conference on acoustics, speech and signal processing (ICASSP) (pp. 4945\u20134949).","DOI":"10.1109\/ICASSP.2016.7472618"},{"key":"1292_CR16","unstructured":"Hannun, A. Y., Maas, A. L., Jurafsky, D., & Ng, A.Y. (2014). First-pass large vocabulary continuous speech recognition using bi-directional recurrent DNNs. arXiv: http:\/\/arXiv.org\/abs\/1408.2873 ."},{"key":"1292_CR17","unstructured":"Hannun, A., Case, C., Casper, J., Catanzaro, B., Diamos, G., Elsen, E., & Ng, A.Y (2014). Deep speech: Scaling up end-to-end speech recognition. arXiv: http:\/\/arXiv.org\/abs\/1412.5567 ."},{"key":"1292_CR18","doi-asserted-by":"crossref","unstructured":"Sak, H., Senior, A., Rao, K., Irsoy, O., Graves, A., Beaufays, F., & Schalkwyk, J. (2015). Learning acoustic frame labeling for speech recognition with recurrent neural networks. In IEEE International conference on acoustics, speech and signal processing (ICASSP) (pp. 4280\u20134284).","DOI":"10.1109\/ICASSP.2015.7178778"},{"key":"1292_CR19","unstructured":"Amodei, D., Ananthanarayanan, S., Anubhai, R., Battenberg, E., Case, C., & Chen, J. (2016). Deep speech 2: End-to-end speech recognition in english and mandarin. In International conference on machine learning (pp. 173\u2013182)."},{"key":"1292_CR20","doi-asserted-by":"crossref","unstructured":"Miao, Y., Gowayyed, M., & Metze, F. (2015). EESEN: End-to-end speech recognition using deep RNN models and WFST-based decoding. In IEEE Workshop on automatic speech recognition and understanding (ASRU) (pp. 167\u2013174).","DOI":"10.1109\/ASRU.2015.7404790"},{"key":"1292_CR21","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, H., Cai, X., & Xu, B. (2015). Towards end-to-end speech recognition for chinese mandarin using long short-term memory recurrent neural networks. In Sixteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2015-717"},{"key":"1292_CR22","unstructured":"Bahdanau, D., Serdyuk, D., Brakel, P., Ke, N. R., Chorowski, J., Courville, A., & Bengio, Y (2015). Task loss estimation for sequence prediction. arXiv: http:\/\/arXiv.org\/abs\/1511.06456 ."},{"key":"1292_CR23","unstructured":"Sak, H., Senior, A., Rao, K., & Beaufays, F (2015). Fast and accurate recurrent neural network acoustic models for speech recognition. arXiv: http:\/\/arXiv.org\/abs\/1507.06947 ."},{"key":"1292_CR24","unstructured":"Sak, H., de Chaumont Quitry, F., & Rao, K (2015). Acoustic modelling with cd-ctc-smbr lstm rnns. In IEEE workshop on automatic speech recognition and understanding (ASRU) (pp. 604\u2013609)."},{"key":"1292_CR25","doi-asserted-by":"crossref","unstructured":"Kingsbury, B. (2009). Lattice-based optimization of sequence classification criteria for neural-network acoustic modeling. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 3761\u20133764).","DOI":"10.1109\/ICASSP.2009.4960445"},{"key":"1292_CR26","doi-asserted-by":"crossref","unstructured":"Povey, D., & Woodland, P. C. (2002). Minimum phone error and I-smoothing for improved discriminative training. In IEEE international conference on acoustics, speech, and signal processing (ICASSP) (Vol. 1, pp.1\u2013105).","DOI":"10.1109\/ICASSP.2002.1005687"},{"key":"1292_CR27","doi-asserted-by":"crossref","unstructured":"Povey, D., Peddinti, V., Galvez, D., Ghahrmani, P., Manohar, V., Na, X., & Khudanpur, S (2016). Purely sequence-trained neural networks for ASR based on lattice-free MMI. In INTERSPEECH (pp. 2751\u20132755).","DOI":"10.21437\/Interspeech.2016-595"},{"key":"1292_CR28","doi-asserted-by":"crossref","unstructured":"Graves, A., Fern\u00e1ndez, S., Gomez, F., & Schmidhuber, J. (2006). Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In Proceedings of the 31st international conference on machine learning (ICML) (pp. 369\u2013376).","DOI":"10.1145\/1143844.1143891"},{"issue":"1","key":"1292_CR29","doi-asserted-by":"crossref","first-page":"69","DOI":"10.1006\/csla.2001.0184","volume":"16","author":"M Mohri","year":"2002","unstructured":"Mohri, M., Pereira, F., & Riley, M (2002). Weighted finite-state transducers in speech recognition. Computer Speech & Language, 16(1), 69\u201388.","journal-title":"Computer Speech & Language"},{"issue":"1","key":"1292_CR30","doi-asserted-by":"crossref","first-page":"90","DOI":"10.1109\/TASLP.2016.2625459","volume":"25","author":"Z Chen","year":"2017","unstructured":"Chen, Z., Zhuang, Y., Qian, Y., & Yu, K. (2017). Phone synchronous speech recognition with CTC lattices. IEEE\/ACM Transactions on Audio Speech and Language Processing (TASLP), 25(1), 90\u2013101.","journal-title":"IEEE\/ACM Transactions on Audio Speech and Language Processing (TASLP)"},{"key":"1292_CR31","doi-asserted-by":"crossref","unstructured":"Sak, H., Vinyals, O., Heigold, G., Senior, A., McDermott, E., Monga, R., & Mao, M (2014). Sequence discriminative distributed training of long short-term memory recurrent neural networks In Fifteenth annual conference of the international speech communication association.","DOI":"10.21437\/Interspeech.2014-305"},{"issue":"5","key":"1292_CR32","doi-asserted-by":"crossref","first-page":"1023","DOI":"10.1109\/TASLP.2017.2678162","volume":"25","author":"N Kanda","year":"2017","unstructured":"Kanda, N., Lu, X., & Kawai, H (2017). Maximum-a-posteriori-based decoding for end-to-end acoustic models. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 25(5), 1023\u20131034.","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"1292_CR33","unstructured":"Harper, M (2014). IARPA babel program. https:\/\/www.iarpa.gov\/index.php\/research-programs\/babel ."},{"key":"1292_CR34","unstructured":"https:\/\/www.nist.gov\/itl\/iad\/mig\/openkws16-evaluation ."},{"key":"1292_CR35","unstructured":"Povey, D., Ghoshal, A., Boulianne, G., Burget, L., Glembek, O., Goel, N., & Silovsky, J. (2011). The Kaldi speech recognition toolkit. In IEEE workshop on automatic speech recognition and understanding (ASRU)."},{"key":"1292_CR36","doi-asserted-by":"crossref","unstructured":"Miao, Y., & Metze, F. (2015). On speaker adaptation of long short-term memory recurrent neural networks. In INTERSPEECH (pp. 1101\u20131105).","DOI":"10.21437\/Interspeech.2015-290"},{"key":"1292_CR37","doi-asserted-by":"crossref","unstructured":"Miao, Y., Gowayyed, M., Na, X., Ko, T., Metze, F., & Waibel, A. (2016). An empirical exploration of CTC acoustic models. In IEEE International conference on acoustics, speech and signal processing (ICASSP) (pp. 2623\u20132627).","DOI":"10.1109\/ICASSP.2016.7472152"},{"key":"1292_CR38","doi-asserted-by":"crossref","unstructured":"Peddinti, V., Povey, D., & Khudanpur, S. (2015). A time delay neural network architecture for efficient modeling of long temporal contexts. In INTERSPEECH (pp. 2440\u20132444).","DOI":"10.21437\/Interspeech.2015-647"},{"key":"1292_CR39","doi-asserted-by":"crossref","unstructured":"Huang, J. T., Li, J., Yu, D., Deng, L., & Gong, Y. (2013). Cross-language knowledge transfer using multilingual deep neural network with shared hidden layers. In IEEE International conference on acoustics, speech and signal processing (ICASSP) (pp. 7304\u20137308).","DOI":"10.1109\/ICASSP.2013.6639081"}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-017-1292-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-017-1292-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-017-1292-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,3]],"date-time":"2022-08-03T20:25:29Z","timestamp":1659558329000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-017-1292-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,9,30]]},"references-count":39,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2018,7]]}},"alternative-id":["1292"],"URL":"https:\/\/doi.org\/10.1007\/s11265-017-1292-0","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"value":"1939-8018","type":"print"},{"value":"1939-8115","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,9,30]]}}}