{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,13]],"date-time":"2025-12-13T23:07:31Z","timestamp":1765667251574},"publisher-location":"Cham","reference-count":28,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030275280"},{"type":"electronic","value":"9783030275297"}],"license":[{"start":{"date-parts":[[2019,1,1]],"date-time":"2019-01-01T00:00:00Z","timestamp":1546300800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-27529-7_29","type":"book-chapter","created":{"date-parts":[[2019,8,5]],"date-time":"2019-08-05T13:03:09Z","timestamp":1565010189000},"page":"332-341","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":21,"title":["Towards End-to-End Speech Recognition with Deep Multipath Convolutional Neural Networks"],"prefix":"10.1007","author":[{"given":"Wei","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minghao","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zilong","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chen","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Cao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,8,6]]},"reference":[{"key":"29_CR1","unstructured":"Lecun, Y., Bengio, Y.: Convolutional networks for images, speech, and time series. In: The Handbook of Brain Theory and Neural Networks. MIT Press, USA (1995)"},{"issue":"10","key":"29_CR2","doi-asserted-by":"publisher","first-page":"1533","DOI":"10.1109\/TASLP.2014.2339736","volume":"22","author":"HO Abdel","year":"2014","unstructured":"Abdel, H.O., Mohamed, A.R., Jiang, H.: Convolutional neural networks for speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 22(10), 1533\u20131545 (2014)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"1","key":"29_CR3","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1109\/TASL.2011.2109382","volume":"20","author":"A Mohamed","year":"2012","unstructured":"Mohamed, A., Dahl, G.E., Hinton, G.E.: Acoustic modeling using deep belief networks. IEEE Trans. Audio Speech Lang. Process. 20(1), 14\u201322 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"6","key":"29_CR4","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"GE Hinton","year":"2012","unstructured":"Hinton, G.E., Deng, L., Yu, D.: Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process. Mag. 29(6), 82\u201397 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"29_CR5","unstructured":"Abdel, H.O., Mohamed, A.R., Jiang, H.: Applying convolutional neural networks concepts to hybrid NN-HMM model for speech recognition. In: International Conference on Acoustics, Speech and Signal Processing, pp. 4277\u20134280. IEEE, Kyoto, May 2012"},{"key":"29_CR6","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Mohamed, A.R., Kingsbury, B.: Deep convolutional neural networks for LVCSR. In: International Conference on Acoustics, Speech and Signal Processing, pp. 8614\u20138618. IEEE, Vancouver, May 2013","DOI":"10.1109\/ICASSP.2013.6639347"},{"key":"29_CR7","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Pezeshki, M., Brakel, P.: Towards end-to-end speech recognition with deep convolutional neural networks. arXiv preprint arXiv:1701.02720 , January 2017","DOI":"10.21437\/Interspeech.2016-1446"},{"key":"29_CR8","doi-asserted-by":"crossref","unstructured":"Qian, Y.M., Woodland, P.C.: Very deep convolutional neural networks for robust speech recognition. In: Spoken Language Technology Workshop, pp. 481\u2013488. IEEE, Berkeley, June 2017","DOI":"10.1109\/SLT.2016.7846307"},{"key":"29_CR9","doi-asserted-by":"crossref","unstructured":"Bahdanau, D., Chorowski, J., Serdyuk, D.: End-to-End attention-based large vocabulary speech recognition. In: International Conference on Acoustics, Speech and Signal Processing, pp. 4945\u20134949. IEEE, Shanghai, March 2016","DOI":"10.1109\/ICASSP.2016.7472618"},{"key":"29_CR10","doi-asserted-by":"crossref","unstructured":"Miao, Y.J., Gowayyed, M., Metze, F.: EESEN: end-to-end speech recognition using deep RNN models and WFST-based decoding. arXiv preprint arXiv:1507.08240 , October 2015","DOI":"10.1109\/ASRU.2015.7404790"},{"key":"29_CR11","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"180","DOI":"10.1007\/978-3-319-25816-4_15","volume-title":"Chinese Computational Linguistics and Natural Language Processing Based on Naturally Annotated Big Data","author":"H Zhang","year":"2015","unstructured":"Zhang, H., Bao, F., Gao, G.: Mongolian speech recognition based on deep neural networks. In: Sun, M., Liu, Z., Zhang, M., Liu, Y. (eds.) CCL 2015. LNCS (LNAI), vol. 9427, pp. 180\u2013188. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-25816-4_15"},{"issue":"8","key":"29_CR12","doi-asserted-by":"publisher","first-page":"1393","DOI":"10.1109\/TASLP.2018.2825432","volume":"26","author":"T Tan","year":"2018","unstructured":"Tan, T., Qian, Y.M., Hu, H.: Adaptive very deep convolutional residual network for noise robust speech recognition. IEEE\/ACM Trans. Audio Speech Lang. Process. 26(8), 1393\u20131405 (2018)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"29_CR13","doi-asserted-by":"crossref","unstructured":"Graves, A., Santiago, F., Gomez, F.: Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: International Conference on Machine Learning, pp. 369\u2013376. IEEE, Pittsburgh, June 2006","DOI":"10.1145\/1143844.1143891"},{"key":"29_CR14","unstructured":"Graves, A., Mohamed, A., Hinton, G.E.: Speech recognition with deep recurrent neural networks. In: International Conference on Acoustics, Speech and Signal Processing, pp. 6645\u20136649. IEEE, Hong Kong, April 2003"},{"key":"29_CR15","unstructured":"Hannun, A., Case, C., Casper, J.: Deep speech: scaling up end-to-end speech recognition. arXiv preprint arXiv:1412.5567 (2014)"},{"key":"29_CR16","unstructured":"Amodei, D., Anubhai, R., Battenberg, F.: Deep speech 2: end-to-end speech recognition in English and Mandarin. arXiv preprint arXiv:1512.02595 (2015)"},{"key":"29_CR17","unstructured":"Wang, Y., Deng, X., Pu, S.: Residual convolutional CTC networks for automatic speech recognition. arXiv preprint arXiv:1702.07793 , February 2017"},{"key":"29_CR18","doi-asserted-by":"crossref","unstructured":"Li, J., Zhang, H., Cai, X.Y.: Towards end-to-end speech recognition for Chinese Mandarin using long short-term memory recurrent neural networks. In: Interspeech 2015, pp. 3615\u20133619. IEEE, Berlin, September 2015","DOI":"10.21437\/Interspeech.2015-717"},{"key":"29_CR19","doi-asserted-by":"crossref","unstructured":"Zhou, S.Y., Dong, L.H., Xu, S., Xu, B.: Syllable-based sequence-to-sequence speech recognition with the transformer in Mandarin Chinese. arXiv preprint arXiv:1804.10752 , June 2018","DOI":"10.21437\/Interspeech.2018-1107"},{"key":"29_CR20","doi-asserted-by":"crossref","unstructured":"Zou, W., Jiang, D.W., Zhao, S.J., Li, X.G.: A comparable study of modeling units for end-to-end Mandarin speech recognition. arXiv preprint arXiv:1805.03832 , May 2018","DOI":"10.1109\/ISCSLP.2018.8706661"},{"key":"29_CR21","doi-asserted-by":"crossref","unstructured":"Dong, L.H., Xu, S., Xu, B.: Speech-transformer: a no-recurrence sequence-to-sequence model for speech recognition. In: International Conference on Acoustics, Speech and Signal Processing, pp. 4437\u20134441. IEEE, Calgary, April 2018","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"29_CR22","unstructured":"Kingma, D., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 , July 2015"},{"issue":"1","key":"29_CR23","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G.E., Krizhevsky, A.: Dropout: a simple way to prevent neural networks from overfitting. J. Mach. Learn. Res. 15(1), 1929\u20131958 (2014)","journal-title":"J. Mach. Learn. Res."},{"key":"29_CR24","volume-title":"Machine Learning","author":"ZH Zhou","year":"2016","unstructured":"Zhou, Z.H.: Machine Learning. Tsinghua University Press, Beijing (2016)"},{"key":"29_CR25","unstructured":"Simonyan, K., Andrew, Z.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2015)"},{"key":"29_CR26","unstructured":"Awni, Y.H., Andrew, L.M., Daniel, J.: First-pass large vocabulary continuous speech recognition using bi-directional recurrent DNNs. arXiv preprint arXiv:1408.2873 , December 2014"},{"key":"29_CR27","unstructured":"Wang, D., Zhang, X.: THCHS-30: a free chinese speech corpus. arXiv preprint arXiv:1512.01882 , December 2015"},{"key":"29_CR28","unstructured":"Zhang, L.M., Wang, Y.Z., Zhang, B.Q.: Chinese Mandarin recognition and improvement based on CTC criterion. Comput. Eng. (2019)"}],"container-title":["Lecture Notes in Computer Science","Intelligent Robotics and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-27529-7_29","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,25]],"date-time":"2022-09-25T05:13:06Z","timestamp":1664082786000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-27529-7_29"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019]]},"ISBN":["9783030275280","9783030275297"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-27529-7_29","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2019]]},"assertion":[{"value":"6 August 2019","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIRA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Robotics and Applications","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Shenyang","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 August 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 August 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icira2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.icira2019.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}