{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,7,7]],"date-time":"2023-07-07T22:15:38Z","timestamp":1688768138977},"reference-count":37,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"12","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2022,12,1]]},"DOI":"10.1587\/transinf.2022edp7043","type":"journal-article","created":{"date-parts":[[2022,11,30]],"date-time":"2022-11-30T22:18:29Z","timestamp":1669846709000},"page":"2112-2118","source":"Crossref","is-referenced-by-count":1,"title":["Robust Speech Recognition Using Teacher-Student Learning Domain Adaptation"],"prefix":"10.1587","volume":"E105.D","author":[{"given":"Han","family":"MA","sequence":"first","affiliation":[{"name":"School of Information Science and Technology, Zhejiang Sci-Tech University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiaoling","family":"ZHANG","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, Zhejiang Sci-Tech University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Roubing","family":"TANG","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, Zhejiang Sci-Tech University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lu","family":"ZHANG","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, Zhejiang Sci-Tech University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yubo","family":"JIA","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, Zhejiang Sci-Tech University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] M.B. Hoy, \u201cAlexa, siri, cortana, and more: an introduction to voice assistants,\u201d Medical reference services quarterly, vol.37, no.1, pp.81-88, 2018. 10.1080\/02763869.2018.1404391","DOI":"10.1080\/02763869.2018.1404391"},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] G. Hinton, L. Deng, D. Yu, G.E. Dahl, A.-r. Mohamed, N. Jaitly, A. Senior, V. Vanhoucke, P. Nguyen, T.N. Sainath, and B. Kingsbury, \u201cDeep neural networks for acoustic modeling in speech recognition: The shared views of four research groups,\u201d IEEE Signal processing magazine, vol.29, no.6, pp.82-97, 2012. 10.1109\/msp.2012.2205597","DOI":"10.1109\/MSP.2012.2205597"},{"key":"3","doi-asserted-by":"publisher","unstructured":"[3] O. Abdel-Hamid, A.-r. Mohamed, H. Jiang, L. Deng, G. Penn, and D. Yu, \u201cConvolutional neural networks for speech recognition,\u201d IEEE\/ACM Transactions on audio, speech, and language processing, vol.22, no.10, pp.1533-1545, 2014. 10.1109\/taslp.2014.2339736","DOI":"10.1109\/TASLP.2014.2339736"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] H. Sak, A.W. Senior, and F. Beaufays, \u201cLong short-term memory recurrent neural network architectures for large scale acoustic modeling,\u201d Interspeech 2014, pp.338-342, 2014. 10.21437\/interspeech.2014-80","DOI":"10.21437\/Interspeech.2014-80"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] V. Peddinti, D. Povey, and S. Khudanpur, \u201cA time delay neural network architecture for efficient modeling of long temporal contexts,\u201d Sixteenth annual conference of the international speech communication association, pp.3214-3218, 2015. 10.21437\/interspeech.2015-647","DOI":"10.21437\/Interspeech.2015-647"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] T.N. Sainath, O. Vinyals, A. Senior, and H. Sak, \u201cConvolutional, long short-term memory, fully connected deep neural networks,\u201d 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp.4580-4584, IEEE, 2015. 10.1109\/icassp.2015.7178838","DOI":"10.1109\/ICASSP.2015.7178838"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] J. Li, V. Lavrukhin, B. Ginsburg, R. Leary, O. Kuchaiev, J.M. Cohen, H. Nguyen, and R.T. Gadde, \u201cJasper: An end-to-end convolutional neural acoustic model,\u201d arXiv preprint arXiv:1904.03288, 2019.","DOI":"10.21437\/Interspeech.2019-1819"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] A. Graves, \u201cSequence transduction with recurrent neural networks,\u201d arXiv preprint arXiv:1211.3711, 2012.","DOI":"10.1007\/978-3-642-24797-2_3"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] A. Graves, A.-r. Mohamed, and G. Hinton, \u201cSpeech recognition with deep recurrent neural networks,\u201d 2013 IEEE international conference on acoustics, speech and signal processing, pp.6645-6649, Ieee, 2013. 10.1109\/icassp.2013.6638947","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"10","unstructured":"[10] W. Chan, N. Jaitly, Q.V. Le, and O. Vinyals, \u201cListen, attend and spell,\u201d arXiv preprint arXiv:1508.01211, 2015."},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] L. Dong, S. Xu, and B. Xu, \u201cSpeech-transformer: a no-recurrence sequence-to-sequence model for speech recognition,\u201d 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.5884-5888, IEEE, 2018. 10.1109\/icassp.2018.8462506","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] W. Han, Z. Zhang, Y. Zhang, J. Yu, C.C. Chiu, J. Qin, A. Gulati, R. Pang, and Y. Wu, \u201cContextnet: Improving convolutional neural networks for automatic speech recognition with global context,\u201d arXiv preprint arXiv:2005.03191, 2020.","DOI":"10.21437\/Interspeech.2020-2059"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] A. Gulati, J. Qin, C.C. Chiu, N. Parmar, Y. Zhang, J. Yu, W. Han, S. Wang, Z. Zhang, Y. Wu, et al., \u201cConformer: Convolution-augmented transformer for speech recognition,\u201d arXiv preprint arXiv:2005.08100, 2020.","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"14","doi-asserted-by":"publisher","unstructured":"[14] J. Li, L. Deng, Y. Gong, and R. Haeb-Umbach, \u201cAn overview of noise-robust automatic speech recognition,\u201d IEEE\/ACM Transactions on Audio, Speech, and Language Processing, vol.22, no.4, pp.745-777, 2014. 10.1109\/taslp.2014.2304637","DOI":"10.1109\/TASLP.2014.2304637"},{"key":"15","doi-asserted-by":"publisher","unstructured":"[15] S. Sun, B. Zhang, L. Xie, and Y. Zhang, \u201cAn unsupervised deep domain adaptation approach for robust speech recognition,\u201d Neurocomputing, vol.257, pp.79-87, 2017. 10.1016\/j.neucom.2016.11.063","DOI":"10.1016\/j.neucom.2016.11.063"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] Z. Meng, Z. Chen, V. Mazalov, J. Li, and Y. Gong, \u201cUnsupervised adaptation with domain separation networks for robust speech recognition,\u201d 2017 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.214-221, IEEE, 2017. 10.1109\/asru.2017.8268938","DOI":"10.1109\/ASRU.2017.8268938"},{"key":"17","unstructured":"[17] G. Hinton, O. Vinyals, and J. Dean, \u201cDistilling the knowledge in a neural network,\u201d arXiv preprint arXiv:1503.02531, 2015."},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] Y. Chebotar and A. Waters, \u201cDistilling knowledge from ensembles of neural networks for speech recognition,\u201d Interspeech, pp.3439-3443, 2016. 10.21437\/interspeech.2016-1190","DOI":"10.21437\/Interspeech.2016-1190"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] Y. Kim and A.M. Rush, \u201cSequence-level knowledge distillation,\u201d arXiv preprint arXiv:1606.07947, 2016.","DOI":"10.18653\/v1\/D16-1139"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] R. Pang, T. Sainath, R. Prabhavalkar, S. Gupta, Y. Wu, S. Zhang, and C.-C. Chiu, \u201cCompression of end-to-end models,\u201d Interspeech 2018, pp.27-31, 2018. 10.21437\/interspeech.2018-1025","DOI":"10.21437\/Interspeech.2018-1025"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] J. Li, M.L. Seltzer, X. Wang, R. Zhao, and Y. Gong, \u201cLarge-scale domain adaptation via teacher-student learning,\u201d arXiv preprint arXiv:1708.05466, 2017.","DOI":"10.21437\/Interspeech.2017-519"},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] L. Mo\u0161ner, M. Wu, A. Raju, S.H.K. Parthasarathi, K. Kumatani, S. Sundaram, R. Maas, and B. Hoffmeister, \u201cImproving noise robustness of automatic speech recognition via parallel data and teacher-student learning,\u201d ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.6475-6479, IEEE, 2019. 10.1109\/icassp.2019.8683422","DOI":"10.1109\/ICASSP.2019.8683422"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] Z. Meng, J. Li, Y. Zhao, and Y. Gong, \u201cConditional teacher-student learning,\u201d ICASSP 2019-2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.6445-6449, IEEE, 2019. 10.1109\/icassp.2019.8683438","DOI":"10.1109\/ICASSP.2019.8683438"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] Z. Meng, J. Li, Y. Gaur, and Y. Gong, \u201cDomain adaptation via teacher-student learning for end-to-end speech recognition,\u201d 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.268-275, IEEE, 2019. 10.1109\/asru46091.2019.9003776","DOI":"10.1109\/ASRU46091.2019.9003776"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] H. Zhu, J. Zhao, Y. Ren, L. Wang, and P. Zhang, \u201cDomain adaptation using class similarity for robust speech recognition,\u201d arXiv preprint arXiv:2011.02782, 2020.","DOI":"10.21437\/Interspeech.2020-3087"},{"key":"26","unstructured":"[26] A. Romero, N. Ballas, S.E. Kahou, A. Chassang, C. Gatta, and Y. Bengio, \u201cFitnets: Hints for thin deep nets,\u201d arXiv preprint arXiv:1412.6550, 2014."},{"key":"27","unstructured":"[27] S. Zagoruyko and N. Komodakis, \u201cPaying more attention to attention: Improving the performance of convolutional neural networks via attention transfer,\u201d arXiv preprint arXiv:1612.03928, 2016."},{"key":"28","doi-asserted-by":"crossref","unstructured":"[28] N. Passalis and A. Tefas, \u201cLearning deep representations with probabilistic knowledge transfer,\u201d Proceedings of the European Conference on Computer Vision (ECCV), vol.11215, pp.283-299, 2018. 10.1007\/978-3-030-01252-6_17","DOI":"10.1007\/978-3-030-01252-6_17"},{"key":"29","unstructured":"[29] B. Zhang, D. Wu, C. Yang, X. Chen, Z. Peng, X. Wang, Z. Yao, X. Wang, F. Yu, L. Xie, et al., \u201cWenet: Production first and production ready end-to-end speech recognition toolkit,\u201d arXiv preprint arXiv:2102.01547, 2021."},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] S. Kim, T. Hori, and S. Watanabe, \u201cJoint ctc-attention based end-to-end speech recognition using multi-task learning,\u201d 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp.4835-4839, IEEE, 2017. 10.1109\/icassp.2017.7953075","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"31","doi-asserted-by":"publisher","unstructured":"[31] S. Watanabe, T. Hori, S. Kim, J.R. Hershey, and T. Hayashi, \u201cHybrid ctc\/attention architecture for end-to-end speech recognition,\u201d IEEE Journal of Selected Topics in Signal Processing, vol.11, no.8, pp.1240-1253, 2017. 10.1109\/jstsp.2017.2763455","DOI":"10.1109\/JSTSP.2017.2763455"},{"key":"32","unstructured":"[32] Z. Yuan, Z. Lyu, J. Li, and X. Zhou, \u201cAn improved hybrid ctc-attention model for speech recognition,\u201d arXiv preprint arXiv:1810.12020, 2018."},{"key":"33","doi-asserted-by":"crossref","unstructured":"[33] A. Sriram, H. Jun, Y. Gaur, and S. Satheesh, \u201cRobust speech recognition using generative adversarial networks,\u201d 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp.5639-5643, IEEE, 2018. 10.1109\/icassp.2018.8462456","DOI":"10.1109\/ICASSP.2018.8462456"},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] S. Zhang, C.-T. Do, R. Doddipatla, and S. Renals, \u201cLearning noise invariant features through transfer learning for robust end-to-end speech recognition,\u201d ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.7024-7028, IEEE, 2020. 10.1109\/icassp40776.2020.9053169","DOI":"10.1109\/ICASSP40776.2020.9053169"},{"key":"35","doi-asserted-by":"crossref","unstructured":"[35] H. Bu, J. Du, X. Na, B. Wu, and H. Zheng, \u201cAishell-1: An open-source mandarin speech corpus and a speech recognition baseline,\u201d 2017 20th Conference of the Oriental Chapter of the International Coordinating Committee on Speech Databases and Speech I\/O Systems and Assessment (O-COCOSDA), pp.1-5, IEEE, 2017. 10.1109\/icsda.2017.8384449","DOI":"10.1109\/ICSDA.2017.8384449"},{"key":"36","doi-asserted-by":"crossref","unstructured":"[36] D.S. Park, W. Chan, Y. Zhang, C.C. Chiu, B. Zoph, E.D. Cubuk, and Q.V. Le, \u201cSpecaugment: A simple data augmentation method for automatic speech recognition,\u201d arXiv preprint arXiv:1904.08779, 2019.","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"37","doi-asserted-by":"crossref","unstructured":"[37] S. Watanabe, T. Hori, S. Karita, T. Hayashi, J. Nishitoba, Y. Unno, N. Enrique Yalta Soplin, J. Heymann, M. Wiesner, N. Chen, A. Renduchintala, and T. Ochiai, \u201cESPnet: End-to-end speech processing toolkit,\u201d Proceedings of Interspeech, pp.2207-2211, 2018. 10.21437\/interspeech.2018-1456","DOI":"10.21437\/Interspeech.2018-1456"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E105.D\/12\/E105.D_2022EDP7043\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,3]],"date-time":"2022-12-03T04:11:44Z","timestamp":1670040704000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E105.D\/12\/E105.D_2022EDP7043\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,1]]},"references-count":37,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2022]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2022edp7043","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,12,1]]},"article-number":"2022EDP7043"}}