{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T07:08:41Z","timestamp":1781334521222,"version":"3.54.1"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2019,1,14]],"date-time":"2019-01-14T00:00:00Z","timestamp":1547424000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J AUDIO SPEECH MUSIC PROC."],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1186\/s13636-018-0146-4","type":"journal-article","created":{"date-parts":[[2019,1,14]],"date-time":"2019-01-14T09:04:46Z","timestamp":1547456686000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":16,"title":["Dual supervised learning for non-native speech recognition"],"prefix":"10.1186","volume":"2019","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0051-2762","authenticated-orcid":false,"given":"Kacper","family":"Radzikowski","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Robert","family":"Nowak","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Le","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Osamu","family":"Yoshie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,1,14]]},"reference":[{"key":"146_CR1","first-page":"5934","volume-title":"The microsoft 2017 conversational speech recognition system","author":"W. Xiong","year":"2018","unstructured":"W. Xiong, L. Wu, J. Droppo, X. Huang, A. Stolcke, in The microsoft 2017 conversational speech recognition system. Proc. IEEE ICASSP (IEEECalgary, 2018), pp. 5934\u20135938. \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/8461870\n                    \n                  ."},{"key":"146_CR2","first-page":"1","volume":"1","author":"N. Dave","year":"2013","unstructured":"N. Dave, Feature extraction methods lpc, plp and mfcc in speech recognition. International Journal of Advanced Research Engineering and Technology. 1:, 1\u20135 (2013).","journal-title":"International Journal of Advanced Research Engineering and Technology"},{"issue":"4","key":"146_CR3","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1109\/TASL.2010.2064307","volume":"19","author":"N Dehak","year":"2011","unstructured":"N Dehak, D. R. P. J. Kenny, P. O. P. Dumouchel, Front-end factor analysis for speaker verification. IEEE Transactions on Audio, Speech and Language Proceedings. 19(4), 788\u2013798 (2011).","journal-title":"IEEE Transactions on Audio, Speech and Language Proceedings"},{"issue":"1, January 2013","key":"146_CR4","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1016\/j.csl.2012.01.008","volume":"27","author":"M. Li","year":"2013","unstructured":"M. Li, N. S. K. J. Han, Automatic speaker age and gender recognition using acoustic and prosodic level information fusion. Computer Speech Lang. 27(1, January 2013), 151\u201367 (2013). \n                    https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0885230812000101\n                    \n                  .","journal-title":"Computer Speech Lang"},{"key":"146_CR5","volume-title":"Glottal closure and opening instant detection from speech signals","author":"T. T. D. Drugman","year":"2009","unstructured":"T. T. D. Drugman, Glottal closure and opening instant detection from speech signals (International Speech Communication Association (ISCA), Brighton, 2009). \n                    https:\/\/www.isca-speech.org\/archive\/interspeech_2009\/i09_2891.html\n                    \n                  ."},{"key":"146_CR6","unstructured":"G. P. A. Shi, M. Shanechi, On the importance of phase in human speech recognition. Audio, Speech, and Language Processing, IEEE Transactions. 14(5) (2006). Published in: IEEE Transactions on Audio, Speech, and Language Processing. \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/1678004\n                    \n                  ."},{"key":"146_CR7","unstructured":"L. M. Tomokiyo, Recognizing non-native speech: Characterizing and adapting to non-native usage in lvcsr. PhD thesis, Carnegie Mellon University, (2001)."},{"key":"146_CR8","volume-title":"Automatic speech recognition for non-native speakers. PhD thesis","author":"T. P. Tan","year":"2008","unstructured":"T. P. Tan, Automatic speech recognition for non-native speakers. PhD thesis (Universit\u00e9 Joseph-Fourier, Grenoble, 2008). \n                    https:\/\/hal.inria.fr\/tel-00294973\/\n                    \n                  ."},{"key":"146_CR9","volume-title":"Non-native english speaker\u2019s speech correction, based on domain focused document","author":"R. Kacper","year":"2016","unstructured":"R. Kacper, W. Le, Y. Osamu, in Non-native english speaker\u2019s speech correction, based on domain focused document. Proceedings of the Conference of Institute of Electrical Engineers of Japan, Electronics and Information Systems Division (The Institute of Electrical Engineers of JapanKobe, 2016). \n                    https:\/\/ieej.ixsq.nii.ac.jp\/ej\/?active_action=repository_view_main_item_detail&page_id=13&block_id=18&item_id=88519&item_no=1\n                    \n                  ."},{"key":"146_CR10","first-page":"276","volume-title":"Non-native english speakers\u2019 speech correction, based on domain focused document","author":"R. Kacper","year":"2016","unstructured":"R. Kacper, W. Le, Y. Osamu, in Non-native english speakers\u2019 speech correction, based on domain focused document. Proceedings of the 18th International Conference on Information Integration and Web-based Applications and Services, iiWAS (ACMNew York, 2016), pp. 276\u2013281."},{"key":"146_CR11","volume-title":"Proceedings of the conference of institute of electrical engineers of japan, electronics and information systems division","author":"R. Kacper","year":"2017","unstructured":"R. Kacper, Y. O. W. Le, in Proceedings of the conference of institute of electrical engineers of japan, electronics and information systems division. Non-native speech recognition using characteristic speech features, with respect to nationality (The Institute of Electrical Engineers of JapanTakamatsu, 2017). \n                    https:\/\/ieej.ixsq.nii.ac.jp\/ej\/?action=pages_view_main&active_action=repository_view_main_item_detail&item_id=97730&item_no=1&page_id=13&block_id=18\n                    \n                  ."},{"key":"146_CR12","first-page":"3789","volume-title":"Proceedings of Machine Learning Research. vol. 70, Dual supervised learning","author":"Y. Xia","year":"2017","unstructured":"Y. Xia, T. Qin, W. Chen, J. Bian, N. Yu, T. -Y. Liu, in Proceedings of Machine Learning Research. vol. 70, Dual supervised learning. Proceedings of the 34th International Conference on Machine Learning (International Convention CentreSydney, 2017), pp. 3789\u20133798."},{"key":"146_CR13","doi-asserted-by":"publisher","unstructured":"K. Livescu, J. Glass, in IEEE International Conference on Acoustics, Speech and Signal Processing. Lexical modeling of non-native speech for automatic speech recognition. Published in: 2000 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No.00CH37100). (IEEE, Istanbul, 2000). \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/862074\n                    \n                  . \n                    https:\/\/doi.org\/10.1109\/ICASSP.2007.367243\n                    \n                  .","DOI":"10.1109\/ICASSP.2007.367243"},{"issue":"6, June 1995","key":"146_CR14","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1109\/97.388911","volume":"2","author":"F. Bimbot Dept. Signal","year":"1995","unstructured":"F. Bimbot Dept. Signal, P.F..R.P..E.L..B.A. ENST: Variable-length sequence modeling: multigrams. IEEE Signal Processing Letters. IEEE Signal Proc Lett. 2(6, June 1995), 111\u201313 (1995). \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/388911\n                    \n                  .","journal-title":"IEEE Signal Proc Lett"},{"key":"146_CR15","volume-title":"Language modeling by variable length sequences: theoretical formulation and evaluation of multigrams","author":"S. Deligne Telecom Paris","year":"1995","unstructured":"S. Deligne Telecom Paris, in Language modeling by variable length sequences: theoretical formulation and evaluation of multigrams. FFB., in International Conference on Acoustics, Speech, and Signal Processing (IEEE ConferenceDetroit, 1995). \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/479391\n                    \n                  ."},{"key":"146_CR16","unstructured":"T. Tan, L. Besacier, Acoustic model interpolation for non-native speech recognition. Proceedings on ICASSP. Published in: 2000 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No.00CH37100). (IEEE, Honolulu, 2007)."},{"key":"146_CR17","doi-asserted-by":"crossref","unstructured":"G. E. Hinton, L. Deng, D. Yu, G. E. Dahl, A. Mohamed, et al., N.J.: Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Processing Magazine. 29(6) (2012). \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/6296526\n                    \n                  . \n                    http:\/\/dx.doi.org\/10.1109\/MSP.2012.2205597\n                    \n                  .","DOI":"10.1109\/MSP.2012.2205597"},{"key":"146_CR18","volume-title":"Proceedings in Interspeech","author":"T.G.J.","year":"2014","unstructured":"T.G.J., Z.Z., F.W., B.S., G.R., in Proceedings in Interspeech. Robust speech recognition using long short-term memory recurrent neural networks for hybrid acoustic modelling (International Speech Communication Association (ISCA)Singapore, 2014). \n                    https:\/\/www.isca-speech.org\/archive\/interspeech_2014\/i14_0631.html\n                    \n                  ."},{"issue":"8","key":"146_CR19","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S. Hochreiter","year":"1997","unstructured":"S. Hochreiter, J. Schmidhuber, Long short-term memory. Neural Computer. 9(8), 1735\u20131780 (1997).","journal-title":"Neural Computer"},{"key":"146_CR20","first-page":"1045","volume-title":"Recurrent neural network based language model","author":"T. Mikolov","year":"2010","unstructured":"T. Mikolov, M. Karafi\u00e1t, L. Burget, J. \u010cernock\u00fd, S. Khudanpur, in Recurrent neural network based language model. Proceedings of the 11th Annual Conference of the International Speech Communication Association (INTERSPEECH 2010) vol. 2010 (International Speech Communication Association (ISCA)Makuhari, Chiba, Japan, 2010), pp. 1045\u20131048. \n                    https:\/\/www.isca-speech.org\/archive\/interspeech_2010\/i10_1045.html\n                    \n                  . \n                    http:\/\/www.proceedings.com\/10100.html\n                    \n                  ."},{"key":"146_CR21","first-page":"1017","volume-title":"Generating text with recurrent neural networks","author":"I. Sutskever","year":"2011","unstructured":"I. Sutskever, J. Martens, G. Hinton, in Generating text with recurrent neural networks. Proceedings of the 28th International Conference on International Conference on Machine Learning. ICML\u201911 (OmnipressUSA, 2011), pp. 1017\u20131024. \n                    http:\/\/dl.acm.org\/citation.cfm?id=3104482.3104610\n                    \n                  ."},{"key":"146_CR22","unstructured":"A. Graves. Generating sequences with recurrent neural networks, (2013). \n                    https:\/\/arxiv.org\/abs\/1308.0850\n                    \n                  . (arXiv:1308.0850)."},{"key":"146_CR23","unstructured":"A. van den Oord, S. Dieleman, H. Zen, K. Simonyan, O. Vinyals, A. Graves, N. Kalchbrenner, A. Senior, K. Kavukcuoglu, in Wavenet: A generative model for raw audio. Arxiv, (2016). \n                    https:\/\/arxiv.org\/abs\/1609.03499\n                    \n                  . (arXiv:1609.03499)."},{"key":"146_CR24","unstructured":"H. Z. K. Tokuda, Directly modeling speech waveforms by neural networks for statistical parametric speech synthesis. Proceedings ICASSP. 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). (IEEE, Brisbane, 2015). \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/7178765\n                    \n                  ."},{"key":"146_CR25","unstructured":"X. Liu, Deep Convolutional and LSTM Neural Networks for Acoustic Modelling in Automatic Speech Recognition. \n                    http:\/\/cs231n.stanford.edu\/reports\/2017\/pdfs\/804.pdf\n                    \n                  . Accessed 20 July 2018."},{"key":"146_CR26","unstructured":"W. Song, End-to-end deep neural network for automatic speech recognition. Published in: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU). (IEEE, Conference, Scottsdale, AZ, USA, 2015). \n                    http:\/\/cs224d.stanford.edu\/reports\/SongWilliam.pdf\n                    \n                  . \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/7404790\n                    \n                  ."},{"key":"146_CR27","volume-title":"Eesen: End-to-end speech recognition using deep rnn models and wfst-based decoding","author":"Y. Miao","year":"2015","unstructured":"Y. Miao, M. Gowayyed, F. Metze, in Eesen: End-to-end speech recognition using deep rnn models and wfst-based decoding. Published in: 2015 IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU) (IEEEScottsdale, 2015). \n                    https:\/\/arxiv.org\/abs\/1507.08240\n                    \n                  ."},{"key":"146_CR28","unstructured":"A. Graves, N. Jaitly, in Towards end-to-end speech recognition with recurrent neural networks. Proceedings of the 31st International Conference on International Conference on Machine Learning - Volume 32. ICML\u201914 (JMLR.org, 2014), pp. 1764\u20131772. \n                    http:\/\/dl.acm.org\/citation.cfm?id=3044805.3045089\n                    \n                  ."},{"key":"146_CR29","unstructured":"X. Tian, J. Zhang, Z. Ma, Y. He, J. Wei, P. Wu, W. Situ, S. Li, Y. Zhang, Arxiv, (2017). \n                    https:\/\/arxiv.org\/abs\/1703.07090\n                    \n                  . (arXiv:1703.07090)."},{"key":"146_CR30","doi-asserted-by":"crossref","unstructured":"A. Graves, G. H. A. Mohamed, Speech recognition with deep recurrent neural networks. Proc ICASSP IEEE. Published in: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing. (IEEE Conference, Vancouver, 2013). \n                    https:\/\/ieeexplore.ieee.org\/abstract\/document\/6638947\n                    \n                  .","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"146_CR31","unstructured":"D. Amodei, R. Anubhai, E. Battenberg, C. Case, J. Casper, B. Catanzaro, J. Chen, M. Chrzanowski, A. Coates, G. Diamos, E. Elsen, J. Engel, L. Fan, C. Fougner, T. Han, A. Hannun, B. Jun, P. LeGresley, L. Lin, S. Narang, A. Ng, S. Ozair, R. Prenger, J. Raiman, S. Satheesh, D. Seetapun, S. Sengupta, Y. Wang, Z. Wang, C. Wang, B. Xiao, D. Yogatama, J. Zhan, Z. Zhu, in Proceedings of The 33rd International Conference on Machine Learning, Vol. 48. Deep Speech 2: End-to-End Speech Recognition in English and Mandarin (PMLR, 2015), pp. 173\u201382. \n                    http:\/\/proceedings.mlr.press\/v48\/amodei16.html\n                    \n                  ."}],"container-title":["EURASIP Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-018-0146-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1186\/s13636-018-0146-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-018-0146-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,1,13]],"date-time":"2020-01-13T19:08:16Z","timestamp":1578942496000},"score":1,"resource":{"primary":{"URL":"https:\/\/asmp-eurasipjournals.springeropen.com\/articles\/10.1186\/s13636-018-0146-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,1,14]]},"references-count":31,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["146"],"URL":"https:\/\/doi.org\/10.1186\/s13636-018-0146-4","relation":{},"ISSN":["1687-4722"],"issn-type":[{"value":"1687-4722","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,1,14]]},"assertion":[{"value":"25 June 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 December 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 January 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Not applicable.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The authors declare that they have no competing interests.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}},{"value":"Springer Nature remains neutral with regard to jurisdictional claims in published maps and institutional affiliations.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Publisher\u2019s Note"}}],"article-number":"3"}}