{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T11:07:38Z","timestamp":1772708858217,"version":"3.50.1"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2015,9,10]],"date-time":"2015-09-10T00:00:00Z","timestamp":1441843200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2016,5]]},"DOI":"10.1007\/s11042-015-2935-4","type":"journal-article","created":{"date-parts":[[2015,9,11]],"date-time":"2015-09-11T00:31:53Z","timestamp":1441931513000},"page":"5109-5124","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Distant-talking accent recognition by combining GMM and DNN"],"prefix":"10.1007","volume":"75","author":[{"given":"Khomdet","family":"Phapatanaburi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Longbiao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ryota","family":"Sakagami","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaofeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ximin","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Masahiro","family":"Iwahashi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,9,10]]},"reference":[{"issue":"4","key":"2935_CR1","doi-asserted-by":"crossref","first-page":"353","DOI":"10.1016\/0167-6393(96)00024-6","volume":"18","author":"L Arslan","year":"1996","unstructured":"Arslan L, Hansen J (1996) Language accent classification in American English. Speech Commun 18(4):353\u2013367","journal-title":"Speech Commun"},{"key":"2935_CR2","doi-asserted-by":"crossref","unstructured":"Chen T, Huang C, Chang E, Wang J (2001) Automatic accent recognition using Gaussian mixture models. Proc. of IEEE Workshop on Automatic Speech Recognition and Understanding, 343\u2013346","DOI":"10.1109\/ASRU.2001.1034657"},{"key":"2935_CR3","unstructured":"Chen N, Shen W, Campbell J (2010) A linguistically-informative approach to dialect recognition using for automatic accent classification. Proc. of IEEE International Conference in acoustic speech and signal processing (ICASSP), 5014\u20135017"},{"key":"2935_CR4","doi-asserted-by":"crossref","unstructured":"Choueiter G, Zweig G, Nguyen P (2008) An empirical study of automatic dialect classification. Proc. of IEEE International Conference in acoustic speech and signal processing (ICASSP)","DOI":"10.1109\/ICASSP.2008.4518597"},{"key":"2935_CR5","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1111\/j.2517-6161.1977.tb01600.x","volume":"39","author":"AP Dempster","year":"1997","unstructured":"Dempster AP, Laird NM, Rubin DB (1997) Maximum likehood from incomplate data via EM algorithum. J R Stat Soc Ser B 39:1\u201338","journal-title":"J R Stat Soc Ser B"},{"key":"2935_CR6","doi-asserted-by":"crossref","unstructured":"Deshpande S, Chikkerur S, Govindaraju V (2005) Accent classification in speech. Proc. of IEEE Workshop on Automatic recognition advanced Technologies","DOI":"10.1109\/AUTOID.2005.10"},{"key":"2935_CR7","doi-asserted-by":"crossref","unstructured":"Fohr D, Illina I (2007) Text-independent foreign accent classification using statistical methods. Proc. of IEEE International Conference on Signal Processing and Communications (ICSPC 2007), 812\u2013815","DOI":"10.1109\/ICSPC.2007.4728443"},{"key":"2935_CR8","unstructured":"Garofolo JS, Lamel LF, Fisher WM, Fiscus JG, Pallett DS, Dahlgren NL, Zue V (1990) Timit acoustic-phonetic continuous speech corpus. National Institute of Standards and Technology Disc 1-1.1, NTIS Order No. PB91-5050651996, 91"},{"key":"2935_CR9","doi-asserted-by":"crossref","first-page":"1581","DOI":"10.1121\/1.387812","volume":"71","author":"V Gupta","year":"1982","unstructured":"Gupta V, Mermelstein P (1982) Effect of speaker accent on the performance of a speaker-independent isolated word recognition. J Acoust Soc Am 71:1581\u20131587","journal-title":"J Acoust Soc Am"},{"issue":"6","key":"2935_CR10","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton G, Deng L, Yu D, Dahl GE, Mohamed A, Jaitly N, Senior A, Vanhoucke V, Nguyen P, Sainath TN, Kingsbury B (2012) Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups. IEEE Signal Process Mag 29(6):82\u201397","journal-title":"IEEE Signal Process Mag"},{"issue":"5786","key":"2935_CR11","doi-asserted-by":"crossref","first-page":"504","DOI":"10.1126\/science.1127647","volume":"313","author":"G Hinton","year":"2006","unstructured":"Hinton G, Salakhutdinov R (2006) Reducing the dimensionality of data with neural networks. Science 313(5786):504\u2013507","journal-title":"Science"},{"issue":"3","key":"2935_CR12","doi-asserted-by":"crossref","first-page":"244","DOI":"10.1016\/j.specom.2007.09.004","volume":"50","author":"H Hirsch","year":"2008","unstructured":"Hirsch H, Finster H (2008) A new approach for the adaptation of HMMs to reverberation and background noise. Speech Comm 50(3):244\u2013263","journal-title":"Speech Comm"},{"issue":"2","key":"2935_CR13","first-page":"1","volume":"2","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Image net classification with deep convolutional neural networks. NIPS 2(2):1\u20134","journal-title":"NIPS"},{"key":"2935_CR14","doi-asserted-by":"crossref","unstructured":"Lazaridis A, Khoury E, Goldman J (2014) Swiss French regional accent recognition. The Speaker and Language Recognition Workshop June 16\u201319, in Joensuu, finland, Proceedings","DOI":"10.21437\/Odyssey.2014-17"},{"key":"2935_CR15","doi-asserted-by":"crossref","unstructured":"Minematsu N, Okabe K, Ogaki K, Hirose K (2011) measurement of objective intelligibility of japanese accented English Using ERJ (English Read by Japanese) Database. Proc Interspeech, 1481\u20131484","DOI":"10.21437\/Interspeech.2011-310"},{"key":"2935_CR16","doi-asserted-by":"crossref","first-page":"12","DOI":"10.1109\/TASL.2011.2109382","volume":"20","author":"A Mohamed","year":"2012","unstructured":"Mohamed A, Dahl GE, Hinton G (2012) Acoustic modeling using deep belief networks. IEEE Trans Audio Speech Lang Process 20:12\u201322","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"2935_CR17","unstructured":"Nakamura S, Hiyane K, Asano F, Nishiura T, Yamada T (2000) Acoustical sound database in real environments for sound scene understanding and hands-free speech recognition. Proc LREC2000, 965\u2013968"},{"key":"2935_CR18","unstructured":"Nishiura T et al (2008) Evaluation framework for distant-talking speech recognition under reverberant environments. Proc Interspeech, 968\u2013971"},{"issue":"1\u20132","key":"2935_CR19","doi-asserted-by":"crossref","first-page":"91","DOI":"10.1016\/0167-6393(95)00009-D","volume":"17","author":"DA Reynolds","year":"1995","unstructured":"Reynolds DA (1995) Speaker identification and verification using Gaussian mixture speaker models. Speech Comm 17(1\u20132):91\u2013108","journal-title":"Speech Comm"},{"issue":"7","key":"2935_CR20","first-page":"1676","volume":"18","author":"A Sehr","year":"2010","unstructured":"Sehr A, Maas R, Kellermann W (2010) Reverberation model-based decoding in the logmelspec domain for robust distant-talking speech recognition. IEEE Trans ASLP 18(7):1676\u20131691","journal-title":"IEEE Trans ASLP"},{"key":"2935_CR21","unstructured":"Tsai MY, Lee LS (2003) Pronunciation variation analysis based on acoustic and phonemic distancemeasures with application examples on Mandarin Chinese. Proc. of IEEE Workshop Autom. Speech Recogn. Understand 117\u2013122"},{"key":"2935_CR22","doi-asserted-by":"crossref","unstructured":"Ueda Y, Wang L, Kai A, Xiao X, Chng E, Li H (2015) Single-channel dereverberation for distant-talking speech recognition by combining denoising autoencoder and temporal structure normalization. J Signal Process Syst","DOI":"10.1109\/ISCSLP.2014.6936613"},{"key":"2935_CR23","doi-asserted-by":"crossref","unstructured":"Wang L, Kitaoka N, Nakagawa S (2006) Robust distant speech recognition by combining multiple microphone-array processing with position-dependent CMN. EURASIP J Appl Signal Process. 95491: 1\u201311","DOI":"10.1155\/ASP\/2006\/95491"},{"issue":"6","key":"2935_CR24","doi-asserted-by":"crossref","first-page":"501","DOI":"10.1016\/j.specom.2007.04.004","volume":"49","author":"L Wang","year":"2007","unstructured":"Wang L, Kitaoka N, Nakagawa S (2007) Robust distant speaker recognition based on position-dependent CMN by combining speaker-specific GMM with speaker-adapted HMM. Speech Comm 49(6):501\u2013513","journal-title":"Speech Comm"},{"key":"2935_CR25","first-page":"1","volume":"12","author":"L Wang","year":"2012","unstructured":"Wang L, Odani K, Kai A (2012) Dereverberation and denoising based on generalized spectral subtraction by nutil-channel LMS algorithm using a small-scale microphone array. EURASIP J Adv Signal Process 12:1\u201311","journal-title":"EURASIP J Adv Signal Process"},{"issue":"3","key":"2935_CR26","first-page":"774","volume":"14","author":"M Wu","year":"2006","unstructured":"Wu M, Wang D (2006) A two-stage algorithm for one-microphone reverberant speech enhancement. IEEE Trans ASLP 14(3):774\u2013784","journal-title":"IEEE Trans ASLP"},{"key":"2935_CR27","doi-asserted-by":"crossref","unstructured":"Yamada T, Wang L, Kai A (2013) Improvement of distant-talking speaker identification using bottleneck features of DNN. Proc Interspeech, 3661\u20133664","DOI":"10.21437\/Interspeech.2013-686"},{"issue":"1","key":"2935_CR28","doi-asserted-by":"crossref","first-page":"65","DOI":"10.1016\/j.csl.2014.11.008","volume":"31","author":"T Yoshioka","year":"2015","unstructured":"Yoshioka T, Gales MJF (2015) Environmentally robust ASR front-end for deep neural network acoustic models. Comput Speech Lang 31(1):65\u201386","journal-title":"Comput Speech Lang"},{"issue":"6","key":"2935_CR29","doi-asserted-by":"crossref","first-page":"114","DOI":"10.1109\/MSP.2012.2205029","volume":"29","author":"T Yoshioka","year":"2012","unstructured":"Yoshioka T, Sehr A, Delcroix M, Kinoshita K, Maas R, Nakatani T, Kellermann W (2012) Making machines understand us in reverberant rooms: robustness against reverberation for automatic speech recognition. IEEE Signal Process Mag 29(6):114\u2013126","journal-title":"IEEE Signal Process Mag"},{"key":"2935_CR30","unstructured":"Young S, Kershow D, Odell J, Ollason D, Valtchev V, Woodland P (2000) The HTK book. Cambridge University (for HTK version 3.0)"},{"key":"2935_CR31","doi-asserted-by":"crossref","unstructured":"Zhang Z, Wang L, Kai A, Odani K, Li W, Iwahashi M (2015) Deep neural network-based bottleneck feature and denoising autoencoder-based dereverberation for distant-talking speaker identification. EURASIP J Audio Speech Music Process 12","DOI":"10.1186\/s13636-015-0056-7"},{"key":"2935_CR32","first-page":"1","volume":"15","author":"Z Zhang","year":"2014","unstructured":"Zhang Z, Wang L, Kai A (2014) Distant-talking speaker identification by generalized spectral subtraction-based dereverberation and its efficient computation. EURASIP J Audio Speech Music Process 15:1\u201312","journal-title":"EURASIP J Audio Speech Music Process"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-015-2935-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-015-2935-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-015-2935-4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,11]],"date-time":"2024-06-11T01:45:49Z","timestamp":1718070349000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-015-2935-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,9,10]]},"references-count":32,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2016,5]]}},"alternative-id":["2935"],"URL":"https:\/\/doi.org\/10.1007\/s11042-015-2935-4","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2015,9,10]]}}}