{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,27]],"date-time":"2025-07-27T07:32:16Z","timestamp":1753601536869,"version":"3.37.3"},"reference-count":53,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2016,9,1]],"date-time":"2016-09-01T00:00:00Z","timestamp":1472688000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2016,9]]},"DOI":"10.1109\/taslp.2016.2582260","type":"journal-article","created":{"date-parts":[[2016,6,20]],"date-time":"2016-06-20T20:30:24Z","timestamp":1466454624000},"page":"1665-1676","source":"Crossref","is-referenced-by-count":4,"title":["On the Use of Acoustic Unit Discovery for Language Recognition"],"prefix":"10.1109","volume":"24","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2564-5214","authenticated-orcid":false,"given":"Stephen H.","family":"Shum","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David F.","family":"Harwath","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Najim","family":"Dehak","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"James R.","family":"Glass","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","first-page":"1471","article-title":"Within-class covariance normalization for SVM-based speaker recognition","author":"hatch","year":"0","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref38","first-page":"209","article-title":"The MITLL NIST LRE 2011 Language recognition system","author":"singer","year":"0","journal-title":"Proc Odyssey"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1988.196610"},{"key":"ref32","first-page":"2715","article-title":"Analyzing hogwild parallel Gaussian Gibbs sampling","author":"johnson","year":"0","journal-title":"Proc Adv Neural Inform Process Syst"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2011.29"},{"article-title":"TIMIT acoustic-phonetic continuous speech corpus LDC93S1","year":"1993","author":"garofolo","key":"ref30"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2064307"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853583"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638949"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.3115\/1075812.1075885"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2006.1660179"},{"year":"2011","key":"ref27","article-title":"The 2011 NIST language recognition evaluation plan (LRE11)"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/S0885-2308(03)00006-8"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1049\/el.2013.1721"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref20","article-title":"Data-driven speech segmentation for language identification and speaker verification","author":"petrovska-delacretaz","year":"0","journal-title":"Proc Non Linear Speech Process"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.1996.481450"},{"article-title":"Unified data-driven approach for audio indexing, retrieval, and recognition","year":"2013","author":"khemiri","key":"ref21"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2002.5743828"},{"key":"ref23","first-page":"2833","article-title":"Phonotactic language recognition using high quality phoneme recognition","author":"matejka","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref26","article-title":"The kaldi speech recognition toolkit","author":"povey","year":"0","journal-title":"Proc IEEE Workshop Autom Speech Recog and Understanding"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.876860"},{"key":"ref50","article-title":"PCA-based feature extraction for phonotactic language recognition","author":"mikolov","year":"0","journal-title":"Proc Odyssey"},{"key":"ref51","first-page":"2913","article-title":"iVector approach to phonotactic language recognition","author":"soufifar","year":"0","journal-title":"Proc INTERSPEECH"},{"article-title":"Overcoming resource limitations in the processing of unlimited speech: Applications to speaker and langauge recognition","year":"2016","author":"shum","key":"ref53"},{"key":"ref52","first-page":"74","article-title":"Regularized subspace n-gram model for phonotactic ivector extraction","author":"soufifar","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1992.225858"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2013.05.002"},{"year":"2015","key":"ref40","article-title":"The 2015 NIST language recognition evaluation plan (LRE15)"},{"key":"ref12","first-page":"857","article-title":"Language recognition via i-vectors and dimensionality reduction","author":"dehak","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref13","first-page":"40","article-title":"A nonparametric Bayesian approach to acoustic model discovery","author":"lee","year":"0","journal-title":"Proc Annual Meeting of the Assoc Computational Linguistics"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639245"},{"key":"ref15","first-page":"1676","article-title":"Towards spoken term discovery at scale with zero resources","author":"jansen","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref16","first-page":"1693","article-title":"Towards unsupervised training of speaker independent acoustic models","author":"jansen","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/89.985546"},{"key":"ref18","first-page":"2265","article-title":"Unsupervised learning of acoustic unit descriptors for audio content representation and classification","author":"chaudhuri","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-60087-6_32"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853887"},{"key":"ref3","first-page":"2150","article-title":"Spoken language recognition based on senone posteriors","author":"ferrer","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2015.2420092"},{"key":"ref5","first-page":"299","article-title":"Neural network bottleneck features for language identification","author":"matejka","year":"0","journal-title":"Proc Odyssey"},{"key":"ref8","first-page":"389","article-title":"Multilingual bottleneck features for language recognition","author":"fer","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref7","first-page":"1146","article-title":"A unified deep neural network for speaker and language recognition","author":"richardson","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472619"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2496226"},{"key":"ref46","first-page":"1","article-title":"Language ID-based training of multilingual stacked bottleneck features","author":"zhang","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref45","first-page":"7654","article-title":"Adaptation of multilingual stacked bottleneck neural network structure for new language","author":"grezl","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7179087"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639241"},{"key":"ref42","first-page":"89","article-title":"Approaches to language identification using gaussian mixture models and shifted delta cepstral features","author":"torres-carrasquillo","year":"0","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref41","first-page":"196","article-title":"The MITLL NIST LRE 2015 language recognition system","author":"torres-carrasquillo","year":"0","journal-title":"Proc Odyssey"},{"key":"ref44","first-page":"3169","article-title":"The zero resource speech challenge 2015","author":"versteegh","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref43","article-title":"Language identification using shifted delta cepstrum","author":"bielefeld","year":"0","journal-title":"Proc 14th Annual Speech Research Symposium"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/7497021\/07494635.pdf?arnumber=7494635","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T16:11:45Z","timestamp":1642003905000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7494635\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,9]]},"references-count":53,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2016.2582260","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"type":"print","value":"2329-9290"},{"type":"electronic","value":"2329-9304"}],"subject":[],"published":{"date-parts":[[2016,9]]}}}