{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:02:04Z","timestamp":1776880924030,"version":"3.51.2"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T00:00:00Z","timestamp":1673222400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T00:00:00Z","timestamp":1673222400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,1,9]]},"DOI":"10.1109\/slt54892.2023.10023187","type":"proceedings-article","created":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T13:54:03Z","timestamp":1674827643000},"page":"221-228","source":"Crossref","is-referenced-by-count":38,"title":["Towards End-to-End Unsupervised Speech Recognition"],"prefix":"10.1109","author":[{"given":"Alexander H.","family":"Liu","sequence":"first","affiliation":[{"name":"MIT CSAIL"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei-Ning","family":"Hsu","sequence":"additional","affiliation":[{"name":"Meta AI"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michael","family":"Auli","sequence":"additional","affiliation":[{"name":"Meta AI"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexei","family":"Baevski","sequence":"additional","affiliation":[{"name":"Meta AI"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","article-title":"The DARPA TIMIT Acoustic-Phonetic Continuous Speech Corpus CDROM","author":"garofolo","year":"1993","journal-title":"Linguistic Data Consortium"},{"key":"ref35","article-title":"Unsupervised cross-lingual representation learning for speech recognition","volume":"abs 2006 13979","author":"conneau","year":"2020","journal-title":"ArXiv"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-4009"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2019.06.005"},{"key":"ref15","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","author":"baevski","year":"0","journal-title":"Proc of NeurIPS"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2017.06.008"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"ref36","article-title":"Pykaldi: A python wrap-per for kaldi","author":"can","year":"0","journal-title":"Proc of ICASSP"},{"key":"ref31","article-title":"Unsupervised pre-training transfers well across languages","author":"riviere","year":"0","journal-title":"Proc of ICASSP"},{"key":"ref30","author":"ito","year":"2017","journal-title":"The LJ speech dataset"},{"key":"ref11","article-title":"Completely unsupervised speech recognition by a generative adversarial network harmonized with iteratively refined hidden markov models","author":"chen","year":"0","journal-title":"Proc of Interspeech"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053571"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2398"},{"key":"ref32","first-page":"5410","article-title":"Almost unsupervised text to speech and automatic speech recognition","author":"ren","year":"2019","journal-title":"International Con-ference on Machine Learning PMLR"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2059"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"ref17","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"0","journal-title":"Int Conference on Machine Learning"},{"key":"ref39","first-page":"269","article-title":"Finite-state transducers in language and speech processing","volume":"23","author":"mohri","year":"1997","journal-title":"Computational Linguistics"},{"key":"ref16","article-title":"Improved training of wasserstein gans","author":"gulrajani","year":"0","journal-title":"Proc of NIPS"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00146"},{"key":"ref19","article-title":"Fully convolutional speech recognition","volume":"abs 1812 6864","author":"zeghidour","year":"2018","journal-title":"ar Xiv"},{"key":"ref18","article-title":"Deep speech 2: End-to-end speech recog-nition in english and mandarin","author":"amodei","year":"0","journal-title":"Proc of ICML"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1470"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1800"},{"key":"ref26","article-title":"MIs: A large-scale multilingual dataset for speech research","author":"pratap","year":"0","journal-title":"Proc of Interspeech"},{"key":"ref25","article-title":"Pushing the limits of semi -supervised learning for automatic speech recognition","author":"zhang","year":"0","journal-title":"Proc of NeurIPS SAS Workshop"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461704"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414460"},{"key":"ref22","article-title":"End-to-end ASR: from Supervised to Semi-Supervised Learning with Modern Architectures","author":"synnaeve","year":"0","journal-title":"Proc of ICML workshop on Self-supervision in Audio and Speech (SAS)"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref28","article-title":"Common voice: A massively-multilingual speech cor-pus","author":"ardila","year":"0","journal-title":"Proc of LREC"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref29","author":"park","year":"2019","journal-title":"G2pe"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2922832"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462002"},{"key":"ref9","article-title":"Unsupervised cross-modal alignment of speech and text embedding spaces","author":"chung","year":"0","journal-title":"Proc of NIPS"},{"key":"ref4","article-title":"Completely unsupervised phoneme recog-nition by adversarially learning mapping relationships from audio embeddings","author":"liu","year":"0","journal-title":"Proc of Interspeech"},{"key":"ref3","author":"paul","year":"2016","journal-title":"Ethnologue Languages of the world nineteenth edition"},{"key":"ref6","article-title":"Unsupervised speech recognition","author":"baevski","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref5","article-title":"Unsupervised speech recognition via segmental empirical output distribution matching","author":"yeh","year":"0","journal-title":"Proc of ICLR"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1006\/csla.2001.0184"}],"event":{"name":"2022 IEEE Spoken Language Technology Workshop (SLT)","location":"Doha, Qatar","start":{"date-parts":[[2023,1,9]]},"end":{"date-parts":[[2023,1,12]]}},"container-title":["2022 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10022052\/10022330\/10023187.pdf?arnumber=10023187","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,20]],"date-time":"2023-02-20T17:08:20Z","timestamp":1676912900000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10023187\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,1,9]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/slt54892.2023.10023187","relation":{},"subject":[],"published":{"date-parts":[[2023,1,9]]}}}