{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T10:39:07Z","timestamp":1730198347636,"version":"3.28.0"},"reference-count":19,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,10,31]],"date-time":"2023-10-31T00:00:00Z","timestamp":1698710400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,10,31]],"date-time":"2023-10-31T00:00:00Z","timestamp":1698710400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,10,31]]},"DOI":"10.1109\/apsipaasc58517.2023.10317133","type":"proceedings-article","created":{"date-parts":[[2023,11,20]],"date-time":"2023-11-20T19:07:46Z","timestamp":1700507266000},"page":"978-982","source":"Crossref","is-referenced-by-count":0,"title":["Progressive Multi-scale Self-supervised Learning for Speech Recognition"],"prefix":"10.1109","author":[{"given":"Genshun","family":"Wan","sequence":"first","affiliation":[{"name":"University of Science and Technology of China,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hang","family":"Chen","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tan","family":"Liu","sequence":"additional","affiliation":[{"name":"iFLYTEK Co. Ltd.,iFLYTEK Research,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenxi","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jia","family":"Pan","sequence":"additional","affiliation":[{"name":"iFLYTEK Co. Ltd.,iFLYTEK Research,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongfu","family":"Ye","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"173","article-title":"Deep speech 2: End-to-end speech recognition in english and mandarin","volume-title":"International conference on machine learning","author":"Amodei"},{"article-title":"Speechstew: Simply mix all available speech recognition data to train one large neural network","year":"2021","author":"Chan","key":"ref2"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054224"},{"article-title":"Vq-wav2vec: Self-supervised learning of discrete speech representations","year":"2019","author":"Baevski","key":"ref4"},{"key":"ref5","first-page":"12 449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"Baevski","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3095662"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2605"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/icassp43922.2022.9747022"},{"article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","year":"2018","author":"Devlin","key":"ref11"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1800"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-740"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P14-1062"},{"article-title":"Hierarchical multiscale recurrent neural networks","year":"2016","author":"Chung","key":"ref16"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref19","article-title":"Wav2letter++: The fastest open-source speech recognition system","volume":"abs\/1812.07625","author":"Pratap","year":"2018","journal-title":"CoRR"}],"event":{"name":"2023 Asia Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC)","start":{"date-parts":[[2023,10,31]]},"location":"Taipei, Taiwan","end":{"date-parts":[[2023,11,3]]}},"container-title":["2023 Asia Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10317071\/10317095\/10317133.pdf?arnumber=10317133","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,2]],"date-time":"2024-03-02T18:42:53Z","timestamp":1709404973000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10317133\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,31]]},"references-count":19,"URL":"https:\/\/doi.org\/10.1109\/apsipaasc58517.2023.10317133","relation":{},"subject":[],"published":{"date-parts":[[2023,10,31]]}}}