{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,16]],"date-time":"2025-11-16T14:13:19Z","timestamp":1763302399794,"version":"3.45.0"},"publisher-location":"Singapore","reference-count":26,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819533510","type":"print"},{"value":"9789819533527","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,17]],"date-time":"2025-11-17T00:00:00Z","timestamp":1763337600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,17]],"date-time":"2025-11-17T00:00:00Z","timestamp":1763337600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-3352-7_18","type":"book-chapter","created":{"date-parts":[[2025,11,16]],"date-time":"2025-11-16T14:08:54Z","timestamp":1763302134000},"page":"221-233","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mongolian Speech Recognition Based on\u00a0Semi-supervised Learning and\u00a0Syllable Subword Modeling Units"],"prefix":"10.1007","author":[{"given":"Yuan","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yonghe","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"ZhenJie","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feilong","family":"Bao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,17]]},"reference":[{"key":"18_CR1","first-page":"27826","volume":"34","author":"A Baevski","year":"2021","unstructured":"Baevski, A., Hsu, W.N., Conneau, A., Auli, M.: Unsupervised speech recognition. ANIPS 34, 27826\u201327839 (2021)","journal-title":"ANIPS"},{"key":"18_CR2","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., Auli, M.: wav2vec 2.0: a framework for self-supervised learning of speech representations. NeurIPS 33, 12449\u201312460 (2020)"},{"key":"18_CR3","doi-asserted-by":"crossref","unstructured":"Bao, F., Gao, G., Yan, X., Wang, W.: Segmentation-based Mongolian LVCSR approach. In: ICASSP, pp. 8136\u20138139 (2013)","DOI":"10.1109\/ICASSP.2013.6639250"},{"key":"18_CR4","doi-asserted-by":"crossref","unstructured":"Chou, J.C., Chien, C.M., Hsu, W.N.: Toward joint language modeling for speech units and text. arXiv preprint arXiv:2310.08715 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.438"},{"key":"18_CR5","doi-asserted-by":"crossref","unstructured":"Dua, M., Akanksha, Dua, S.: Noise robust automatic speech recognition: review and analysis. Int. J. Speech Technol. 26(2), 475\u2013519 (2023)","DOI":"10.1007\/s10772-023-10033-0"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Gulati, A., Qin, J., Chiu, C.-C., et al.: Conformer: convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"18_CR7","doi-asserted-by":"crossref","unstructured":"Janhunen, J.: The Mongolic Languages. Routledge (2006)","DOI":"10.4324\/9780203987919"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Liang, K., Liu, B., Hu, Y., et al.: MnTTS2: an open-source multi-speaker Mongolian text-to-speech synthesis dataset. In: National Conference on Man-Machine Speech Communication, pp. 318\u2013329. Springer (2022)","DOI":"10.1007\/978-981-99-2401-1_28"},{"key":"18_CR9","doi-asserted-by":"crossref","unstructured":"Liu, A.H., Hsu, W.N., Auli, M., Baevski, A.: Towards end-to-end unsupervised speech recognition. In: SLT, pp. 221\u2013228 (2023)","DOI":"10.1109\/SLT54892.2023.10023187"},{"key":"18_CR10","unstructured":"Liu, R., Bao, F., Gao, G., Zhang, H., Wang, Y.: A LSTM approach with sub-word embeddings for Mongolian phrase break prediction. In: ICCL, pp. 2448\u20132455 (2018)"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Liu, R., Bao, F., Gao, G., Zhang, H., Wang, Y.: Phonologically aware BiLSTM model for Mongolian phrase break prediction with attention mechanism. In: PRICAI, pp. 217\u2013231. Springer (2018)","DOI":"10.1007\/978-3-319-97304-3_17"},{"key":"18_CR12","first-page":"1075","volume":"32","author":"R Liu","year":"2024","unstructured":"Liu, R., Hu, Y., Zuo, H., Luo, Z., Wang, L., Gao, G.: Text-to-speech for low-resource agglutinative language with morphology-aware language model pre-training. IEEE\/ACM TASLP 32, 1075\u20131087 (2024)","journal-title":"IEEE\/ACM TASLP"},{"key":"18_CR13","first-page":"274","volume":"29","author":"R Liu","year":"2020","unstructured":"Liu, R., Sisman, B., Bao, F., Yang, G., et al.: Exploiting morphological and phonological features to improve prosodic phrasing for Mongolian speech synthesis. TASLP 29, 274\u2013285 (2020)","journal-title":"TASLP"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Polyak, A., Adi, Y., Jade, C., et al.: Speech resynthesis from discrete disentangled self-supervised representations. arXiv preprint arXiv:2104.00355 (2021)","DOI":"10.21437\/Interspeech.2021-475"},{"issue":"1","key":"18_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s13636-021-00199-3","volume":"2021","author":"K Radzikowski","year":"2021","unstructured":"Radzikowski, K., Wang, L., Yoshie, O., Nowak, R.: Accent modification for speech recognition of non-native speakers using neural style transfer. EURASIP J. Audio Speech Music Process. 2021(1), 1\u201310 (2021). https:\/\/doi.org\/10.1186\/s13636-021-00199-3","journal-title":"EURASIP J. Audio Speech Music Process."},{"key":"18_CR16","doi-asserted-by":"crossref","unstructured":"Singh, K., Manohar, V., Xiao, A., Edunov, S., et al.: Large scale weakly and semi-supervised learning for low-resource video ASR. arXiv preprint arXiv:2005.07850 (2020)","DOI":"10.21437\/Interspeech.2020-1917"},{"key":"18_CR17","unstructured":"Sutskever, I., Vinyals, O., Le, Q.V.: Sequence to sequence learning with neural networks. NeurIPS 27 (2014)"},{"key":"18_CR18","doi-asserted-by":"crossref","unstructured":"Tian, Z., Yi, J., Tao, J., et al.: Spike-triggered non-autoregressive transformer for end-to-end speech recognition. arXiv preprint arXiv:2005.07903 (2020)","DOI":"10.21437\/Interspeech.2020-2086"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Wang, Y., Bao, F., Zhang, H., Gao, G.: Joint alignment learning-attention based model for grapheme-to-phoneme conversion. In: ICASSP, pp. 7788\u20137792 (2021)","DOI":"10.1109\/ICASSP39728.2021.9413679"},{"key":"18_CR20","unstructured":"Wang, Y., Wang, W.: Khalkha Mongolian dialect speech dataset (version 1), March 2023. Dataset"},{"key":"18_CR21","doi-asserted-by":"crossref","unstructured":"Wu, F., et al.: Wav2Seq: pre-training speech-to-text encoder-decoder models using pseudo languages. In: ICASSP, pp.\u00a01\u20135 (2022)","DOI":"10.1109\/ICASSP49357.2023.10096988"},{"key":"18_CR22","doi-asserted-by":"crossref","unstructured":"Wu, Y., Wang, Y., Zhang, H., Bao, F., Gao, G.: MNASR: a free speech corpus for Mongolian speech recognition and accompanied baselines. In: O-COCOSDA, pp.\u00a01\u20136 (2022)","DOI":"10.1109\/O-COCOSDA202257103.2022.9997919"},{"issue":"10","key":"18_CR23","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3617830","volume":"22","author":"W Yonghe","year":"2023","unstructured":"Yonghe, W., Bao, F., Gao, G.: A comparative study on selecting acoustic modeling units for WFST-based Mongolian speech recognition. TALLIP 22(10), 1\u201320 (2023)","journal-title":"TALLIP"},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Yusuyin, S., Ma, T., Huang, H., Zhao, W., Ou, Z.: Whistle: data-efficient multilingual and crosslingual speech recognition via weakly phonetic supervision. arXiv preprint arXiv:2406.02166 (2024)","DOI":"10.1109\/TASLPRO.2025.3550683"},{"key":"18_CR25","doi-asserted-by":"crossref","unstructured":"Zhang, B., Wu, D., Peng, Z., et al.: WeNet 2.0: more productive end-to-end speech recognition toolkit. arXiv preprint arXiv:2203.15455 (2022)","DOI":"10.21437\/Interspeech.2022-483"},{"key":"18_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Park, D.S., Han, W., et al.: BigSSL: exploring the frontier of large-scale semi-supervised learning for automatic speech recognition. JSTSP 16(6), 1519\u20131532 (2022)","DOI":"10.1109\/JSTSP.2022.3182537"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Chinese Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-3352-7_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,16]],"date-time":"2025-11-16T14:08:58Z","timestamp":1763302138000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-3352-7_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,17]]},"ISBN":["9789819533510","9789819533527"],"references-count":26,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-3352-7_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,17]]},"assertion":[{"value":"17 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLPCC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"CCF International Conference on Natural Language Processing and Chinese Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nlpcc2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/tcci.ccf.org.cn\/conference\/2025\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}