{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:48:42Z","timestamp":1776887322206,"version":"3.51.2"},"publisher-location":"Singapore","reference-count":37,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819783663","type":"print"},{"value":"9789819783670","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,29]],"date-time":"2024-11-29T00:00:00Z","timestamp":1732838400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,29]],"date-time":"2024-11-29T00:00:00Z","timestamp":1732838400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8367-0_15","type":"book-chapter","created":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T11:56:27Z","timestamp":1732794987000},"page":"243-258","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["DialectMoE: An End-to-End Multi-dialect Speech Recognition Model with\u00a0Mixture-of-Experts"],"prefix":"10.1007","author":[{"given":"Jie","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengxiang","family":"Gao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengtao","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ling","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenjun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,29]]},"reference":[{"key":"15_CR1","doi-asserted-by":"publisher","first-page":"131858","DOI":"10.1109\/ACCESS.2021.3112535","volume":"9","author":"A S","year":"2021","unstructured":"S, A., et al.: Automatic speech recognition: systematic literature review. IEEE Access 9, 131858\u2013131876 (2021)","journal-title":"IEEE Access"},{"key":"15_CR2","doi-asserted-by":"publisher","first-page":"975","DOI":"10.1007\/s10579-020-09505-5","volume":"54","author":"E Alsharhan","year":"2020","unstructured":"Alsharhan, E., Ramsay, A.: Investigating the effects of gender, dialect, and training size on the performance of Arabic speech recognition. Lang. Resour. Eval. 54, 975\u2013998 (2020)","journal-title":"Lang. Resour. Eval."},{"key":"15_CR3","doi-asserted-by":"crossref","unstructured":"Bu, H., Du, J., Na, X., Wu, B., Zheng, H.: AISHELL-1: an open-source mandarin speech corpus and a speech recognition baseline. In: 2017 20th Conference of the Oriental Chapter of the International Coordinating Committee on Speech Databases and Speech I\/O Systems and Assessment (O-COCOSDA), pp.\u00a01\u20135. IEEE (2017)","DOI":"10.1109\/ICSDA.2017.8384449"},{"issue":"10","key":"15_CR4","doi-asserted-by":"publisher","first-page":"1429","DOI":"10.3390\/e24101429","volume":"24","author":"Z Dan","year":"2022","unstructured":"Dan, Z., Zhao, Y., Bi, X., Wu, L., Ji, Q.: Multi-task transformer with adaptive cross-entropy loss for multi-dialect speech recognition. Entropy 24(10), 1429 (2022)","journal-title":"Entropy"},{"key":"15_CR5","unstructured":"Du, N., et\u00a0al.: GLaM: efficient scaling of language models with mixture-of-experts. In: International Conference on Machine Learning, pp. 5547\u20135569. PMLR (2022)"},{"key":"15_CR6","doi-asserted-by":"crossref","unstructured":"Elfeky, M., Bastani, M., Velez, X., Moreno, P., Waters, A.: Towards acoustic model unification across dialects. In: 2016 IEEE Spoken Language Technology Workshop (SLT), pp. 624\u2013628. IEEE (2016)","DOI":"10.1109\/SLT.2016.7846328"},{"key":"15_CR7","first-page":"28441","volume":"35","author":"Z Fan","year":"2022","unstructured":"Fan, Z., et al.: M$$^3$$ViT: mixture-of-experts vision transformer for efficient multi-task learning with model-accelerator co-design. Adv. Neural. Inf. Process. Syst. 35, 28441\u201328457 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"1","key":"15_CR8","first-page":"5232","volume":"23","author":"W Fedus","year":"2022","unstructured":"Fedus, W., Zoph, B., Shazeer, N.: Switch transformers: scaling to trillion parameter models with simple and efficient sparsity. J. Mach. Learn. Res. 23(1), 5232\u20135270 (2022)","journal-title":"J. Mach. Learn. Res."},{"key":"15_CR9","unstructured":"Gotmare, A., Keskar, N.S., Xiong, C., Socher, R.: A closer look at deep learning heuristics: learning rate restarts, warmup and distillation. arXiv preprint arXiv:1810.13243 (2018)"},{"key":"15_CR10","doi-asserted-by":"crossref","unstructured":"Gulati, A., et\u00a0al.: Conformer: convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"15_CR11","unstructured":"Hinsvark, A., et\u00a0al.: Accented speech recognition: a survey. arXiv preprint arXiv:2104.10747 (2021)"},{"key":"15_CR12","unstructured":"Ho, D.A.: Chinese dialects. In: The Oxford Handbook of Chinese Linguistics, pp. 149\u2013159 (2015)"},{"key":"15_CR13","doi-asserted-by":"crossref","unstructured":"Hori, T., Watanabe, S., Hershey, J.R.: Joint CTC\/attention decoding for end-to-end speech recognition. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics, vol. 1: Long Papers, pp. 518\u2013529 (2017)","DOI":"10.18653\/v1\/P17-1048"},{"key":"15_CR14","doi-asserted-by":"crossref","unstructured":"Humphries, J.J., Woodland, P.C., Pearce, D.: Using accent-specific pronunciation modelling for robust speech recognition. In: Proceeding of Fourth International Conference on Spoken Language Processing, ICSLP 1996, vol.\u00a04, pp. 2324\u20132327. IEEE (1996)","DOI":"10.21437\/ICSLP.1996-588"},{"issue":"1","key":"15_CR15","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1162\/neco.1991.3.1.79","volume":"3","author":"RA Jacobs","year":"1991","unstructured":"Jacobs, R.A., Jordan, M.I., Nowlan, S.J., Hinton, G.E.: Adaptive mixtures of local experts. Neural Comput. 3(1), 79\u201387 (1991)","journal-title":"Neural Comput."},{"key":"15_CR16","doi-asserted-by":"crossref","unstructured":"Jiang, R.: Chinese dialect recognition based on transfer learning. In: INTERSPEECH 2023 (2023). https:\/\/api.semanticscholar.org\/CorpusID:260922138","DOI":"10.21437\/Interspeech.2023-1554"},{"key":"15_CR17","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"15_CR18","doi-asserted-by":"crossref","unstructured":"Kwon, Y., Chung, S.W.: MoLE: mixture of language experts for multi-lingual automatic speech recognition. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10096227"},{"key":"15_CR19","doi-asserted-by":"crossref","unstructured":"Li, B., et al.: Multi-dialect speech recognition with a single sequence-to-sequence model. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4749\u20134753. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461886"},{"key":"15_CR20","doi-asserted-by":"crossref","unstructured":"Li, S., Lu, X., Ding, C., Shen, P., Kawahara, T., Kawai, H.: Investigating radical-based end-to-end speech recognition systems for Chinese dialects and Japanese. In: INTERSPEECH, pp. 2200\u20132204 (2019)","DOI":"10.21437\/Interspeech.2019-2104"},{"key":"15_CR21","doi-asserted-by":"publisher","first-page":"9411","DOI":"10.1007\/s11042-020-10073-7","volume":"80","author":"M Malik","year":"2021","unstructured":"Malik, M., Malik, M.K., Mehmood, K., Makhdoom, I.: Automatic speech recognition: a survey. Multimedia Tools Appl. 80, 9411\u20139457 (2021)","journal-title":"Multimedia Tools Appl."},{"key":"15_CR22","doi-asserted-by":"crossref","unstructured":"Park, D.S., et al.: SpecAugment: a simple data augmentation method for automatic speech recognition. arXiv preprint arXiv:1904.08779 (2019)","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"15_CR23","unstructured":"Peng, Y., Dalmia, S., Lane, I., Watanabe, S.: BranchFormer: parallel MLP-attention architectures to capture local and global context for speech recognition and understanding. In: International Conference on Machine Learning, pp. 17627\u201317643. PMLR (2022)"},{"key":"15_CR24","doi-asserted-by":"crossref","unstructured":"Ren, Z., Yang, G., Xu, S.: Two-stage training for Chinese dialect recognition. arXiv preprint arXiv:1908.02284 (2019)","DOI":"10.21437\/Interspeech.2019-1522"},{"key":"15_CR25","first-page":"8583","volume":"34","author":"C Riquelme","year":"2021","unstructured":"Riquelme, C., et al.: Scaling vision with sparse mixture of experts. Adv. Neural. Inf. Process. Syst. 34, 8583\u20138595 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"15_CR26","unstructured":"Sehoon, K., et\u00a0al.: SqueezeFormer: an efficient transformer for automatic speech recognition (2023)"},{"key":"15_CR27","unstructured":"Shazeer, N., et al.: Outrageously large neural networks: the sparsely-gated mixture-of-experts layer. arXiv preprint arXiv:1701.06538 (2017)"},{"key":"15_CR28","doi-asserted-by":"crossref","unstructured":"Singh, Y., Pillay, A., Jembere, E.: Features of speech audio for accent recognition. In: 2020 International Conference on Artificial Intelligence, Big Data, Computing and Data Communication Systems (icABCD), pp.\u00a01\u20136. IEEE (2020)","DOI":"10.1109\/icABCD49160.2020.9183893"},{"key":"15_CR29","unstructured":"Sproat, R., et\u00a0al.: Dialectal Chinese speech recognition. In: CLSP Summer Workshop (2004)"},{"key":"15_CR30","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132826 (2016)","DOI":"10.1109\/CVPR.2016.308"},{"issue":"8","key":"15_CR31","doi-asserted-by":"publisher","first-page":"1018","DOI":"10.3390\/sym11081018","volume":"11","author":"D Wang","year":"2019","unstructured":"Wang, D., Wang, X., Lv, S.: An overview of end-to-end automatic speech recognition. Symmetry 11(8), 1018 (2019)","journal-title":"Symmetry"},{"key":"15_CR32","doi-asserted-by":"crossref","unstructured":"Wang, X., Long, Y., Li, Y., Wei, H.: Multi-pass training and cross-information fusion for low-resource end-to-end accented speech recognition. arXiv preprint arXiv:2306.11309 (2023)","DOI":"10.21437\/Interspeech.2023-142"},{"key":"15_CR33","doi-asserted-by":"crossref","unstructured":"You, Z., Feng, S., Su, D., Yu, D.: SpeechMoE: scaling to large acoustic models with dynamic routing mixture of experts. arXiv preprint arXiv:2105.03036 (2021)","DOI":"10.21437\/Interspeech.2021-478"},{"key":"15_CR34","doi-asserted-by":"crossref","unstructured":"You, Z., Feng, S., Su, D., Yu, D.: SpeechMoE2: mixture-of-experts model with improved routing. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7217\u20137221. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9747065"},{"key":"15_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, B., et al.: WeNet 2.0: more productive end-to-end speech recognition toolkit. arXiv preprint arXiv:2203.15455 (2022)","DOI":"10.21437\/Interspeech.2022-483"},{"key":"15_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, F., Xie, X., Quan, X.: Chinese dialect speech recognition based on end-to-end machine learning. In: 2022 International Conference on Machine Learning, Control, and Robotics (MLCR), pp. 14\u201318. IEEE (2022)","DOI":"10.1109\/MLCR57210.2022.00012"},{"key":"15_CR37","doi-asserted-by":"crossref","unstructured":"Zilvan, V., Heryana, A., Yuliani, A.R., Krisnandi, D., Yuwana, R.S., Pardede, H.F.: Front-end based robust speech recognition methods: a review. In: Proceedings of the 2021 International Conference on Computer, Control, Informatics and Its Applications, pp. 136\u2013140 (2021)","DOI":"10.1145\/3489088.3489121"}],"container-title":["Lecture Notes in Computer Science","Chinese Computational Linguistics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8367-0_15","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T12:06:55Z","timestamp":1732795615000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8367-0_15"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,29]]},"ISBN":["9789819783663","9789819783670"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8367-0_15","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,29]]},"assertion":[{"value":"29 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"No potential conflict of interest was reported by the authors.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"CCL","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China National Conference on Chinese Computational Linguistics","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Taiyuan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 July 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 July 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cncl2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/cips-cl.org\/static\/CCL2024\/en\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}