{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T08:51:24Z","timestamp":1781772684610,"version":"3.54.5"},"reference-count":31,"publisher":"China Science Publishing & Media Ltd.","issue":"2","content-domain":{"domain":["engine.scichina.com"],"crossmark-restriction":false},"short-container-title":["DI"],"published-print":{"date-parts":[[2026,6,1]]},"DOI":"10.3724\/2096-7004.di.2025.0139","type":"journal-article","created":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T07:08:14Z","timestamp":1758524894000},"page":"20250139","update-policy":"https:\/\/doi.org\/10.1360\/scp-crossmark-policy-page","source":"Crossref","is-referenced-by-count":0,"title":["Dynamic Language Routing Mixture of Experts Model for Multilingual Speech Recognition"],"prefix":"10.3724","volume":"8","author":[{"given":"Junchen","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yonghe","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feilong","family":"Bao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guanglai","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"2026","published-online":{"date-parts":[[2025,9,22]]},"reference":[{"key":"null","unstructured":"Radford A., Kim J. W., Xu T., Brockman G., McLeavey C., and Sutskever I., \u201cRobust speech recognition via large-scale weak supervision,\u201d in\u00a0 International Conference on Machine Learning. PMLR, 2023, pp. 28492\u201328518."},{"key":"null","unstructured":"Zhang Y. \u00a0 et al., \u201cGoogle usm: Scaling automatic speech recognition beyond 100 languages,\u201d 2023. [Online]. Available: arXiv:2303.01037."},{"key":"null","unstructured":"Pratap V. \u00a0 et al., \u201cScaling speech technology to 1,000+ languages,\u201d\u00a0 Journal of Machine Learning Research, vol. 25, no. 97, pp. 1\u201352, 2024."},{"key":"null","unstructured":"Saif A., Chen L., Cui X., Lu S., Kingsbury B., and Chen T., \u201cM2asr: Multilingual multi-task automatic speech recognition via multi-objective optimization,\u201d in\u00a0 Interspeech, vol. 2024, 2024, pp. 1240\u20131244."},{"key":"null","unstructured":"Ferraz T. P., Boito M. Z., Brun C., and Nikoulina V., \u201cMultilingual distillwisher: Efficient distillation of multi-task speech models via language-specific experts,\u201d in\u00a0 ICASSP 2024, 2024, pp. 10716\u201310720."},{"key":"null","unstructured":"Kashiwagi Y., Futami H., Tsunoo E., Arora S., and Watanabe S., \u201cRapid language adaptation for multilingual e2e speech recognition using encoder prompting,\u201d in\u00a0 Proc. Interspeech 2024, 2024, pp. 2900\u20132904."},{"key":"null","unstructured":"Liu W., Hou J., Yang D., Cao M., and Lee T., \u201cA parameter-efficient language extension framework for multilingual asr,\u201d in\u00a0 Proc. Interspeech 2024, 2024, pp. 3929\u20133933."},{"key":"null","unstructured":"Chen P. \u00a0 et al., \u201cBa-moe: Boundary-aware mixture-of-experts adapter for code-switching speech recognition,\u201d in\u00a0 2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), 2023, pp. 1\u20137."},{"key":"null","unstructured":"Wu Y., Peng Y., Lu Y., Chang X., Song R., and Watanabe S., \u201cRobust audiovisual speech recognition models with mixture-of-experts,\u201d in\u00a0 2024 IEEE Spoken Language Technology Workshop (SLT), 2024, pp. 43\u201348."},{"key":"null","unstructured":"Wang W., Ma G., Li Y., and Du B., \u201cLanguage-routing mixture of experts for multilingual and code-switching speech recognition,\u201d in\u00a0 Proc. Interspeech 2023, 2023, pp. 1389\u20131393."},{"key":"null","unstructured":"Ma G., Wang W., Zhou L., Yang Y., Li Y., and Du B., \u201cBlr-moe: Boosted language-routing mixture of experts for domain-robust multilingual e2e asr,\u201d 2025. [Online]. Available: arXiv:2501.12602."},{"key":"null","unstructured":"Huang H. \u00a0 et al., \u201cDynamic language group-based moe: Enhancing code-switching speech recognition with hierarchical routing,\u201d in\u00a0 ICASSP 2025, 2025, pp. 1\u20135."},{"key":"null","unstructured":"You Z., Feng S., Su D., and Yu D., \u201cSpeechmoe: Scaling to large acoustic models with dynamic routing mixture of experts,\u201d in\u00a0 Proc. Interspeech 2021, 2021, pp. 2077\u20132081."},{"key":"null","unstructured":"You Z., Feng S., Su D., and Yu D., \u201cSpeechmeo2: Mixture-of-experts model with improved routing,\u201d in\u00a0 ICASSP 2022, 2022, pp. 7217\u20137221."},{"key":"null","unstructured":"You Z., Feng S., Su D., and Yu D., \u201c3m: Multi-loss, multi-path and multi-level neural networks for speech recognition,\u201d in\u00a0 2022 13th International Symposium on Chinese Spoken Language Processing (ISCSLP), 2022, pp. 170\u2013174."},{"key":"null","unstructured":"Kwon Y. and Chung S.-W., \u201cMole: Mixture of language experts for multi-lingual automatic speech recognition,\u201d in\u00a0 ICASSP 2023, 2023, pp. 1\u20135."},{"key":"null","unstructured":"Cao S., Wang X., Zhang Y., Zhang X., and Ma L., \u201cM-moe: Mixture of mixture-of-expert model for ctc-based streaming multilingual asr,\u201d in\u00a0 ICASSP 2025, 2025, pp. 1\u20135."},{"key":"null","unstructured":"Kannan A. \u00a0 et al., \u201cLarge-scale multilingual speech recognition with a streaming end-to-end model,\u201d in\u00a0 Proc. Interspeech 2019, 2019, pp. 2130\u20132134."},{"key":"null","unstructured":"Li B. \u00a0 et al., \u201cEfficient domain adaptation for speech foundation models,\u201d in\u00a0 ICASSP 2023, 2023, pp. 1\u20135."},{"key":"null","unstructured":"Feng P., Zhang X., Zhao J., Wang Y., and Huang B., \u201cRelation extraction based on prompt information and feature reuse,\u201d\u00a0 Data Intelligence, vol. 5, no. 3, pp. 817\u2013833, 2023. [Online]. Available: https:\/\/doi.org\/10.1162\/dint_a_00192."},{"key":"null","unstructured":"Bao M. and De Q., \u201cConstruction of mongolian near-synonymous compound qualitative adjective sets,\u201d\u00a0 Data Intelligence, vol. 7, no. 1, pp. 221\u2013236, 2025. [Online]. Available: https:\/\/doi.org\/10.3724\/2096-7004.di.2025.0008."},{"key":"null","unstructured":"Hu K., Li B., Sainath T., Zhang Y., and Beaufays F., \u201cMixture-of-expert conformer for streaming multilingual asr,\u201d in\u00a0 Proc. Interspeech 2023, 2023, pp. 3327\u20133331."},{"key":"null","unstructured":"Hori T., Watanabe S., and Hershey J. R., \u201cJoint ctc\/attention decoding for end-to-end speech recognition,\u201d in\u00a0 Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), 2017, pp. 518\u2013529."},{"key":"null","unstructured":"Panayotov V., Chen G., Povey D., and Khudanpur S., \u201cLibrispeech: An asr corpus based on public domain audio books,\u201d in\u00a0 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), 2015, pp. 5206\u20135210."},{"key":"null","unstructured":"Bu H., Du J., Na X., Wu J., and Zheng M. L., \u201cAishell-1: An open-source mandarin speech corpus and a speech recognition baseline,\u201d in\u00a0 Proceedings of the 20th Conference of the Oriental Chapter of the International Coordinating Committee on Speech Databases and Speech I\/O Systems and Assessment (O-COCOSDA), 2017, pp. 1\u20135."},{"key":"null","unstructured":"Wu Y., Wang Y., Zhang H., Bao F., and Gao G., \u201cMnasr: A free speech corpus for mongolian speech recognition and accompanied baselines,\u201d in\u00a0 2022 25th Conference of the Oriental COCOSDA International Committee for the Co-ordination and Standardisation of Speech Databases and Assessment Techniques (O-COCOSDA), 2022, pp. 1\u20136."},{"key":"null","unstructured":"Zhang B. \u00a0 et al., \u201cWenet 2.0: More productive end-to-end speech recognition toolkit,\u201d in\u00a0 Proc. Interspeech 2022, 2022, pp. 1661\u20131665."},{"key":"null","unstructured":"Ko T., Peddinti V., Povey D., and Khudanpur S., \u201cAudio augmentation for speech recognition,\u201d in\u00a0 Interspeech, 2015, pp. 3586\u20133589."},{"key":"null","unstructured":"Park D. S. \u00a0 et al., \u201cSpecaugment: A simple data augmentation method for automatic speech recognition,\u201d in\u00a0 Interspeech, 2019, pp. 2613\u20132617."},{"key":"null","unstructured":"Kingma D. P. and Ba J. L., \u201cAdam: A method for stochastic optimization,\u201d in\u00a0 ICLR, 2015."},{"key":"null","unstructured":"Vaswani A. \u00a0 et al., \u201cAttention is all you need,\u201d in\u00a0 Adv. Neural Inf. Process. Syst. (NeurIPS), 2017, pp. 5998\u20136008."}],"container-title":["Data Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/297C19FDBE634578A55F03DC2B625B4F","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0139","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/297C19FDBE634578A55F03DC2B625B4F","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T07:58:30Z","timestamp":1781769510000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0139"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,22]]},"references-count":31,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2025,9,22]]},"published-print":{"date-parts":[[2026,6,1]]}},"URL":"https:\/\/doi.org\/10.3724\/2096-7004.di.2025.0139","relation":{},"ISSN":["2096-7004"],"issn-type":[{"value":"2096-7004","type":"print"}],"subject":[],"published":{"date-parts":[[2025,9,22]]}}}