{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:06Z","timestamp":1779228366205,"version":"3.51.4"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62067008"],"award-info":[{"award-number":["62067008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100010726","name":"Northwest Normal University","doi-asserted-by":"publisher","award":["NWNU-LKQN2024-11"],"award-info":[{"award-number":["NWNU-LKQN2024-11"]}],"id":[{"id":"10.13039\/100010726","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101956","type":"journal-article","created":{"date-parts":[[2026,2,24]],"date-time":"2026-02-24T07:45:30Z","timestamp":1771919130000},"page":"101956","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Cro-MTVITS: An end-to-end cross-lingual speech synthesis model for Mandarin and multi-dialect Tibetan based on VITS"],"prefix":"10.1016","volume":"100","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3692-4921","authenticated-orcid":false,"given":"Weizhao","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mengjuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junzhi","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongwu","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101956_b1","first-page":"1409","article-title":"Unit selection in tibetan speech synthesis","author":"Cairangzhuoma","year":"2015","journal-title":"J. Softw."},{"key":"10.1016\/j.csl.2026.101956_b2","doi-asserted-by":"crossref","unstructured":"Cao, Y., Liu, S., Wu, X., Kang, S., Liu, P., Wu, Z., Liu, X., Su, D., Yu, D., Meng, H., 2020. Code-switched speech synthesis using bilingual phonetic posteriorgram with only monolingual corpora. In: 2020 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 7619\u20137623.","DOI":"10.1109\/ICASSP40776.2020.9053094"},{"key":"10.1016\/j.csl.2026.101956_b3","doi-asserted-by":"crossref","unstructured":"Cao, Y., Wu, X., Liu, S., Yu, J., Li, X., Wu, Z., Liu, X., Meng, H., 2019. End-to-end code-switched tts with mix of monolingual recordings. In: 2019 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 6935\u20136939.","DOI":"10.1109\/ICASSP.2019.8682927"},{"key":"10.1016\/j.csl.2026.101956_b4","doi-asserted-by":"crossref","first-page":"705","DOI":"10.1109\/TASLPRO.2025.3530270","article-title":"Neural codec language models are zero-shot text to speech synthesizers","volume":"33","author":"Chen","year":"2025","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"2","key":"10.1016\/j.csl.2026.101956_b5","first-page":"75","article-title":"Neural network based tibetan speech synthesis","volume":"33","author":"Dou","year":"2019","journal-title":"J. Chin. Inf. Process."},{"key":"10.1016\/j.csl.2026.101956_b6","series-title":"Cosyvoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens","author":"Du","year":"2024"},{"key":"10.1016\/j.csl.2026.101956_b7","doi-asserted-by":"crossref","unstructured":"Fan, Y., Qian, Y., Soong, F.K., He, L., 2016. Speaker and language factorization in dnn-based tts synthesis. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 5540\u20135544.","DOI":"10.1109\/ICASSP.2016.7472737"},{"key":"10.1016\/j.csl.2026.101956_b8","doi-asserted-by":"crossref","unstructured":"Fan, Y., Qian, Y., Xie, F.-L., Soong, F.K., 2014. Tts synthesis with bidirectional lstm based recurrent neural networks. In: Interspeech. pp. 1964\u20131968.","DOI":"10.21437\/Interspeech.2014-443"},{"key":"10.1016\/j.csl.2026.101956_b9","doi-asserted-by":"crossref","unstructured":"Guo, W., Yang, H., Gan, Z., 2018. A dnn-based mandarin-tibetan cross-lingual speech synthesis. In: 2018 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference. APSIPA ASC, pp. 1702\u20131707.","DOI":"10.23919\/APSIPA.2018.8659668"},{"key":"10.1016\/j.csl.2026.101956_b10","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101956_b11","unstructured":"Jiang, Z., Liu, J., Ren, Y., He, J., Ye, Z., Ji, S., Yang, Q., Zhang, C., Wei, P., Wang, C., Yin, X., Ma, Z., Zhao, Z., 2024. Mega-tts 2: Boosting prompting mechanisms for zero-shot speech synthesis. In: International Conference on Learning Representations. Vol. 2024, pp. 57919\u201357939."},{"key":"10.1016\/j.csl.2026.101956_b12","series-title":"Mega-tts: Zero-shot text-to-speech at scale with intrinsic inductive bias","author":"Jiang","year":"2023"},{"key":"10.1016\/j.csl.2026.101956_b13","series-title":"International Conference on Machine Learning","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","author":"Kim","year":"2021"},{"issue":"9","key":"10.1016\/j.csl.2026.101956_b14","first-page":"82","article-title":"An end-to-end tibetan speech synthesis method","volume":"38","author":"Lhakpa","year":"2024","journal-title":"J. Chin. Inf. Process."},{"key":"10.1016\/j.csl.2026.101956_b15","doi-asserted-by":"crossref","unstructured":"Li, B., Zhang, Y., Sainath, T., Wu, Y., Chan, W., 2019. Bytes are all you need: End-to-end multilingual speech recognition and synthesis with bytes. In: 2019 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 5621\u20135625.","DOI":"10.1109\/ICASSP.2019.8682674"},{"issue":"5","key":"10.1016\/j.csl.2026.101956_b16","article-title":"End-to-end speech synthesis for tibetan lhasa dialect","volume":"1187","author":"Luo","year":"2019","journal-title":"J. Phys.: Conf. Ser."},{"key":"10.1016\/j.csl.2026.101956_b17","doi-asserted-by":"crossref","unstructured":"McAuliffe, M., Socolof, M., Mihuc, S., Wagner, M., Sonderegger, M., 2017. Montreal forced aligner: Trainable text-speech alignment using kaldi. In: Interspeech. Vol. 2017, pp. 498\u2013502.","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"10.1016\/j.csl.2026.101956_b18","doi-asserted-by":"crossref","unstructured":"Nachmani, E., Wolf, L., 2019. Unsupervised polyglot text-to-speech. In: 2019 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 7055\u20137059.","DOI":"10.1109\/ICASSP.2019.8683519"},{"key":"10.1016\/j.csl.2026.101956_b19","doi-asserted-by":"crossref","unstructured":"Polyak, A., Adi, Y., Copet, J., Kharitonov, E., Lakhotia, K., Hsu, W.-N., Mohamed, A., Dupoux, E., 2021. Speech resynthesis from discrete disentangled self-supervised representations. In: Interspeech. pp. 3615\u20133619.","DOI":"10.21437\/Interspeech.2021-475"},{"key":"10.1016\/j.csl.2026.101956_b20","doi-asserted-by":"crossref","unstructured":"Qian, Y., Fan, Y., Hu, W., Soong, F.K., 2014. On the training aspects of deep neural network (dnn) for parametric tts synthesis. In: 2014 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 3829\u20133833.","DOI":"10.1109\/ICASSP.2014.6854318"},{"key":"10.1016\/j.csl.2026.101956_b21","unstructured":"Ren, Y., Hu, C., Tan, X., Qin, T., Zhao, S., Zhao, Z., Liu, T.-Y., 2021. Fastspeech 2: Fast and high-quality end-to-end text to speech. In: International Conference on Learning Representations."},{"key":"10.1016\/j.csl.2026.101956_b22","series-title":"Mparrottts: Multilingual multi-speaker text to speech synthesis in low resource setting","author":"Shah","year":"2023"},{"key":"10.1016\/j.csl.2026.101956_b23","doi-asserted-by":"crossref","unstructured":"Shen, J., Pang, R., Weiss, R.J., Schuster, M., Jaitly, N., Yang, Z., Chen, Z., Zhang, Y., Wang, Y., R. Skerrv-Ryan, R.A. Saurous, Agiomvrgiannakis, Y., Wu, Y., 2018. Natural tts synthesis by conditioning wavenet on mel spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 4779\u20134783.","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"10.1016\/j.csl.2026.101956_b24","doi-asserted-by":"crossref","unstructured":"Shi, Y., Bu, H., Xu, X., Zhang, S., Li, M., 2021. Aishell-3: A multi-speaker mandarin tts corpus. In: Interspeech. pp. 2756\u20132760.","DOI":"10.21437\/Interspeech.2021-755"},{"key":"10.1016\/j.csl.2026.101956_b25","series-title":"International Conference on Machine Learning","first-page":"4693","article-title":"Towards end-to-end prosody transfer for expressive speech synthesis with tacotron","author":"Skerry-Ryan","year":"2018"},{"issue":"6","key":"10.1016\/j.csl.2026.101956_b26","doi-asserted-by":"crossref","first-page":"4234","DOI":"10.1109\/TPAMI.2024.3356232","article-title":"Naturalspeech: End-to-end text-to-speech synthesis with human-level quality","volume":"46","author":"Tan","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"5","key":"10.1016\/j.csl.2026.101956_b27","doi-asserted-by":"crossref","first-page":"1234","DOI":"10.1109\/JPROC.2013.2251852","article-title":"Speech synthesis based on hidden markov models","volume":"101","author":"Tokuda","year":"2013","journal-title":"Proc. IEEE"},{"key":"10.1016\/j.csl.2026.101956_b28","doi-asserted-by":"crossref","unstructured":"Traber, C., Huber, K., Nedir, K., Pfister, B., Keller, E., Zellner, B., 1999. From multilingual to polyglot speech synthesis. In: Eurospeech. pp. 835\u2013838.","DOI":"10.21437\/Eurospeech.1999-216"},{"key":"10.1016\/j.csl.2026.101956_b29","first-page":"776","article-title":"Realizing mandarin-tibetan bilingual speech synthesis by speaker adaptive training","volume":"53","author":"Wang","year":"2013","journal-title":"J. Tsinghua Univ."},{"key":"10.1016\/j.csl.2026.101956_b30","doi-asserted-by":"crossref","unstructured":"Wu, Z., Yu, H., Li, G., Wan, S., 2013. Hmm-based tibetan lhasa speech synthesis system. In: Proceedings of 2013 3rd International Conference on Computer Science and Network Technology. pp. 92\u201395.","DOI":"10.1109\/ICCSNT.2013.6967070"},{"issue":"1","key":"10.1016\/j.csl.2026.101956_b31","article-title":"End-to-end speech synthesis for tibetan multidialect","volume":"2021","author":"Xu","year":"2021","journal-title":"Complexity"},{"key":"10.1016\/j.csl.2026.101956_b32","doi-asserted-by":"crossref","unstructured":"Yamauchi, K., Saito, Y., Saruwatari, H., 2024. Cross-dialect text-to-speech in pitch-accent language incorporating multi-dialect phoneme-level bert. In: 2024 IEEE Spoken Language Technology Workshop. SLT, pp. 750\u2013757.","DOI":"10.1109\/SLT61566.2024.10832155"},{"issue":"22","key":"10.1016\/j.csl.2026.101956_b33","doi-asserted-by":"crossref","first-page":"9927","DOI":"10.1007\/s11042-014-2117-9","article-title":"Using speaker adaptive training to realize mandarin-tibetan cross-lingual speech synthesis","volume":"74","author":"Yang","year":"2015","journal-title":"Multimedia Tools Appl."},{"key":"10.1016\/j.csl.2026.101956_b34","unstructured":"Yang, Z., Xu, Z., Cui, Y., Wang, B., Lin, M., Wu, D., Chen, Z., 2022. Cino: A chinese minority pre-trained language model. In: Proceedings of the 29th International Conference on Computational Linguistics. pp. 3937\u20133949."},{"key":"10.1016\/j.csl.2026.101956_b35","doi-asserted-by":"crossref","unstructured":"Yu, Q., Liu, P., Wu, Z., Ang, S.K., Meng, H., Cai, L., 2016. Learning cross-lingual information with multilingual blstm for speech synthesis of low-resource languages. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 5545\u20135549.","DOI":"10.1109\/ICASSP.2016.7472738"},{"issue":"6","key":"10.1016\/j.csl.2026.101956_b36","doi-asserted-by":"crossref","first-page":"1713","DOI":"10.1109\/TASL.2012.2187195","article-title":"Statistical parametric speech synthesis based on speaker and language factorization","volume":"20","author":"Zen","year":"2012","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101956_b37","doi-asserted-by":"crossref","unstructured":"Zen, H., Sak, H., 2015. Unidirectional long short-term memory recurrent neural network with recurrent output layer for low-latency speech synthesis. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 4470\u20134474.","DOI":"10.1109\/ICASSP.2015.7178816"},{"key":"10.1016\/j.csl.2026.101956_b38","doi-asserted-by":"crossref","unstructured":"Zen, H., Senior, A., Schuster, M., 2013. Statistical parametric speech synthesis using deep neural networks. In: 2013 IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, pp. 7962\u20137966.","DOI":"10.1109\/ICASSP.2013.6639215"},{"key":"10.1016\/j.csl.2026.101956_b39","doi-asserted-by":"crossref","unstructured":"Zhan, H., Zhang, H., Ou, W., Lin, Y., 2021. Improve cross-lingual text-to-speech synthesis on monolingual corpora with pitch contour information. In: Interspeech. pp. 1599\u20131603.","DOI":"10.21437\/Interspeech.2021-474"},{"key":"10.1016\/j.csl.2026.101956_b40","first-page":"1251","article-title":"Speech synthesis for tibetan amdo dialect base on complete end-to-end","author":"Zhang","year":"2025","journal-title":"J. Appl. Acoust."},{"key":"10.1016\/j.csl.2026.101956_b41","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Weiss, R.J., Zen, H., Wu, Y., Chen, Z., Skerry-Ryan, R., Jia, Y., Rosenberg, A., Ramabhadran, B., 2019. Learning to speak fluently in a foreign language: Multilingual speech synthesis and cross-language voice cloning. In: Interspeech. pp. 2080\u20132084.","DOI":"10.21437\/Interspeech.2019-2668"},{"issue":"23","key":"10.1016\/j.csl.2026.101956_b42","doi-asserted-by":"crossref","first-page":"12185","DOI":"10.3390\/app122312185","article-title":"Meta-learning for mandarin-tibetan cross-lingual speech synthesis","volume":"12","author":"Zhang","year":"2022","journal-title":"Appl. Sci."},{"issue":"9","key":"10.1016\/j.csl.2026.101956_b43","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3616012","article-title":"Improving sequence-to-sequence tibetan speech synthesis with prosodic information","volume":"22","author":"Zhang","year":"2023","journal-title":"ACM Trans. Asian Low-Resour. Lang. Inf. Process."},{"key":"10.1016\/j.csl.2026.101956_b44","doi-asserted-by":"crossref","unstructured":"Zhang, W., Yang, H., Bu, X., 2019. Kham dialect speech synthesis based on deep learning. In: 2019 International Joint Conference on Information, Media and Engineering. IJCIME, pp. 280\u2013283.","DOI":"10.1109\/IJCIME49369.2019.00063"},{"key":"10.1016\/j.csl.2026.101956_b45","doi-asserted-by":"crossref","first-page":"140305","DOI":"10.1109\/ACCESS.2019.2940125","article-title":"Lhasa-tibetan speech synthesis using end-to-end model","volume":"7","author":"Zhao","year":"2019","journal-title":"IEEE Access"},{"issue":"15","key":"10.1016\/j.csl.2026.101956_b46","doi-asserted-by":"crossref","first-page":"6834","DOI":"10.3390\/app14156834","article-title":"Tibetan speech synthesis based on pre-traind mixture alignment fastspeech2","volume":"14","author":"Zhou","year":"2024","journal-title":"Appl. Sci."},{"issue":"2","key":"10.1016\/j.csl.2026.101956_b47","first-page":"24","article-title":"A dataset of tibetan dialect speech synthesis","volume":"7","author":"Zhuoma","year":"2022","journal-title":"China Sci. Data"},{"key":"10.1016\/j.csl.2026.101956_b48","doi-asserted-by":"crossref","unstructured":"Zu, B., Cai, R., Cai, Z., Pengmao, Z., 2022. Research on tibetan speech synthesis based on fastspeech2. In: 2022 3rd International Conference on Pattern Recognition and Machine Learning. PRML, pp. 241\u2013244.","DOI":"10.1109\/PRML56267.2022.9882187"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000197?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000197?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:11:24Z","timestamp":1779225084000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000197"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":48,"alternative-id":["S0885230826000197"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101956","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Cro-MTVITS: An end-to-end cross-lingual speech synthesis model for Mandarin and multi-dialect Tibetan based on VITS","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101956","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101956"}}