{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T14:23:25Z","timestamp":1779287005376,"version":"3.51.4"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"15","license":[{"start":{"date-parts":[[2024,2,27]],"date-time":"2024-02-27T00:00:00Z","timestamp":1708992000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,27]],"date-time":"2024-02-27T00:00:00Z","timestamp":1708992000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National key R &D Program of China","doi-asserted-by":"crossref","award":["2020AAA0107904"],"award-info":[{"award-number":["2020AAA0107904"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Major Scientific Research Project of the State Language Commission in the 13th Five-Year Plan","award":["WT135-38"],"award-info":[{"award-number":["WT135-38"]}]},{"name":"Key Support Project of NSFC-Liaoning Joint Foundation","award":["U1908216"],"award-info":[{"award-number":["U1908216"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2024,5]]},"DOI":"10.1007\/s00521-024-09547-8","type":"journal-article","created":{"date-parts":[[2024,2,27]],"date-time":"2024-02-27T20:02:20Z","timestamp":1709064140000},"page":"8641-8656","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["A multitask co-training framework for improving speech translation by leveraging speech recognition and machine translation tasks"],"prefix":"10.1007","volume":"36","author":[{"given":"Yue","family":"Zhou","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxuan","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8163-7139","authenticated-orcid":false,"given":"Xiaodong","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,2,27]]},"reference":[{"key":"9547_CR1","doi-asserted-by":"crossref","unstructured":"Fang Q, Ye R, Li L, Feng Y, Wang M (2022) Stemm: self-learning with speech-text manifold mixup for speech translation. In: Proc ACL, pp 7050\u20137062","DOI":"10.18653\/v1\/2022.acl-long.486"},{"key":"9547_CR2","doi-asserted-by":"crossref","unstructured":"Dong Q, Ye R, Wang M, Zhou H, Xu S, Xu B, Li L (2021) Listen, understand and translate: Triple supervision decouples end-to-end speech-to-text translation. In: Proc AAAI","DOI":"10.1609\/aaai.v35i14.17509"},{"key":"9547_CR3","doi-asserted-by":"crossref","unstructured":"Zhang P, Ge N, Chen B, Fan K (2019) Lattice transformer for speech translation. In: Proc ACL, pp 6475\u20136484","DOI":"10.18653\/v1\/P19-1649"},{"key":"9547_CR4","doi-asserted-by":"crossref","unstructured":"Lam TK, Schamoni S, Riezler S (2021) Cascaded models with cyclic feedback for direct speech translation. In: Proc ICASSP, pp 7508\u20137512 . IEEE","DOI":"10.1109\/ICASSP39728.2021.9413719"},{"key":"9547_CR5","doi-asserted-by":"crossref","unstructured":"Dong Q, Wang F, Yang Z, Chen W, Xu S, Xu B (2019) Adapting translation models for transcript disfluency detection. In: Proc AAAI, vol 33, pp 6351\u20136358","DOI":"10.1609\/aaai.v33i01.33016351"},{"key":"9547_CR6","doi-asserted-by":"crossref","unstructured":"Sperber M, Neubig G, Niehues J, Waibel A (2017) Neural lattice-to-sequence models for uncertain inputs. In: Proc EMNLP, pp 1380\u20131389","DOI":"10.18653\/v1\/D17-1145"},{"key":"9547_CR7","doi-asserted-by":"crossref","unstructured":"Wang C, Wu Y, Liu S, Yang Z, Zhou M (2020) Bridging the gap between pre-training and fine-tuning for end-to-end speech translation. In: Proc AAAI","DOI":"10.18653\/v1\/2020.acl-main.344"},{"key":"9547_CR8","doi-asserted-by":"crossref","unstructured":"Wang C, Wu Y, Liu S, Zhou M, Yang Z (2020) Curriculum pre-training for end-to-end speech translation. In: Proc ACL, pp 3728\u20133738","DOI":"10.18653\/v1\/2020.acl-main.344"},{"key":"9547_CR9","doi-asserted-by":"crossref","unstructured":"Tang Y, Pino J, Li X, Wang C, Genzel D (2021) Improving speech translation by understanding and learning from the auxiliary text translation task. In: Proc ACL","DOI":"10.18653\/v1\/2021.acl-long.328"},{"key":"9547_CR10","doi-asserted-by":"crossref","unstructured":"Han C, Wang M, Ji H, Li L (2021) Learning shared semantic space for speech-to-text translation. In: Proc ACL - findings, pp 2214\u20132225","DOI":"10.18653\/v1\/2021.findings-acl.195"},{"key":"9547_CR11","doi-asserted-by":"crossref","unstructured":"Ye R, Wang M, Li L (2021) End-to-end speech translation via cross-modal progressive training","DOI":"10.21437\/Interspeech.2021-1065"},{"key":"9547_CR12","doi-asserted-by":"crossref","unstructured":"Weiss RJ, Chorowski J, Jaitly N, Wu Y, Chen Z (2017) Sequence-to-sequence models can directly translate foreign speech. In: Proc Interspeech, pp 2625\u20132629","DOI":"10.21437\/Interspeech.2017-503"},{"key":"9547_CR13","doi-asserted-by":"crossref","unstructured":"Anastasopoulos A, Chiang D (2018) Tied multitask learning for neural speech translation. In: Proc NAACL-HLT","DOI":"10.18653\/v1\/N18-1008"},{"key":"9547_CR14","doi-asserted-by":"crossref","unstructured":"Bahar P, Bieschke T, Ney H (2019) A comparative study on end-to-end speech to text translation. In: Proc ASRU","DOI":"10.1109\/ASRU46091.2019.9003774"},{"key":"9547_CR15","doi-asserted-by":"crossref","unstructured":"Bansal S, Kamper H, Livescu K, Lopez A, Goldwater S (2019) Pre-training on high-resource speech recognition improves low-resource speech-to-text translation. In: Proc NAACL-HLT, pp 58\u201368","DOI":"10.18653\/v1\/N19-1006"},{"key":"9547_CR16","doi-asserted-by":"crossref","unstructured":"Tang Y, Pino J, Wang C, Ma X, Genzel D (2021) A general multi-task learning framework to leverage text data for speech to text tasks. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6209\u20136213. IEEE","DOI":"10.1109\/ICASSP39728.2021.9415058"},{"key":"9547_CR17","doi-asserted-by":"crossref","unstructured":"Ko Y, Sudoh K, Sakti S, Nakamura S (2021) ASR posterior-based loss for multi-task end-to-end speech translation. In: Interspeech, pp 2272\u20132276","DOI":"10.21437\/Interspeech.2021-1105"},{"key":"9547_CR18","doi-asserted-by":"crossref","unstructured":"Indurthi S, Han H, Lakumarapu NK, Lee B, Chung I, Kim S, Kim C (2020) Data efficient direct speech-to-text translation with modality agnostic meta-learning. In: Proceedings of ICASSP. IEEE","DOI":"10.1109\/ICASSP40776.2020.9054759"},{"key":"9547_CR19","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. In: Proceedings NeurIPS"},{"key":"9547_CR20","doi-asserted-by":"crossref","unstructured":"Le H, Pino J, Wang C, Gu J, Schwab D, Besacier L (2020) Dual-decoder transformer for joint automatic speech recognition and multilingual speech translation. In: Proc of COLING, pp 3520\u20133533","DOI":"10.18653\/v1\/2020.coling-main.314"},{"key":"9547_CR21","doi-asserted-by":"crossref","unstructured":"Du Y, Zhang Z, Wang W, Chen B, Xie J, Xu T (2022) Regularizing end-to-end speech translation with triangular decomposition agreement. In: Proc AAAI","DOI":"10.1609\/aaai.v36i10.21303"},{"key":"9547_CR22","doi-asserted-by":"crossref","unstructured":"Gaido M, Di Gangi MA, Negri M, Turchi M (2020) End-to-end speech-translation with knowledge distillation: Fbk@ iwslt2020. In: Proc INTERSPEECH, pp 80\u201388","DOI":"10.18653\/v1\/2020.iwslt-1.8"},{"key":"9547_CR23","doi-asserted-by":"crossref","unstructured":"Graves A, Fern\u00e1ndez S, Gomez F, Schmidhuber J (2006) Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: Proc ICML, pp 369\u2013376","DOI":"10.1145\/1143844.1143891"},{"key":"9547_CR24","unstructured":"B\u00e9rard A, Pietquin O, Besacier L, Servan C (2016) Listen and translate: a proof of concept for end-to-end speech-to-text translation. In: NIPS workshop on end-to-end learning for speech and audio processing"},{"key":"9547_CR25","doi-asserted-by":"crossref","unstructured":"Cheng Y, Tu Z, Meng F, Zhai J, Liu Y (2018) Towards robust neural machine translation. In: Proc ACL, pp 1756\u20131766","DOI":"10.18653\/v1\/P18-1163"},{"key":"9547_CR26","doi-asserted-by":"publisher","first-page":"1521","DOI":"10.1007\/s00521-018-3466-5","volume":"31","author":"S Lokesh","year":"2019","unstructured":"Lokesh S, Malarvizhi Kumar P, Ramya Devi M, Parthasarathy P, Gokulnath C (2019) An automatic Tamil speech recognition system by using bidirectional recurrent neural network with self-organizing map. Neural Comput. Appl. 31:1521\u20131531","journal-title":"Neural Comput. Appl."},{"key":"9547_CR27","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1007\/s00521-018-3741-5","volume":"32","author":"R Qing-dao-er-ji","year":"2020","unstructured":"Qing-dao-er-ji R, Su YL, Liu WW (2020) Research on the LSTM Mongolian and Chinese machine translation based on morpheme encoding. Neural Comput Appl 32:41\u201349","journal-title":"Neural Comput Appl"},{"key":"9547_CR28","doi-asserted-by":"crossref","unstructured":"Gulati A, Qin J, Chiu C-C, Parmar N, Zhang Y, Yu J, Han W, Wang S, Zhang Z, Wu Y, et al. (2020) Conformer: convolution-augmented transformer for speech recognition. Proc Interspeech, 5036\u20135040","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"9547_CR29","doi-asserted-by":"crossref","unstructured":"Weiss RJ, Chorowski J, Jaitly N, Wu Y, Chen Z (2017) Sequence-to-sequence models can directly translate foreign speech. In: INTERSPEECH.https:\/\/arxiv.org\/pdf\/1703.08581.pdf","DOI":"10.21437\/Interspeech.2017-503"},{"key":"9547_CR30","doi-asserted-by":"crossref","unstructured":"Vila LC, Escolano C, Fonollosa JA, Costa-Jussa MR (2018) End-to-end speech translation with the transformer. In: IberSPEECH, pp 60\u201363","DOI":"10.21437\/IberSPEECH.2018-13"},{"key":"9547_CR31","doi-asserted-by":"crossref","unstructured":"Salesky E, Sperber M, Waibel A (2019) Fluent translations from disfluent speech in end-to-end speech translation. In: Proc of NAACL-HLT, pp. 2786\u20132792","DOI":"10.18653\/v1\/N19-1285"},{"key":"9547_CR32","doi-asserted-by":"crossref","unstructured":"Ren Y, Liu J, Tan X, Zhang C, Qin T, Zhao Z, Liu T-Y (2020) Simulspeech: end-to-end simultaneous speech to text translation. In: Proc ACL, pp 3787\u20133796","DOI":"10.18653\/v1\/2020.acl-main.350"},{"key":"9547_CR33","doi-asserted-by":"crossref","unstructured":"Zhao J, Luo W, Chen B, Gilman A (2021) Mutual-learning improves end-to-end speech translation. In: Proc EMNLP, pp 3989\u20133994","DOI":"10.18653\/v1\/2021.emnlp-main.325"},{"key":"9547_CR34","doi-asserted-by":"crossref","unstructured":"Pino J, Xu Q, Ma X, Dousti MJ, Tang Y (2020) Self-training for end-to-end speech translation. In: Proc Interspeech, pp 1476\u20131480","DOI":"10.21437\/Interspeech.2020-2938"},{"key":"9547_CR35","doi-asserted-by":"crossref","unstructured":"Alinejad A, Sarkar A (2020) Effectively pretraining a speech translation decoder with machine translation data. In: Proc EMNLP, pp 8014\u20138020","DOI":"10.18653\/v1\/2020.emnlp-main.644"},{"key":"9547_CR36","doi-asserted-by":"crossref","unstructured":"Xu C, Hu B, Li Y, Zhang Y, Huang S, Ju Q, Xiao T, Zhu J (2021) Stacked acoustic-and-textual encoding: integrating the pre-trained models into speech translation encoders. In: Proc ACL, pp 2619\u20132630","DOI":"10.18653\/v1\/2021.acl-long.204"},{"key":"9547_CR37","doi-asserted-by":"crossref","unstructured":"Vydana HK, Karafi\u00e1t M, Zmolikova K, Burget L, \u010cernock\u1ef3 H (2021) Jointly trained transformers models for spoken language translation. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 7513\u20137517 . IEEE","DOI":"10.1109\/ICASSP39728.2021.9414159"},{"key":"9547_CR38","doi-asserted-by":"crossref","unstructured":"Lam TK, Schamoni S, Riezler S (2022) Sample, translate, recombine: Leveraging audio alignments for data augmentation in end-to-end speech translation. In: Proc ACL - Short Papers, pp 245\u2013254","DOI":"10.18653\/v1\/2022.acl-short.27"},{"key":"9547_CR39","doi-asserted-by":"publisher","first-page":"194","DOI":"10.1016\/j.neunet.2022.01.016","volume":"148","author":"C Mi","year":"2022","unstructured":"Mi C, Xie L, Zhang Y (2022) Improving data augmentation for low resource speech-to-text translation with diverse paraphrasing. Neural Netw 148:194\u2013205","journal-title":"Neural Netw"},{"key":"9547_CR40","first-page":"1128","volume":"2019","author":"Y Liu","year":"2019","unstructured":"Liu Y, Xiong H, Zhang J, He Z, Wu H, Wang H, Zong C (2019) End-to-end speech translation with knowledge distillation. Proc Interspeech 2019:1128\u20131132","journal-title":"Proc Interspeech"},{"key":"9547_CR41","doi-asserted-by":"crossref","unstructured":"Inaguma H, Kawahara T, Watanabe S (2021) Source and target bidirectional knowledge distillation for end-to-end speech translation. In: Proceedings of the 2021 conference of the North American chapter of the association for computational linguistics: human language technologies, pp 1872\u20131881","DOI":"10.18653\/v1\/2021.naacl-main.150"},{"key":"9547_CR42","doi-asserted-by":"crossref","unstructured":"Indurthi S, Han H, Lakumarapu NK, Lee B, Chung I, Kim S, Kim C (2020) End-end speech-to-text translation with modality agnostic meta-learning. In: Proc. ICASSP, pp 7904\u20137908 . IEEE","DOI":"10.1109\/ICASSP40776.2020.9054759"},{"key":"9547_CR43","doi-asserted-by":"crossref","unstructured":"Bahar P, Bieschke T, Ney H (2019) A comparative study on end-to-end speech to text translation. In: ASRU","DOI":"10.1109\/ASRU46091.2019.9003774"},{"key":"9547_CR44","doi-asserted-by":"crossref","unstructured":"Jia Y, Johnson M, Macherey W, Weiss RJ, Cao Y, Chiu C-C, Ari N, Laurenzo S, Wu Y (2019) Leveraging weakly supervised data to improve end-to-end speech-to-text translation. In: ICASSP 2019-2019 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 7180\u20137184 . IEEE","DOI":"10.1109\/ICASSP.2019.8683343"},{"key":"9547_CR45","doi-asserted-by":"publisher","first-page":"10403","DOI":"10.1007\/s00521-019-04577-z","volume":"32","author":"R-K Lu","year":"2020","unstructured":"Lu R-K, Liu J-W, Lian S-M, Zuo X (2020) Multi-view representation learning in multi-task scene. Neural Comput Appl 32:10403\u201310422","journal-title":"Neural Comput Appl"},{"key":"9547_CR46","unstructured":"Di Gangi MA, Cattoni R, Bentivogli L, Negri M, Turchi M (2019) Must-c: a multilingual speech translation corpus. In: Proc NAACL-HLT, pp 2012\u20132017"},{"key":"9547_CR47","doi-asserted-by":"publisher","unstructured":"Panayotov V, Chen G, Povey D, Khudanpur S (2015) Librispeech: An ASR corpus based on public domain audio books. In: 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 5206\u20135210. https:\/\/doi.org\/10.1109\/ICASSP.2015.7178964","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"9547_CR48","doi-asserted-by":"crossref","unstructured":"Inaguma H, Kiyono S, Duh K, Karita S, Yalta N, Hayashi T, Watanabe S (2020) Espnet-st: all-in-one speech translation toolkit. In: Proc ACL, pp 302\u2013311","DOI":"10.18653\/v1\/2020.acl-demos.34"},{"key":"9547_CR49","doi-asserted-by":"crossref","unstructured":"Park DS, Chan W, Zhang Y, Chiu C-C, Zoph B, Cubuk ED, Le QV (2019) Specaugment: a simple data augmentation method for automatic speech recognition. In: Proc Interspeech","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"9547_CR50","doi-asserted-by":"crossref","unstructured":"Ko T, Peddinti V, Povey D, Khudanpur S (2015) Audio augmentation for speech recognition. In: Proc. Interspeech","DOI":"10.21437\/Interspeech.2015-711"},{"key":"9547_CR51","unstructured":"Kingma DP, Ba J (2015) Adam: a method for stochastic optimization. In: Proc. ICLR"},{"key":"9547_CR52","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J (2002) Bleu: a method for automatic evaluation of machine translation. In: Proc. ACL, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"9547_CR53","unstructured":"Wang C, Tang Y, Ma X, Wu A, Okhonko D, Pino J (2020) Fairseq s2t: Fast speech-to-text modeling with fairseq. In: Proc NAACL - demonstrations, pp 33\u201339"},{"key":"9547_CR54","doi-asserted-by":"crossref","unstructured":"Zhao C, Wang M, Dong Q, Ye R, Li L (2021) Neurst: Neural speech translation toolkit. In: Proceedings of the 59th annual meeting of the association for computational linguistics and the 11th international joint conference on natural language processing: system demonstrations, pp 55\u201362","DOI":"10.18653\/v1\/2021.acl-demo.7"},{"key":"9547_CR55","doi-asserted-by":"crossref","unstructured":"Zhang B, Titov I, Haddow B, Sennrich R (2020) Adaptive feature selection for end-to-end speech translation. In: Proc. EMNLP - FIndings, pp 2533\u20132544","DOI":"10.18653\/v1\/2020.findings-emnlp.230"},{"key":"9547_CR56","doi-asserted-by":"crossref","unstructured":"Papi S, Gaido M, Negri M, Turchi M (2021) Speechformer: Reducing information loss in direct speech translation. In: Proceedings of the 2021 conference on empirical methods in natural language processing, pp 1698\u20131706","DOI":"10.18653\/v1\/2021.emnlp-main.127"},{"key":"9547_CR57","doi-asserted-by":"crossref","unstructured":"Li X, Wang C, Tang Y, Tran C, Tang Y, Pino J, Baevski A, Conneau A, Auli M (2021) Multilingual speech translation from efficient finetuning of pretrained models. In: Proc ACL, pp 827\u2013838","DOI":"10.18653\/v1\/2021.acl-long.68"},{"key":"9547_CR58","doi-asserted-by":"crossref","unstructured":"Chen J, Ma M, Zheng R, Huang L (2021) Specrec: an alternative solution for improving end-to-end speech-to-text translation via spectrogram reconstruction. In: Proc Interspeech, pp 2232\u20132236","DOI":"10.21437\/Interspeech.2021-733"},{"key":"9547_CR59","doi-asserted-by":"crossref","unstructured":"Le H, Pino J, Wang C, Gu J, Schwab D, Besacier L (2021) Lightweight adapter tuning for multilingual speech translation. In: Proc ACL - short papers, pp 817\u2013824","DOI":"10.18653\/v1\/2021.acl-short.103"},{"key":"9547_CR60","unstructured":"Zheng R, Chen J, Ma M, Huang L (2021) Fused acoustic and text encoding for multimodal bilingual pretraining and speech translation. In: Proc. ICML, pp 12736\u201312746 . PMLR"},{"key":"9547_CR61","unstructured":"Baevski A, Zhou Y, Mohamed A, Auli M (2020) wav2vec 2.0: A framework for self-supervised learning of speech representations"},{"key":"9547_CR62","first-page":"2247","volume":"2021","author":"C Wang","year":"2021","unstructured":"Wang C, Wu A, Gu J, Pino J (2021) Covost 2 and massively multilingual speech translation. Proc Interspeech 2021:2247\u20132251","journal-title":"Proc Interspeech"},{"key":"9547_CR63","unstructured":"Zhang B, Haddow B, Sennrich R (2022) Revisiting end-to-end speech-to-text translation from scratch. In: International conference on machine learning, pp 26193\u201326205. PMLR"},{"key":"9547_CR64","volume-title":"Findings of the association for computational linguistics: ACL 2023","author":"D Zhang","year":"2023","unstructured":"Zhang D, Ye R, Ko T, Wang M, Zhou Y (2023) DUB: Discrete unit back-translation for speech translation. In: Rogers A, Boyd-Graber J, Okazaki N (eds) Findings of the association for computational linguistics: ACL 2023. Association for Computational Linguistics, Toronto"},{"key":"9547_CR65","unstructured":"Gangi MAD, Cattoni R, Bentivogli L, Negri M, Turchi M (2019) MuST-C: a multilingual speech translation corpus. In: NAACL-HLT . https:\/\/www.aclweb.org\/anthology\/N19-1202.pdf"},{"key":"9547_CR66","doi-asserted-by":"publisher","unstructured":"Gaido M, Cettolo M, Negri M, Turchi M (2021) CTC-based compression for direct speech translation. In: Proceedings of the 16th conference of the European chapter of the association for computational linguistics: main volume, pp 690\u2013696. Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/2021.eacl-main.57","DOI":"10.18653\/v1\/2021.eacl-main.57"},{"key":"9547_CR67","doi-asserted-by":"crossref","unstructured":"Dong L, Xu B (2020) Cif: Continuous integrate-and-fire for end-to-end speech recognition. In: ICASSP 2020-2020 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6079\u20136083. IEEE","DOI":"10.1109\/ICASSP40776.2020.9054250"},{"key":"9547_CR68","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.101830","volume":"98","author":"J Lin","year":"2023","unstructured":"Lin J, Song J, Zhou Z, Chen Y, Shi X (2023) Automated scholarly paper review: concepts, technologies, and challenges. Inf Fusion 98:101830","journal-title":"Inf Fusion"},{"key":"9547_CR69","doi-asserted-by":"crossref","unstructured":"Bai P, Zhou Y, Zheng M, Sun W, Shi X (2023) Improving chinese pop song and hokkien gezi opera singing voice synthesis by enhancing local modeling. In: Proceedings of the 2023 conference on empirical methods in natural language processing, pp 3302\u20133312","DOI":"10.18653\/v1\/2023.emnlp-main.200"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09547-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-024-09547-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09547-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,4,16]],"date-time":"2024-04-16T15:26:33Z","timestamp":1713281193000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-024-09547-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,27]]},"references-count":69,"journal-issue":{"issue":"15","published-print":{"date-parts":[[2024,5]]}},"alternative-id":["9547"],"URL":"https:\/\/doi.org\/10.1007\/s00521-024-09547-8","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2,27]]},"assertion":[{"value":"5 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 January 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 February 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interest regarding the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}