{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T15:27:16Z","timestamp":1750951636958,"version":"3.28.0"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,12,16]],"date-time":"2023-12-16T00:00:00Z","timestamp":1702684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,12,16]],"date-time":"2023-12-16T00:00:00Z","timestamp":1702684800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,12,16]]},"DOI":"10.1109\/asru57964.2023.10389659","type":"proceedings-article","created":{"date-parts":[[2024,1,19]],"date-time":"2024-01-19T13:38:40Z","timestamp":1705671520000},"page":"1-8","source":"Crossref","is-referenced-by-count":1,"title":["Bisinger: Bilingual Singing Voice Synthesis"],"prefix":"10.1109","author":[{"given":"Huali","family":"Zhou","sequence":"first","affiliation":[{"name":"Wuhan University,School of Computer Science,Wuhan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yueqian","family":"Lin","sequence":"additional","affiliation":[{"name":"Duke Kunshan University,Suzhou Municipal Key Laboratory of Multimodal Intelligent Systems,Kunshan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yao","family":"Shi","sequence":"additional","affiliation":[{"name":"Duke Kunshan University,Suzhou Municipal Key Laboratory of Multimodal Intelligent Systems,Kunshan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peng","family":"Sun","sequence":"additional","affiliation":[{"name":"Duke Kunshan University,Suzhou Municipal Key Laboratory of Multimodal Intelligent Systems,Kunshan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Li","sequence":"additional","affiliation":[{"name":"Wuhan University,School of Computer Science,Wuhan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1410"},{"key":"ref2","article-title":"Fastspeech: Fast, robust and controllable text to speech","volume":"32","author":"Ren","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP49672.2021.9362104"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"ref5","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","volume-title":"International Conference on Machine Learning","author":"Kim"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747664"},{"article-title":"Recent development of the hmm-based singing voice synthesis system\u2014sinsy","volume-title":"Seventh ISCA Workshop on Speech Synthesis","author":"Oura","key":"ref7"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10039"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096239"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21350"},{"article-title":"Children\u2019s song dataset for singing voice research","volume-title":"International Society for Music Information Retrieval Conference (ISMIR)","author":"Choi","key":"ref11"},{"key":"ref12","first-page":"6914","article-title":"M4singer: A multi-style, multi-singer and musical score provided mandarin singing corpus","volume":"35","author":"Zhang","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2013.6694316"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2022.101427"},{"key":"ref15","first-page":"4009","article-title":"Vocaloid-commercial singing synthesizer based on sample concatenation","volume":"2007","author":"Kenmochi","year":"2007","journal-title":"Interspeech"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-872"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853599"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403249"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2019-2668"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682674"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682927"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2679"},{"key":"ref23","article-title":"Cross-lingual multi-speaker text-to-speech synthesis for voice cloning without using parallel corpus for unseen speakers","author":"Liu","year":"2019","journal-title":"arXiv preprint arXiv:1911.11601"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1464"},{"key":"ref25","article-title":"Learning pronunciation from a foreign language in speech synthesis networks","author":"Lee","year":"2018","journal-title":"arXiv preprint arXiv:1811.09364"},{"key":"ref26","article-title":"Improve bilingual tts using dynamic language and phonology embedding","author":"Yang","year":"2022","journal-title":"arXiv preprint arXiv:2212.03435"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref29","first-page":"17022","article-title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis","volume":"33","author":"Kong","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref30","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"International Conference on Machine Learning","author":"Radford"}],"event":{"name":"2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","start":{"date-parts":[[2023,12,16]]},"location":"Taipei, Taiwan","end":{"date-parts":[[2023,12,20]]}},"container-title":["2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10388490\/10389614\/10389659.pdf?arnumber=10389659","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,23]],"date-time":"2024-01-23T11:35:50Z","timestamp":1706009750000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10389659\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,16]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/asru57964.2023.10389659","relation":{},"subject":[],"published":{"date-parts":[[2023,12,16]]}}}