{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T20:10:12Z","timestamp":1776888612151,"version":"3.51.2"},"publisher-location":"Singapore","reference-count":31,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819787944","type":"print"},{"value":"9789819787951","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8795-1_18","type":"book-chapter","created":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T23:03:50Z","timestamp":1730588630000},"page":"263-277","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["StyleFusion TTS: Multimodal Style-Control and\u00a0Enhanced Feature Fusion for\u00a0Zero-Shot Text-to-Speech Synthesis"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9629-6111","authenticated-orcid":false,"given":"Zhiyong","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9600-868X","authenticated-orcid":false,"given":"Xinnuo","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1034-9972","authenticated-orcid":false,"given":"Zhiqi","family":"Ai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1905-6269","authenticated-orcid":false,"given":"Shugong","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,3]]},"reference":[{"key":"18_CR1","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F.L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et\u00a0al.: Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"18_CR2","unstructured":"Adigwe, A., Tits, N., Haddad, K.E., Ostadabbas, S., Dutoit, T.: The emotional voices database: towards controlling the emotion dimension in voice generation systems. arXiv preprint arXiv:1806.09514 (2018)"},{"key":"18_CR3","unstructured":"Casanova, E., Weber, J., Shulby, C.D., Junior, A.C., G\u00f6lge, E., Ponti, M.A.: YourTTS: Towards zero-shot multi-speaker TTS and zero-shot voice conversion for everyone. In: Proceedings of the 39th International Conference on Machine Learning, pp. 2709\u20132720. PMLR, https:\/\/proceedings.mlr.press\/v162\/casanova22a.html, ISSN: 2640-3498"},{"issue":"1","key":"18_CR4","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1186\/s13636-024-00351-9","volume":"2024","author":"Z Chen","year":"2024","unstructured":"Chen, Z., Ai, Z., Ma, Y., Li, X., Xu, S.: Optimizing feature fusion for improved zero-shot adaptation in text-to-speech synthesis. EURASIP J. Audio Speech Music Process. 2024(1), 28 (2024)","journal-title":"EURASIP J. Audio Speech Music Process."},{"key":"18_CR5","unstructured":"Chevi, R., Aji, A.F.: Daisy-TTS: simulating wider spectrum of emotions via prosody embedding decomposition, http:\/\/arxiv.org\/abs\/2402.14523"},{"key":"18_CR6","unstructured":"emotivoice: Emotivoice system (2024), https:\/\/replicate.com\/bramhooimeijer\/emotivoice"},{"key":"18_CR7","unstructured":"Esser, P., Kulal, S., Blattmann, A., Entezari, R., M\u00fcller, J., Saini, H., Levi, Y., Lorenz, D., Sauer, A., Boesel, F., et\u00a0al.: Scaling rectified flow transformers for high-resolution image synthesis. arXiv preprint arXiv:2403.03206 (2024)"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Guan, W., Li, Y., Li, T., Huang, H., Wang, F., Lin, J., Huang, L., Li, L., Hong, Q.: Mm-tts: Multi-modal prompt based style transfer for expressive text-to-speech synthesis. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a038, pp. 18117\u201318125 (2024)","DOI":"10.1609\/aaai.v38i16.29769"},{"key":"18_CR9","doi-asserted-by":"crossref","unstructured":"Guo, Z., Leng, Y., Wu, Y., Zhao, S., Tan, X.: Prompttts: Controllable text-to-speech with text descriptions. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10096285"},{"key":"18_CR10","unstructured":"Ju, Z., Wang, Y., Shen, K., Tan, X., Xin, D., Yang, D., Liu, Y., Leng, Y., Song, K., Tang, S., et\u00a0al.: Naturalspeech 3: Zero-shot speech synthesis with factorized codec and diffusion models. arXiv preprint arXiv:2403.03100 (2024)"},{"key":"18_CR11","unstructured":"Kang, M., Han, W., Hwang, S.J., Yang, E.: ZET-speech: Zero-shot adaptive emotion-controllable text-to-speech synthesis with diffusion and style-based models, http:\/\/arxiv.org\/abs\/2305.13831"},{"key":"18_CR12","unstructured":"Kim, J., Kong, J., Son, J.: Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In: International Conference on Machine Learning, pp. 5530\u20135540. PMLR (2021)"},{"key":"18_CR13","doi-asserted-by":"crossref","unstructured":"Kong, J., Park, J., Kim, B., Kim, J., Kong, D., Kim, S.: Vits2: Improving quality and efficiency of single-stage text-to-speech with adversarial learning and architecture design. arXiv preprint arXiv:2307.16430 (2023)","DOI":"10.21437\/Interspeech.2023-534"},{"key":"18_CR14","unstructured":"Lee, S.H., Choi, H.Y., Kim, S.B., Lee, S.W.: HierSpeech++: Bridging the gap between semantic and acoustic representation of speech by hierarchical variational inference for zero-shot speech synthesis, http:\/\/arxiv.org\/abs\/2311.12454"},{"key":"18_CR15","doi-asserted-by":"publisher","unstructured":"Lee, Y.H., Cho, N.: PhonMatchNet: Phoneme-guided zero-shot keyword spotting for user-defined keywords. In: Proceedings INTERSPEECH 2023, pp. 3964\u20133968 (2023). https:\/\/doi.org\/10.21437\/Interspeech.2023-597","DOI":"10.21437\/Interspeech.2023-597"},{"key":"18_CR16","unstructured":"Leng, Y., Guo, Z., Shen, K., Tan, X., Ju, Z., Liu, Y., Liu, Y., Yang, D., Zhang, L., Song, K., He, L., Li, X.Y., Zhao, S., Qin, T., Bian, J.: PromptTTS 2: describing and generating voices with text prompt, http:\/\/arxiv.org\/abs\/2309.02285"},{"key":"18_CR17","unstructured":"Li, Y.A., Han, C., Raghavan, V.S., Mischler, G., Mesgarani, N.: StyleTTS 2: Towards human-level text-to-speech through style diffusion and adversarial training with large speech language models"},{"key":"18_CR18","unstructured":"Liu, G., Zhang, Y., Lei, Y., Chen, Y., Wang, R., Li, Z., Xie, L.: PromptStyle: Controllable style transfer for text-to-speech with natural language descriptions, http:\/\/arxiv.org\/abs\/2305.19522"},{"key":"18_CR19","doi-asserted-by":"crossref","unstructured":"Liu, G., Zhang, Y., Lei, Y., Chen, Y., Wang, R., Li, Z., Xie, L.: Promptstyle: Controllable style transfer for text-to-speech with natural language descriptions. arXiv preprint arXiv:2305.19522 (2023)","DOI":"10.21437\/Interspeech.2023-1779"},{"key":"18_CR20","unstructured":"Lyth, D., King, S.: Natural language guidance of high-fidelity text-to-speech with synthetic annotations. arXiv preprint arXiv:2402.01912 (2024)"},{"key":"18_CR21","unstructured":"Qin, Z., Zhao, W., Yu, X., Sun, X.: Openvoice: Versatile instant voice cloning. arXiv preprint arXiv:2312.01479 (2023)"},{"key":"18_CR22","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International conference on machine learning. pp. 8748\u20138763. PMLR (2021)"},{"key":"18_CR23","unstructured":"Wang, C., Chen, S., Wu, Y., Zhang, Z., Zhou, L., Liu, S., Chen, Z., Liu, Y., Wang, H., Li, J., et\u00a0al.: Neural codec language models are zero-shot text to speech synthesizers. arXiv preprint arXiv:2301.02111 (2023)"},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Wang, K., Wu, Q., Song, L., Yang, Z., Wu, W., Qian, C., He, R., Qiao, Y., Loy, C.C.: Mead: A large-scale audio-visual dataset for emotional talking-face generation. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58589-1_42"},{"key":"18_CR25","unstructured":"Yao, J., Yang, Y., Lei, Y., Ning, Z., Hu, Y., Pan, Y., Yin, J., Zhou, H., Lu, H., Xie, L.: PromptVC: Flexible stylistic voice conversion in latent space driven by natural language prompts, http:\/\/arxiv.org\/abs\/2309.09262"},{"key":"18_CR26","unstructured":"Zhang, X., Zhang, D., Li, S., Zhou, Y., Qiu, X.: Speechtokenizer: Unified speech tokenizer for speech large language models. arXiv preprint arXiv:2308.16692 (2023)"},{"key":"18_CR27","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Liu, G., Lei, Y., Chen, Y., Yin, H., Xie, L., Li, Z.: Promptspeaker: Speaker generation based on text descriptions. In: 2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a01\u20137. IEEE (2023)","DOI":"10.1109\/ASRU57964.2023.10389772"},{"key":"18_CR28","doi-asserted-by":"crossref","unstructured":"Zhou, K., Sisman, B., Liu, R., Li, H.: Seen and unseen emotional style transfer for voice conversion with a new emotional speech dataset. In: ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 920\u2013924. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9413391"},{"key":"18_CR29","doi-asserted-by":"publisher","unstructured":"Zhu, X., Lei, Y., Li, T., Zhang, Y., Zhou, H., Lu, H., Xie, L.: METTS: Multilingual emotional text-to-speech by cross-speaker and cross-lingual emotion transfer 32, 1506\u20131518. https:\/\/doi.org\/10.1109\/TASLP.2024.3363444, https:\/\/ieeexplore.ieee.org\/document\/10423864\/","DOI":"10.1109\/TASLP.2024.3363444"},{"key":"18_CR30","unstructured":"Zhu, X., Lei, Y., Song, K., Zhang, Y., Li, T., Xie, L.: Multi-speaker expressive speech synthesis via multiple factors decoupling, http:\/\/arxiv.org\/abs\/2211.10568"},{"key":"18_CR31","unstructured":"Zhu, X., Lv, Y., Lei, Y., Li, T., He, W., Zhou, H., Lu, H., Xie, L.: Vec-tok speech: speech vectorization and tokenization for neural speech generation, http:\/\/arxiv.org\/abs\/2310.07246"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8795-1_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T23:04:33Z","timestamp":1730588673000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8795-1_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,3]]},"ISBN":["9789819787944","9789819787951"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8795-1_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,3]]},"assertion":[{"value":"3 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}