{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T00:53:41Z","timestamp":1743123221223,"version":"3.40.3"},"publisher-location":"Cham","reference-count":37,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030878016"},{"type":"electronic","value":"9783030878023"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-87802-3_33","type":"book-chapter","created":{"date-parts":[[2021,9,21]],"date-time":"2021-09-21T23:36:52Z","timestamp":1632267412000},"page":"360-371","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Assessing Speaker Interpolation in Neural Text-to-Speech"],"prefix":"10.1007","author":[{"given":"Roman","family":"Korostik","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Javier","family":"Latorre","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sivanand","family":"Achanta","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yannis","family":"Stylianou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,9,22]]},"reference":[{"key":"33_CR1","unstructured":"Athar, S., Burnaev, E., Lempitsky, V.: Latent convolutional models. In: 7th International Conference on Learning Representations, ICLR 2019 (2019)"},{"key":"33_CR2","doi-asserted-by":"crossref","unstructured":"Battenberg, E., et al.: Location-relative attention mechanisms for robust long-form speech synthesis. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6194\u20136198. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054106"},{"key":"33_CR3","unstructured":"Chen, Y., et al.: Sample efficient adaptive text-to-speech. arXiv preprint arXiv:1809.10460 (2018)"},{"key":"33_CR4","doi-asserted-by":"crossref","unstructured":"Cooper, E., et al.: Zero-shot multi-speaker text-to-speech with state-of-the-art neural speaker embeddings. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6184\u20136188. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054535"},{"key":"33_CR5","unstructured":"Denil, M., Shakibi, B., Dinh, L., Ranzato, M., De Freitas, N.: Predicting parameters in deep learning. arXiv preprint arXiv:1306.0543 (2013)"},{"key":"33_CR6","unstructured":"Donahue, J., Dieleman, S., Bi\u0144kowski, M., Elsen, E., Simonyan, K.: End-to-end adversarial text-to-speech. arXiv preprint arXiv:2006.03575 (2020)"},{"issue":"7","key":"33_CR7","doi-asserted-by":"publisher","first-page":"e11","DOI":"10.23915\/distill.00011","volume":"3","author":"V Dumoulin","year":"2018","unstructured":"Dumoulin, V., et al.: Feature-wise transformations. Distill 3(7), e11 (2018)","journal-title":"Distill"},{"key":"33_CR8","unstructured":"Dumoulin, V., Shlens, J., Kudlur, M.: A learned representation for artistic style. arXiv preprint arXiv:1610.07629 (2016)"},{"key":"33_CR9","unstructured":"Habib, R., et al.: Semi-supervised generative modeling for controllable speech synthesis. arXiv preprint arXiv:1910.01709 (2019)"},{"key":"33_CR10","unstructured":"Han, Y., Huang, G., Song, S., Yang, L., Wang, H., Wang, Y.: Dynamic neural networks: a survey. arXiv preprint arXiv:2102.04906 (2021)"},{"key":"33_CR11","doi-asserted-by":"crossref","unstructured":"Hsu, W.N., et al.: Disentangling correlated speaker and noise for speech synthesis via data augmentation and adversarial factorization. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5901\u20135905. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8683561"},{"key":"33_CR12","unstructured":"Hsu, W.N., et al.: Hierarchical generative modeling for controllable speech synthesis. arXiv preprint arXiv:1810.07217 (2018)"},{"key":"33_CR13","doi-asserted-by":"crossref","unstructured":"Hu, Q., Marchi, E., Winarsky, D., Stylianou, Y., Naik, D., Kajarekar, S.: Neural text-to-speech adaptation from low quality public recordings. In: Speech Synthesis Workshop, vol. 10 (2019)","DOI":"10.21437\/SSW.2019-5"},{"key":"33_CR14","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: accelerating deep network training by reducing internal covariate shift. In: International Conference on Machine Learning, pp. 448\u2013456. PMLR (2015)"},{"key":"33_CR15","unstructured":"Jia, Y., et al.: Transfer learning from speaker verification to multispeaker text-to-speech synthesis. arXiv preprint arXiv:1806.04558 (2018)"},{"key":"33_CR16","unstructured":"Kalchbrenner, N., et al.: Efficient neural audio synthesis. In: International Conference on Machine Learning, pp. 2410\u20132419. PMLR (2018)"},{"key":"33_CR17","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational Bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"33_CR18","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: ImageNet classification with deep convolutional neural networks. In: Pereira, F., Burges, C.J.C., Bottou, L., Weinberger, K.Q. (eds.) Advances in Neural Information Processing Systems, vol. 25. Curran Associates, Inc. (2012). https:\/\/proceedings.neurips.cc\/paper\/2012\/file\/c399862d3b9d6b76c8436e924a68c45b-Paper.pdf"},{"key":"33_CR19","doi-asserted-by":"crossref","unstructured":"Ma, N., Zhang, X., Huang, J., Sun, J.: WeightNet: revisiting the design space of weight networks. arXiv preprint arXiv:2007.11823 (2020)","DOI":"10.1007\/978-3-030-58555-6_46"},{"key":"33_CR20","doi-asserted-by":"crossref","unstructured":"Moss, H.B., Aggarwal, V., Prateek, N., Gonz\u00e1lez, J., Barra-Chicote, R.: BOFFIN TTS: few-shot speaker adaptation by Bayesian optimization. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7639\u20137643. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054301"},{"issue":"2","key":"33_CR21","doi-asserted-by":"publisher","first-page":"164","DOI":"10.1016\/j.specom.2009.09.004","volume":"52","author":"M Pucher","year":"2010","unstructured":"Pucher, M., Schabus, D., Yamagishi, J., Neubarth, F., Strom, V.: Modeling and interpolation of Austrian German and Viennese dialect in hmm-based speech synthesis. Speech Commun. 52(2), 164\u2013179 (2010)","journal-title":"Speech Commun."},{"key":"33_CR22","doi-asserted-by":"crossref","unstructured":"Raitio, T., Rasipuram, R., Castellani, D.: Controllable neural text-to-speech synthesis using intuitive prosodic features. arXiv preprint arXiv:2009.06775 (2020)","DOI":"10.21437\/Interspeech.2020-2861"},{"key":"33_CR23","unstructured":"Salimans, T., Kingma, D.P.: Weight normalization: a simple reparameterization to accelerate training of deep neural networks. In: Proceedings of the 30th International Conference on Neural Information Processing Systems, pp. 901\u2013909 (2016)"},{"issue":"1","key":"33_CR24","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1162\/neco.1992.4.1.131","volume":"4","author":"J Schmidhuber","year":"1992","unstructured":"Schmidhuber, J.: Learning to control fast-weight memories: an alternative to dynamic recurrent networks. Neural Comput. 4(1), 131\u2013139 (1992)","journal-title":"Neural Comput."},{"key":"33_CR25","doi-asserted-by":"crossref","unstructured":"Shechtman, S., Sorin, A.: Sequence to sequence neural speech synthesis with prosody modification capabilities. In: Proceedings of the 10th ISCA Speech Synthesis Workshop, pp. 275\u2013280 (2019)","DOI":"10.21437\/SSW.2019-49"},{"key":"33_CR26","doi-asserted-by":"crossref","unstructured":"Shen, J., et al.: Natural TTS synthesis by conditioning WaveNet on Mel spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"33_CR27","doi-asserted-by":"crossref","unstructured":"Tachibana, M., Yamagishi, J., Onishi, K., Masuko, T., Kobayashi, T.: HMM-based speech synthesis with various speaking styles using model interpolation. In: Speech Prosody 2004, International Conference (2004)","DOI":"10.21437\/SpeechProsody.2004-94"},{"key":"33_CR28","doi-asserted-by":"crossref","unstructured":"Um, S.Y., Oh, S., Byun, K., Jang, I., Ahn, C., Kang, H.G.: Emotional speech synthesis with rich and granularized control. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7254\u20137258. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053732"},{"key":"33_CR29","unstructured":"Wang, Y., et al.: Uncovering latent style factors for expressive speech synthesis. arXiv preprint arXiv:1711.00520 (2017)"},{"key":"33_CR30","unstructured":"Wang, Y., et al.: Style tokens: unsupervised style modeling, control and transfer in end-to-end speech synthesis. In: International Conference on Machine Learning, pp. 5180\u20135189. PMLR (2018)"},{"key":"33_CR31","unstructured":"Wu, F., Fan, A., Baevski, A., Dauphin, Y.N., Auli, M.: Pay less attention with lightweight and dynamic convolutions. arXiv preprint arXiv:1901.10430 (2019)"},{"key":"33_CR32","doi-asserted-by":"publisher","unstructured":"Xiao, Y., He, L., Ming, H., Soong, F.K.: Improving prosody with linguistic and BERT derived features in multi-speaker based mandarin Chinese neural TTS. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6704\u20136708 (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9054337","DOI":"10.1109\/ICASSP40776.2020.9054337"},{"key":"33_CR33","unstructured":"Yamagishi, J., Veaux, C., MacDonald, K., et al.: CSTR VCTK corpus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92) (2019)"},{"key":"33_CR34","doi-asserted-by":"crossref","unstructured":"Yoshimura, T., Masuko, T., Tokuda, K., Kobayashi, T., Kitamura, T.: Speaker interpolation in HMM-based speech synthesis system. In: Fifth European Conference on Speech Communication and Technology (1997)","DOI":"10.21437\/Eurospeech.1997-655"},{"key":"33_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"818","DOI":"10.1007\/978-3-319-10590-1_53","volume-title":"Computer Vision \u2013 ECCV 2014","author":"MD Zeiler","year":"2014","unstructured":"Zeiler, M.D., Fergus, R.: Visualizing and understanding convolutional networks. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8689, pp. 818\u2013833. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10590-1_53"},{"key":"33_CR36","doi-asserted-by":"crossref","unstructured":"Zen, H., et al.: LibriTTS: a corpus derived from LibriSpeech for text-to-speech. arXiv preprint arXiv:1904.02882 (2019)","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"33_CR37","doi-asserted-by":"crossref","unstructured":"Zhang, Y.J., Pan, S., He, L., Ling, Z.H.: Learning latent representations for style control and transfer in end-to-end speech synthesis. In: ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6945\u20136949. IEEE (2019)","DOI":"10.1109\/ICASSP.2019.8683623"}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-87802-3_33","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,8]],"date-time":"2024-09-08T17:39:57Z","timestamp":1725817197000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-87802-3_33"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030878016","9783030878023"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-87802-3_33","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"22 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SPECOM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Speech and Computer","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"St Petersburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Russia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 September 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"specom2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/specom.nw.ru\/2021\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"163","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"74","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"45% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.5","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held online due to the COVID-19 pandemic.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}