{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T13:13:10Z","timestamp":1742994790145,"version":"3.40.3"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031476334"},{"type":"electronic","value":"9783031476341"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-47634-1_31","type":"book-chapter","created":{"date-parts":[[2023,11,4]],"date-time":"2023-11-04T20:01:39Z","timestamp":1699128099000},"page":"415-427","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["PauseSpeech: Natural Speech Synthesis via\u00a0Pre-trained Language Model and\u00a0Pause-Based Prosody Modeling"],"prefix":"10.1007","author":[{"given":"Ji-Sang","family":"Hwang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sang-Hoon","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Seong-Whan","family":"Lee","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,5]]},"reference":[{"key":"31_CR1","unstructured":"Abbas, A., et al.: Expressive, Variable, and Controllable Duration Modelling in TTS. arXiv preprint arXiv:2206.14165 (2022)"},{"key":"31_CR2","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., Auli, M.: wav2vec 2.0: a framework for self-supervised learning of speech representations. In: Advances in Neural Information Processing Systems, vol. 33, pp. 12449\u201312460 (2020)"},{"key":"31_CR3","unstructured":"Braunschweiler, N., Chen, L.: Automatic detection of inhalation breath pauses for improved pause modelling in HMM-TTS. In: SSW, vol. 8, pp. 1\u20136 (2013)"},{"key":"31_CR4","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: Pre-Training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"31_CR5","unstructured":"Donahue, J., Dieleman, S., Bi\u0144kowski, M., Elsen, E., Simonyan, K.: End-to-End Adversarial Text-to-Speech. arXiv preprint arXiv:2006.03575 (2020)"},{"key":"31_CR6","doi-asserted-by":"crossref","unstructured":"Elmers, M., Werner, R., Muhlack, B., M\u00f6bius, B., Trouvain, J.: Take a breath: respiratory sounds improve recollection in synthetic speech. In: Interspeech, pp. 3196\u20133200 (2021)","DOI":"10.21437\/Interspeech.2021-1496"},{"key":"31_CR7","doi-asserted-by":"crossref","unstructured":"Futamata, K., Park, B., Yamamoto, R., Tachibana, K.: Phrase Break Prediction with Bidirectional Encoder Representations in Japanese Text-to-Speech Synthesis. arXiv preprint arXiv:2104.12395 (2021)","DOI":"10.21437\/Interspeech.2021-252"},{"key":"31_CR8","unstructured":"Goldberg, Y.: Assessing BERT\u2019s Syntactic Abilities. arXiv preprint arXiv:1901.05287 (2019)"},{"issue":"11","key":"31_CR9","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow, I., et al.: Generative adversarial networks. Commun. ACM 63(11), 139\u2013144 (2020)","journal-title":"Commun. ACM"},{"key":"31_CR10","doi-asserted-by":"crossref","unstructured":"Hayashi, T., Watanabe, S., Toda, T., Takeda, K., Toshniwal, S., Livescu, K.: Pre-trained text embeddings for enhanced text-to-speech synthesis. In: Interspeech, pp. 4430\u20134434 (2019)","DOI":"10.21437\/Interspeech.2019-3177"},{"key":"31_CR11","unstructured":"Hewitt, J., Manning, C.D.: A structural probe for finding syntax in word representations. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 4129\u20134138 (2019)"},{"key":"31_CR12","doi-asserted-by":"crossref","unstructured":"Hida, R., Hamada, M., Kamada, C., Tsunoo, E., Sekiya, T., Kumakura, T.: Polyphone disambiguation and accent prediction using pre-trained language models in Japanese TTS front-end. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7132\u20137136. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746212"},{"key":"31_CR13","first-page":"8067","volume":"33","author":"J Kim","year":"2020","unstructured":"Kim, J., Kim, S., Kong, J., Yoon, S.: Glow-TTS: a generative flow for text-to-speech via monotonic alignment search. Adv. Neural. Inf. Process. Syst. 33, 8067\u20138077 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"31_CR14","unstructured":"Kim, J., Kong, J., Son, J.: Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech. In: International Conference on Machine Learning, pp. 5530\u20135540. PMLR (2021)"},{"key":"31_CR15","doi-asserted-by":"crossref","unstructured":"Kim, J.H., Lee, S.H., Lee, J.H., Lee, S.W.: Fre-GAN: adversarial frequency-consistent audio synthesis. In: 22nd Annual Conference of the International Speech Communication Association, INTERSPEECH 2021, pp. 3246\u20133250. International Speech Communication Association (2021)","DOI":"10.21437\/Interspeech.2021-845"},{"issue":"1","key":"31_CR16","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1109\/TNSRE.2019.2946625","volume":"28","author":"KT Kim","year":"2019","unstructured":"Kim, K.T., Guan, C., Lee, S.W.: A subject-transfer framework based on single-trial EMG analysis using convolutional neural networks. IEEE Trans. Neural Syst. Rehabil. Eng. 28(1), 94\u2013103 (2019)","journal-title":"IEEE Trans. Neural Syst. Rehabil. Eng."},{"key":"31_CR17","doi-asserted-by":"crossref","unstructured":"Klimkov, V., et al.: Phrase break prediction for long-form reading TTS: exploiting text structure information. In: Proceedings of Interspeech 2017, pp. 1064\u20131068 (2017)","DOI":"10.21437\/Interspeech.2017-419"},{"key":"31_CR18","first-page":"17022","volume":"33","author":"J Kong","year":"2020","unstructured":"Kong, J., Kim, J., Bae, J.: HiFi-GAN: generative adversarial networks for efficient and high fidelity speech synthesis. Adv. Neural. Inf. Process. Syst. 33, 17022\u201317033 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"31_CR19","doi-asserted-by":"crossref","unstructured":"\u0141a\u0144cucki, A.: FastPitch: parallel text-to-speech with pitch prediction. In: ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6588\u20136592. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9413889"},{"key":"31_CR20","doi-asserted-by":"crossref","unstructured":"Lee, J.H., Lee, S.H., Kim, J.H., Lee, S.W.: PVAE-TTS: adaptive text-to-speech via progressive style adaptation. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6312\u20136316. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9747388"},{"key":"31_CR21","first-page":"16624","volume":"35","author":"SH Lee","year":"2022","unstructured":"Lee, S.H., Kim, S.B., Lee, J.H., Song, E., Hwang, M.J., Lee, S.W.: HierSpeech: bridging the gap between text and speech by hierarchical variational inference using self-supervised representations for speech synthesis. Adv. Neural. Inf. Process. Syst. 35, 16624\u201316636 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"31_CR22","doi-asserted-by":"crossref","unstructured":"Lee, S.H., Yoon, H.W., Noh, H.R., Kim, J.H., Lee, S.W.: Multi-SpectroGAN: high-diversity and high-fidelity spectrogram generation with adversarial style combination for speech synthesis. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 13198\u201313206 (2021)","DOI":"10.1609\/aaai.v35i14.17559"},{"key":"31_CR23","doi-asserted-by":"crossref","unstructured":"Lewis, M., et al.: BART: Denoising Sequence-to-Sequence Pre-Training for Natural Language Generation, Translation, and Comprehension. arXiv preprint arXiv:1910.13461 (2019)","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"31_CR24","doi-asserted-by":"crossref","unstructured":"Liu, N.F., Gardner, M., Belinkov, Y., Peters, M.E., Smith, N.A.: Linguistic knowledge and transferability of contextual representations. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 1073\u20131094 (2019)","DOI":"10.18653\/v1\/N19-1112"},{"key":"31_CR25","unstructured":"Liu, Y., et al.: RoBERTa: A Robustly Optimized BERT Pretraining Approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"31_CR26","unstructured":"Loshchilov, I., Hutter, F.: Decoupled Weight Decay Regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"31_CR27","doi-asserted-by":"crossref","unstructured":"Makarov, P., et al.: Simple and Effective Multi-Sentence TTS with Expressive and Coherent Prosody. arXiv preprint arXiv:2206.14643 (2022)","DOI":"10.21437\/Interspeech.2022-379"},{"key":"31_CR28","doi-asserted-by":"crossref","unstructured":"McAuliffe, M., Socolof, M., Mihuc, S., Wagner, M., Sonderegger, M.: Montreal forced aligner: trainable text-speech alignment using kaldi. In: Interspeech, vol. 2017, pp. 498\u2013502 (2017)","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"31_CR29","doi-asserted-by":"crossref","unstructured":"Oh, H.S., Lee, S.H., Lee, S.W.: DiffProsody: Diffusion-based Latent Prosody Generation for Expressive Speech Synthesis with Prosody Conditional Adversarial Training. arXiv preprint arXiv:2307.16549 (2023)","DOI":"10.1109\/TASLP.2024.3395994"},{"key":"31_CR30","unstructured":"Ren, Y., et al.: FastSpeech 2: fast and high-quality end-to-end text to speech. In: International Conference on Learning Representations (2021)"},{"key":"31_CR31","doi-asserted-by":"crossref","unstructured":"Ren, Y., et al.: ProsoSpeech: enhancing prosody with quantized vector pre-training in text-to-speech. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7577\u20137581. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746883"},{"key":"31_CR32","first-page":"13963","volume":"34","author":"Y Ren","year":"2021","unstructured":"Ren, Y., Liu, J., Zhao, Z.: PortaSpeech: portable and high-quality generative text-to-speech. Adv. Neural. Inf. Process. Syst. 34, 13963\u201313974 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"31_CR33","unstructured":"Ren, Y., et al.: FastSpeech: fast, robust and controllable text to speech. In: Proceedings of the 33rd International Conference on Neural Information Processing Systems, pp. 3171\u20133180 (2019)"},{"key":"31_CR34","doi-asserted-by":"crossref","unstructured":"Ren, Y., Tan, X., Qin, T., Zhao, Z., Liu, T.Y.: Revisiting over-smoothness in text to speech. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 8197\u20138213 (2022)","DOI":"10.18653\/v1\/2022.acl-long.564"},{"key":"31_CR35","unstructured":"Rezende, D., Mohamed, S.: Variational inference with normalizing flows. In: International Conference on Machine Learning, pp. 1530\u20131538. PMLR (2015)"},{"key":"31_CR36","doi-asserted-by":"publisher","first-page":"842","DOI":"10.1162\/tacl_a_00349","volume":"8","author":"A Rogers","year":"2021","unstructured":"Rogers, A., Kovaleva, O., Rumshisky, A.: A primer in BERTology: what we know about how BERT works. Trans. Assoc. Comput. Linguist. 8, 842\u2013866 (2021)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"31_CR37","doi-asserted-by":"crossref","unstructured":"Seshadri, S., Raitio, T., Castellani, D., Li, J.: Emphasis Control for Parallel Neural TTS. arXiv preprint arXiv:2110.03012 (2021)","DOI":"10.21437\/Interspeech.2022-411"},{"key":"31_CR38","doi-asserted-by":"crossref","unstructured":"Shaw, P., Uszkoreit, J., Vaswani, A.: Self-attention with Relative Position Representations. arXiv preprint arXiv:1803.02155 (2018)","DOI":"10.18653\/v1\/N18-2074"},{"key":"31_CR39","doi-asserted-by":"crossref","unstructured":"Shen, J., et al.: Natural TTS synthesis by conditioning wavenet on mel spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783. IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"31_CR40","doi-asserted-by":"crossref","unstructured":"Sz\u00e9kely, \u00c9., Henter, G.E., Beskow, J., Gustafson, J.: Breathing and speech planning in spontaneous speech synthesis. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7649\u20137653. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9054107"},{"key":"31_CR41","doi-asserted-by":"publisher","first-page":"68","DOI":"10.1016\/j.media.2018.01.002","volume":"45","author":"KH Thung","year":"2018","unstructured":"Thung, K.H., Yap, P.T., Adeli, E., Lee, S.W., Shen, D., Initiative, A.D.N., et al.: Conversion and time-to-conversion predictions of mild cognitive impairment using low-rank affinity pursuit denoising and matrix completion. Med. Image Anal. 45, 68\u201382 (2018)","journal-title":"Med. Image Anal."},{"key":"31_CR42","unstructured":"Veaux, C., Yamagishi, J., MacDonald, K., et al.: Superseded-CSTR VCTK Corpus: English Multi-Speaker Corpus for CSTR Voice Cloning Toolkit (2016)"},{"key":"31_CR43","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Tacotron: towards end-to-end speech synthesis. In: Proceedings of Interspeech 2017, pp. 4006\u20134010 (2017)","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"31_CR44","doi-asserted-by":"crossref","unstructured":"Wu, J., Luan, J.: Adversarially Trained Multi-Singer Sequence-to-Sequence Singing Synthesizer. arXiv preprint arXiv:2006.10317 (2020)","DOI":"10.21437\/Interspeech.2020-1109"},{"key":"31_CR45","doi-asserted-by":"crossref","unstructured":"Xu, G., Song, W., Zhang, Z., Zhang, C., He, X., Zhou, B.: Improving prosody modelling with cross-utterance BERT embeddings for end-to-end speech synthesis. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6079\u20136083. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9414102"},{"key":"31_CR46","doi-asserted-by":"crossref","unstructured":"Yang, D., Koriyama, T., Saito, Y., Saeki, T., Xin, D., Saruwatari, H.: Duration-Aware Pause Insertion Using Pre-Trained Language Model for Multi-Speaker Text-to-Speech. arXiv preprint arXiv:2302.13652 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096402"},{"key":"31_CR47","doi-asserted-by":"crossref","unstructured":"Ye, Z., Zhao, Z., Ren, Y., Wu, F.: SyntaSpeech: Syntax-Aware Generative Adversarial Text-to-Speech. arXiv preprint arXiv:2204.11792 (2022)","DOI":"10.24963\/ijcai.2022\/620"},{"issue":"3","key":"31_CR48","doi-asserted-by":"publisher","first-page":"631","DOI":"10.1109\/TASLP.2019.2892235","volume":"27","author":"JX Zhang","year":"2019","unstructured":"Zhang, J.X., Ling, Z.H., Liu, L.J., Jiang, Y., Dai, L.R.: Sequence-to-sequence acoustic modeling for voice conversion. IEEE\/ACM Trans. Audio Speech Lang. Process. 27(3), 631\u2013644 (2019)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-47634-1_31","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T11:29:33Z","timestamp":1730460573000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-47634-1_31"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031476334","9783031476341"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-47634-1_31","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"5 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ACPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Asian Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kitakyushu","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Japan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 November 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"acpr2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ericlab.org\/acpr2023\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"164","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"93","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"57% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}