{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T12:37:24Z","timestamp":1756384644961,"version":"3.37.3"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"41-42","license":[{"start":{"date-parts":[[2020,8,15]],"date-time":"2020-08-15T00:00:00Z","timestamp":1597449600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,8,15]],"date-time":"2020-08-15T00:00:00Z","timestamp":1597449600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2020,11]]},"DOI":"10.1007\/s11042-020-09321-7","type":"journal-article","created":{"date-parts":[[2020,8,15]],"date-time":"2020-08-15T10:06:19Z","timestamp":1597485979000},"page":"30205-30233","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Fast Griffin Lim based waveform generation strategy for text-to-speech synthesis"],"prefix":"10.1007","volume":"79","author":[{"given":"Ankit","family":"Sharma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Puneet","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5130-8249","authenticated-orcid":false,"given":"Vikas","family":"Maddukuri","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nagasai","family":"Madamshetti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"K. G.","family":"Kishore","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sahit Sai Sriram","family":"Kavuru","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Balasubramanian","family":"Raman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Partha Pratim","family":"Roy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,8,15]]},"reference":[{"key":"9321_CR1","unstructured":"Aaron A, Bakis R, Eide EM, Hamza WM (2014) Systems and methods for text-to-speech synthesis using spoken example, November 11 2014. US Patent 8,886,538"},{"key":"9321_CR2","unstructured":"Arik SO, Chrzanowski M, Coates A, Diamos G, Gibiansky A, Kang Y, Li X, Miller J, Ng A, Raiman J, et al. (2017) Deep voice: real-time neural text-to-speech. In: Proceedings of the 34th international conference on machine learning (ICML), vol 70, pp 195\u2013204"},{"issue":"1","key":"9321_CR3","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1109\/LSP.2018.2880284","volume":"26","author":"SO Arik","year":"2018","unstructured":"Arik SO, Jun H, Diamos G (2018) Fast spectrogram inversion using multi-head convolutional neural networks. IEEE Signal Process Lett 26(1):94\u201398","journal-title":"IEEE Signal Process Lett"},{"key":"9321_CR4","volume-title":"The Fourier transform and its applications, vol 31999","author":"RN Bracewell","year":"1986","unstructured":"Bracewell RN, Bracewell RN (1986) The Fourier transform and its applications, vol 31999. McGraw-Hill, New York"},{"key":"9321_CR5","doi-asserted-by":"crossref","unstructured":"Braunschweiler N, Gales MJF, Buchholz S (2010) Lightly supervised recognition for automatic alignment of large coherent speech recordings. In: INTERSPEECH","DOI":"10.21437\/Interspeech.2010-611"},{"issue":"2","key":"9321_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2846092","volume":"34","author":"Z Cheng","year":"2016","unstructured":"Cheng Z, Shen J (2016) On effective location-aware music recommendation. ACM Trans Inf Syst (TOIS) 34(2):1\u201332","journal-title":"ACM Trans Inf Syst (TOIS)"},{"key":"9321_CR7","unstructured":"Coorman G, Deprez F, De Bock M, Fackrell J, Leys S, Rutten P, De Moortel J, Schenk A, Van Coile B (2007) Speech synthesis using concatenation of speech waveforms, May 15 2007. US Patent 7,219,060"},{"key":"9321_CR8","first-page":"15236","volume":"7","author":"P Ghate","year":"2017","unstructured":"Ghate P, Shirbahadurkar SD (2017) A survey on methods of tts and various test for evaluating the quality of synthesized speech. Int J Dev Res 7:15236\u201315239","journal-title":"Int J Dev Res"},{"key":"9321_CR9","unstructured":"Gibiansky A, Arik S, Diamos G, Miller J, Peng K, Ping W, Raiman J, Zhou Y (2017) Deep voice 2: multi-speaker neural text-to-speech. In: Advances in neural information processing systems, pp 2962\u20132970"},{"issue":"2","key":"9321_CR10","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin D, Lim J (1984) Signal estimation from modified short-time fourier transform. IEEE Trans Acoust Speech Signal Process 32(2):236\u2013243","journal-title":"IEEE Trans Acoust Speech Signal Process"},{"key":"9321_CR11","unstructured":"Hunt AJ, Black AW (1996) Unit selection in a concatenative speech synthesis system using a large speech database. In: IEEE international conference on acoustics, speech, and signal processing conference proceedings, vol 1, pp 373\u2013376"},{"key":"9321_CR12","unstructured":"Ito K (2017) The lj speech dataset https:\/\/keithito.com\/LJ-Speech-Dataset\/"},{"key":"9321_CR13","unstructured":"Jia Y, Zhang Y, Weiss R, Wang Q, Shen J, Ren F, Nguyen P, Pang R, Moreno IL, Wu Y, et al. (2018) Transfer learning from speaker verification to multispeaker text-to-speech synthesis. In: Advances in neural information processing systems (NeuroIPS), pp 4480\u20134490"},{"key":"9321_CR14","unstructured":"Jones BT, Guthrie DM, Schaefer L, Martin JD (2017) Real-time speech-to-text conversion in an audio conference session, January 31 2017. US Patent 9,560,206"},{"key":"9321_CR15","unstructured":"Kinsella B (2017) Speech synthesis becomes more humanlike. https:\/\/voicebot.ai\/2017\/12\/21\/speech-synthesis-becomes-humanlike\/"},{"key":"9321_CR16","doi-asserted-by":"crossref","unstructured":"Kim S, Hori T, Watanabe S (2017) Joint ctc-attention based end-to-end speech recognition using multi-task learning. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4835\u20134839","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"9321_CR17","unstructured":"Kumar K, Kumar R, de Boissiere T, Gestin L, Teoh WZ, Sotelo J, de Brebisson A, Bengio Y, Courville AC (2019) Melgan: generative adversarial networks for conditional waveform synthesis. In: Advances in neural information processing systems, pp 14881\u201314892"},{"issue":"23","key":"9321_CR18","doi-asserted-by":"publisher","first-page":"24917","DOI":"10.1007\/s11042-016-4122-7","volume":"76","author":"S Lee","year":"2017","unstructured":"Lee S, Chang J-H (2017) Spectral difference for statistical model-based speech enhancement in speech recognition. Springer Multimed Tools Appl 76 (23):24917\u201324929","journal-title":"Springer Multimed Tools Appl"},{"key":"9321_CR19","unstructured":"Levoy M (1992) Volume rendering using the fourier projection-slice theorem. Computer Systems Laboratory, Stanford University"},{"issue":"6","key":"9321_CR20","doi-asserted-by":"publisher","first-page":"8449","DOI":"10.1007\/s11042-016-3414-2","volume":"76","author":"T Malathi","year":"2017","unstructured":"Malathi T, Bhuyan MK (2017) Performance analysis of gabor wavelet for extracting most informative and efficient features. Springer Multimed Tools Appl 76(6):8449\u20138469","journal-title":"Springer Multimed Tools Appl"},{"key":"9321_CR21","doi-asserted-by":"crossref","unstructured":"Masuko T, Tokuda K, Kobayashi T, Imai S (1997) Voice characteristics conversion for hmm-based speech synthesis system. In: 1997 IEEE international conference on acoustics, speech, and signal processing, vol 3, pp 1611\u20131614","DOI":"10.1109\/ICASSP.1997.598807"},{"issue":"1","key":"9321_CR22","doi-asserted-by":"publisher","first-page":"184","DOI":"10.1109\/LSP.2018.2884026","volume":"26","author":"Y Masuyama","year":"2018","unstructured":"Masuyama Y, Yatabe K, Oikawa Y (2018) Griffin\u2013lim like phase recovery via alternating direction method of multipliers. IEEE Signal Process Lett 26 (1):184\u2013188","journal-title":"IEEE Signal Process Lett"},{"key":"9321_CR23","doi-asserted-by":"crossref","unstructured":"Masuyama Y, Yatabe K, Koizumi Y, Oikawa Y, Harada N (2019) Deep griffin\u2013lim iteration. In: IEEE international conference on acoustics speech and signal processing (ICASSP), pp 61\u201365","DOI":"10.1109\/ICASSP.2019.8682744"},{"key":"9321_CR24","doi-asserted-by":"crossref","unstructured":"Mizuno H, Abe M, Hirokawa T (1993) Waveform-based speech synthesis approach with a formant frequency modification. In: IEEE international conference on acoustics, speech, and signal processing, vol 2, pp 195\u2013198","DOI":"10.1109\/ICASSP.1993.319267"},{"issue":"7","key":"9321_CR25","doi-asserted-by":"publisher","first-page":"1877","DOI":"10.1587\/transinf.2015EDP7457","volume":"99","author":"M Morise","year":"2016","unstructured":"Morise M, Yokomori F, Ozawa K (2016) World: a vocoder-based high-quality speech synthesis system for real-time applications. IEICE Trans Inf Syst 99(7):1877\u20131884","journal-title":"IEICE Trans Inf Syst"},{"key":"9321_CR26","doi-asserted-by":"crossref","unstructured":"Oyamada K, Kameoka H, Kaneko T, Tanaka K, Hojo N, Ando H (2018) Generative adversarial network-based approach to signal reconstruction from magnitude spectrogram. In: 26th IEEE European signal processing conference (EUSIPCO), pp 2514\u20132518","DOI":"10.23919\/EUSIPCO.2018.8553396"},{"key":"9321_CR27","doi-asserted-by":"crossref","unstructured":"Perraudin N, Balazs P, S\u00f8ndergaard PL (2013) A fast griffin-lim algorithm. In: IEEE workshop on applications of signal processing to audio and acoustics, pp 1\u20134","DOI":"10.1109\/WASPAA.2013.6701851"},{"key":"9321_CR28","unstructured":"Ping W, Peng K, Gibiansky A, Arik SO, Kannan A, Narang S, Raiman J, Miller J (2017) Deep voice 3: scaling text-to-speech with convolutional sequence learning. arXiv:1710.07654W"},{"key":"9321_CR29","unstructured":"Prahallad K (2016) Speech technology: spectrogram, cepstrum and mel-frequency analysis. https:\/\/archive.org\/details\/SpectrogramCepstrumAndMel-frequency_636522"},{"key":"9321_CR30","doi-asserted-by":"crossref","unstructured":"Prenger R, Valle R, Catanzaro B (2019) Waveglow: a flow-based generative network for speech synthesis. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 3617\u20133621","DOI":"10.1109\/ICASSP.2019.8683143"},{"issue":"7","key":"9321_CR31","doi-asserted-by":"publisher","first-page":"2429","DOI":"10.1109\/78.224251","volume":"41","author":"S Qian","year":"1993","unstructured":"Qian S, Chen D (1993) Discrete gabor transform. IEEE Trans Signal Process 41(7):2429\u20132438","journal-title":"IEEE Trans Signal Process"},{"issue":"4","key":"9321_CR32","first-page":"650","volume":"82","author":"PL Salza","year":"1996","unstructured":"Salza PL, Foti E, Nebbia L, Oreglia M (1996) Mos and pair comparison combined methods for quality evaluation of text-to-speech systems. Acta Acustica United with Acustica 82(4):650\u2013656","journal-title":"Acta Acustica United with Acustica"},{"key":"9321_CR33","doi-asserted-by":"crossref","unstructured":"Shen J, Pang R, Weiss RJ, Schuster M, Jaitly N, Yang Z, Chen Z, Zhang Y, Wang Y, Skerrv-Ryan RJ, et al. (2018) Natural tts synthesis by conditioning wavenet on mel spectrogram predictions. In: International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 4779\u20134783","DOI":"10.1109\/ICASSP.2018.8461368"},{"issue":"6","key":"9321_CR34","doi-asserted-by":"publisher","first-page":"849","DOI":"10.1109\/TASSP.1987.1165220","volume":"35","author":"HV Sorensen","year":"1987","unstructured":"Sorensen HV, Jones D, Heideman M, Burrus C (1987) Real-valued fast fourier transform algorithms. IEEE Trans Acoust Speech Signal Process 35(6):849\u2013863","journal-title":"IEEE Trans Acoust Speech Signal Process"},{"key":"9321_CR35","unstructured":"Sotelo J, Mehri S, Kumar K, Santos JF, Kastner K, Courville A, Bengio Y (2017) Char2wav: end-to-end speech synthesis"},{"key":"9321_CR36","unstructured":"Sysko (2013) Tatoeba speech dataset. https:\/\/tatoeba.org\/eng\/"},{"key":"9321_CR37","unstructured":"Taigman Y, Wolf L, Polyak A, Nachmani E (2017) Voiceloop: voice fitting and synthesis via a phonological loop. arXiv:1707.06588"},{"issue":"5","key":"9321_CR38","doi-asserted-by":"publisher","first-page":"1234","DOI":"10.1109\/JPROC.2013.2251852","volume":"101","author":"K Tokuda","year":"2013","unstructured":"Tokuda K, Nankaku Y, Toda T, Zen H, Yamagishi J, Oura K (2013) Speech synthesis based on hidden markov models. Proc IEEE 101 (5):1234\u20131252","journal-title":"Proc IEEE"},{"key":"9321_CR39","doi-asserted-by":"crossref","unstructured":"Tokuday K, Zen H (2015) Directly modeling speech waveforms by neural networks for statistical parametric speech synthesis. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4215\u20134219","DOI":"10.1109\/ICASSP.2015.7178765"},{"key":"9321_CR40","unstructured":"van den Oord A, Dieleman S, Zen H, Simonyan K, Vinyals O, Graves A, Kalchbrenner N, Senior A, Kavukcuoglu K (2016) Wavenet: a generative model for raw audio. arXiv:1609.03499"},{"key":"9321_CR41","unstructured":"van den Oord A, Li Y, Babuschkin I, Simonyan K, Vinyals O, Kavukcuoglu K, van den Driessche G, Lockhart E, Cobo LC, Stimberg F, et al. (2017) Parallel wavenet: fast high-fidelity speech synthesis. arXiv:1711.10433"},{"key":"9321_CR42","doi-asserted-by":"crossref","unstructured":"Wang Y, Skerry-Ryan RJ, Stanton D, Wu Y, Weiss RJ, Jaitly N, Yang Z, Xiao Y, Chen Z, Bengio S, et al. (2017) Tacotron: towards end-to-end speech synthesis. arXiv:1703.10135","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"9321_CR43","unstructured":"Wu Y, Schuster M, Chen Z, Le QV, Norouzi M, Macherey W, Krikun M, Cao Y, Gao Q, Macherey K, et al. (2016) Google\u2019s neural machine translation system: Bridging the gap between human and machine translation. arXiv:1609.08144"},{"key":"9321_CR44","doi-asserted-by":"publisher","first-page":"984","DOI":"10.1109\/TASL.2010.2045237","volume":"18","author":"J Yamagishi","year":"2009","unstructured":"Yamagishi J, Usabaev B, King S, Watts O, Dines J, Tian J, Hu R, Guan Y, Oura K, Tokuda K, Karhila R, Kurimo M (2009) Thousands of voices for hmm-based speech synthesis\u2013analysis and application of tts systems built on various asr corpora. IEEE Trans Audio Speech Lang Process 18:984\u20131004","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"9321_CR45","unstructured":"Yu F, Koltun V (2015) Multi-scale context aggregation by dilated convolutions. arXiv:1511.07122"},{"issue":"11","key":"9321_CR46","doi-asserted-by":"publisher","first-page":"1039","DOI":"10.1016\/j.specom.2009.04.004","volume":"51","author":"H Zen","year":"2009","unstructured":"Zen H, Tokuda K, Black AW (2009) Statistical parametric speech synthesis. Elsevier Speech Commun 51(11):1039\u20131064","journal-title":"Elsevier Speech Commun"},{"key":"9321_CR47","doi-asserted-by":"publisher","first-page":"60478","DOI":"10.1109\/ACCESS.2018.2872060","volume":"6","author":"Y Zhao","year":"2018","unstructured":"Zhao Y, Takaki S, Luong H -T, Yamagishi J, Saito D, Minematsu N (2018) Wasserstein gan and waveform loss-based acoustic model training for multi-speaker text-to-speech synthesis systems using a wavenet vocoder. IEEE Access 6:60478\u201360488","journal-title":"IEEE Access"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-09321-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-020-09321-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-09321-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,7]],"date-time":"2022-11-07T02:00:50Z","timestamp":1667786450000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-020-09321-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,8,15]]},"references-count":47,"journal-issue":{"issue":"41-42","published-print":{"date-parts":[[2020,11]]}},"alternative-id":["9321"],"URL":"https:\/\/doi.org\/10.1007\/s11042-020-09321-7","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"type":"print","value":"1380-7501"},{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2020,8,15]]},"assertion":[{"value":"22 October 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 June 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 July 2020","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 August 2020","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}