{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T15:31:27Z","timestamp":1778599887550,"version":"3.51.4"},"publisher-location":"Cham","reference-count":23,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783031209796","type":"print"},{"value":"9783031209802","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-20980-2_7","type":"book-chapter","created":{"date-parts":[[2022,11,12]],"date-time":"2022-11-12T19:03:09Z","timestamp":1668279789000},"page":"64-74","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["An Initial Study on Birdsong Re-synthesis Using Neural Vocoders"],"prefix":"10.1007","author":[{"given":"Rhythm Rajiv","family":"Bhatia","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tomi H.","family":"Kinnunen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,11,10]]},"reference":[{"key":"7_CR1","unstructured":"xeno-canto \u2013 sharing bird sounds from around the world (2017). www.xeno-canto.org\/. Accessed 11 Mar 2021"},{"key":"7_CR2","doi-asserted-by":"crossref","unstructured":"Amador, A., Mindlin, G.B.: Synthetic birdsongs as a tool to induce, and iisten to, replay activity in sleeping birds. Front. Neurosci. 15, 835 (2021)","DOI":"10.3389\/fnins.2021.647978"},{"key":"7_CR3","doi-asserted-by":"publisher","unstructured":"Bonada, J., Lachlan, R., Blaauw, M.: Bird song synthesis based on hidden Markov models. In: Interspeech 2016, pp. 2582\u20132586 (2016). https:\/\/doi.org\/10.21437\/Interspeech.2016-1110","DOI":"10.21437\/Interspeech.2016-1110"},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Dunbar, E., Algayres, R., Karadayi, J., Bernard, et al.: The zero resource speech challenge 2019: TTS without T. arXiv preprint arXiv:1904.11469 (2019)","DOI":"10.21437\/Interspeech.2019-2904"},{"key":"7_CR5","unstructured":"Engel, J., Resnick, C., Roberts, A., Dieleman, et al.: Neural audio synthesis of musical notes with wavenet autoencoders. In: International Conference on Machine Learning, pp. 1068\u20131077. PMLR (2017)"},{"key":"7_CR6","unstructured":"Goodfellow, I.J., et al.: Generative adversarial nets. In: Proceedings of the NIPS. pp. 2672\u20132680 (2014). http:\/\/proceedings.neurips.cc\/paper\/2014\/hash\/5ca3e9b122f61f8f06494c97b1afccf3-Abstract.html"},{"key":"7_CR7","doi-asserted-by":"publisher","unstructured":"Gutscher, L., Pucher, M., Lozo, C., Hoeschele, M., C. Mann, D.: Statistical parametric synthesis of budgerigar songs. In: Proceedings of the 10th ISCA Speech Synthesis Workshop, pp. 127\u2013131 (2019). https:\/\/doi.org\/10.21437\/SSW.2019-23","DOI":"10.21437\/SSW.2019-23"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Haque, A., Guo, M., Verma, P.: Conditional end-to-end audio transforms. arXiv preprint arXiv:1804.00047 (2018)","DOI":"10.21437\/Interspeech.2018-38"},{"key":"7_CR9","unstructured":"Imai, S., et al.: Speech signal processing toolkit (sptk) (2009)"},{"key":"7_CR10","doi-asserted-by":"publisher","unstructured":"Kawahara, H., Morise, M., Takahashi, T., Nisimura, R., Irino, T., Banno, H.: Tandem-STRAIGHT: a temporally stable power spectral representation for periodic signals and applications to interference-free spectrum, f0, and aperiodicity estimation. In: Proceedings of the IEEE ICASSP, pp. 3933\u20133936 (2008). https:\/\/doi.org\/10.1109\/ICASSP.2008.4518514","DOI":"10.1109\/ICASSP.2008.4518514"},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Kawahara, H., Masuda-Katsuse, I., de Cheveign\u00e9, A.: Restructuring speech representations using a pitch-adaptive time-frequency smoothing and an instantaneous-frequency-based F0 extraction: possible role of a repetitive structure in sounds. Speech Commun. 27(3\u20134), 187\u2013207 (1999)","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"7_CR12","doi-asserted-by":"publisher","unstructured":"Moore, R.K.: A real-time parametric general-purpose mammalian vocal synthesiser. In: Interspeech 2016, pp. 2636\u20132640. ISCA (2016). https:\/\/doi.org\/10.21437\/Interspeech.2016-841","DOI":"10.21437\/Interspeech.2016-841"},{"key":"7_CR13","doi-asserted-by":"publisher","first-page":"57","DOI":"10.1016\/j.specom.2016.09.001","volume":"84","author":"M Morise","year":"2016","unstructured":"Morise, M.: D4C, a band-aperiodicity estimator for high-quality speech synthesis. Speech Commun. 84, 57\u201365 (2016)","journal-title":"Speech Commun."},{"key":"7_CR14","doi-asserted-by":"publisher","unstructured":"Morise, M., Yokomori, F., Ozawa, K.: WORLD: a vocoder-based high-quality speech synthesis system for real-time applications. IEICE Trans. Inf. Syst. 99-D(7), 1877\u20131884 (2016). https:\/\/doi.org\/10.1587\/transinf.2015EDP7457","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"7_CR15","unstructured":"van den Oord, A., et al.: Wavenet: a generative model for raw audio. In: The 9th ISCA Speech Synthesis Workshop. Sunnyvale, CA, USA (2016)"},{"key":"7_CR16","doi-asserted-by":"publisher","unstructured":"O\u2019Reilly, C., Marples, N.M., Kelly, D.J., Harte, N.: YIN-bird: improved pitch tracking for bird vocalisations. In: Interspeech, pp. 2641\u20132645. ISCA (2016). https:\/\/doi.org\/10.21437\/Interspeech.2016-90","DOI":"10.21437\/Interspeech.2016-90"},{"key":"7_CR17","unstructured":"Robitza, W.: ffmpeg tool (2015). https:\/\/github.com\/slhck\/ffmpeg-normalize. Accessed 11 March 2021"},{"key":"7_CR18","unstructured":"Salimans, T., Kingma, D.P.: Weight normalization: a simple reparameterization to accelerate training of deep neural networks. arXiv preprint arXiv:1602.07868 (2016)"},{"key":"7_CR19","doi-asserted-by":"crossref","unstructured":"Somervuo, P., H\u00e4rm\u00e4, A., Fagerlund, S.: Parametric representations of bird sounds for automatic species recognition. IEEE Trans. Speech Audio Process. 14(6), 2252\u20132263 (2006)","DOI":"10.1109\/TASL.2006.872624"},{"key":"7_CR20","doi-asserted-by":"publisher","DOI":"10.7717\/peerj.488","volume":"2","author":"D Stowell","year":"2014","unstructured":"Stowell, D., Plumbley, M.D.: Automatic large-scale classification of bird sounds is strongly improved by unsupervised feature learning. PeerJ 2, e488 (2014)","journal-title":"PeerJ"},{"key":"7_CR21","doi-asserted-by":"publisher","unstructured":"Stowell, D., Wood, M., Stylianou, Y., Glotin, H.: Bird detection in audio: a survey and a challenge. In: IEEE International Workshop on MLSP, pp. 1\u20136 (2016). https:\/\/doi.org\/10.1109\/MLSP.2016.7738875","DOI":"10.1109\/MLSP.2016.7738875"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Tjandra, A., Sisman, B., Zhang, M., Sakti, S., Li, H., Nakamura, S.: VQVAE unsupervised unit discovery and multi-scale code2spec inverter for zerospeech challenge 2019. arXiv preprint arXiv:1905.11449 (2019)","DOI":"10.21437\/Interspeech.2019-3232"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Yamamoto, R., Song, E., Kim, J.M.: Parallel WaveGAN: a fast waveform generation model based on generative adversarial networks with multi-resolution spectrogram. In: Proceedings of the IEEE ICASSP, pp. 6199\u20136203 (2020)","DOI":"10.1109\/ICASSP40776.2020.9053795"}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-20980-2_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,12]],"date-time":"2022-11-12T19:04:55Z","timestamp":1668279895000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-20980-2_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031209796","9783031209802"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-20980-2_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"10 November 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SPECOM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Speech and Computer","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Gurugram","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 November 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 November 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"specom2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.specom.co.in","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"99","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"60","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"61% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}