{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T14:44:49Z","timestamp":1742913889531,"version":"3.40.3"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031483080"},{"type":"electronic","value":"9783031483097"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-48309-7_3","type":"book-chapter","created":{"date-parts":[[2023,11,21]],"date-time":"2023-11-21T20:03:21Z","timestamp":1700597001000},"page":"32-42","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Significance of\u00a0Audio Quality in\u00a0Speech-to-Text Translation Systems"],"prefix":"10.1007","author":[{"given":"Tonmoy","family":"Rajkhowa","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Amartya Roy","family":"Chowdhury","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S. R. Mahadeva","family":"Prasanna","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,22]]},"reference":[{"key":"3_CR1","doi-asserted-by":"crossref","unstructured":"Ali, A., Renals, S.: Word error rate estimation for speech recognition: e-WER. In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers), pp. 20\u201324. Association for Computational Linguistics, Melbourne, Australia (2018)","DOI":"10.18653\/v1\/P18-2004"},{"key":"3_CR2","doi-asserted-by":"crossref","unstructured":"Anastasopoulos, A., Chiang, D., Duong, L.: An unsupervised probability model for speech-to-translation alignment of low-resource languages. In: Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing, pp. 1255\u20131263. Association for Computational Linguistics, Austin (2016)","DOI":"10.18653\/v1\/D16-1133"},{"key":"3_CR3","unstructured":"Berard, A., Pietquin, O., Servan, C., Besacier, L.: Listen and translate: a proof of concept for end-to-end speech-to-text translation (2016)"},{"key":"3_CR4","unstructured":"Chadha, H.S., Gupta, A., Shah, P., Chhimwal, N., Dhuriya, A., Gaur, R.: Vakyansh: TTS Indic languages (2022)"},{"key":"3_CR5","unstructured":"Chadha, H.S., et al.: Vakyansh: ASR toolkit for low resource Indic languages. arXiv preprint arXiv:2203.16512 (2022)"},{"key":"3_CR6","unstructured":"Di Gangi, M.A., Cattoni, R., Bentivogli, L., Negri, M., Turchi, M.: MuST-C: a multilingual speech translation corpus. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), pp. 2012\u20132017. Association for Computational Linguistics, Minneapolis (2019)"},{"key":"3_CR7","doi-asserted-by":"publisher","unstructured":"Duong, L., Anastasopoulos, A., Chiang, D., Bird, S., Cohn, T.: An attentional model for speech translation without transcription. In: Proceedings of the 2016 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 949\u2013959. Association for Computational Linguistics, San Diego (2016). https:\/\/doi.org\/10.18653\/v1\/N16-1109, https:\/\/aclanthology.org\/N16-1109","DOI":"10.18653\/v1\/N16-1109"},{"key":"3_CR8","unstructured":"Gaido, M., Negri, M., Cettolo, M., Turchi, M.: Beyond voice activity detection: hybrid audio segmentation for direct speech translation. In: Proceedings of the 4th International Conference on Natural Language and Speech Processing (ICNLSP 2021), pp. 55\u201362. Association for Computational Linguistics, Trento (2021). http:\/\/aclanthology.org\/2021.icnlsp-1.7"},{"key":"3_CR9","unstructured":"Gupta, A., Chadha, H.S., Shah, P., Chhimwal, N., Dhuriya, A., Gaur, R., Raghavan, V.: CLSRIL-23: cross lingual speech representations for Indic languages. arXiv preprint arXiv:2107.07402 (2021)"},{"key":"3_CR10","doi-asserted-by":"publisher","unstructured":"Indurthi, S., et al.: End-end speech-to-text translation with modality agnostic meta-learning. In: ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7904\u20137908 (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9054759","DOI":"10.1109\/ICASSP40776.2020.9054759"},{"key":"3_CR11","doi-asserted-by":"crossref","unstructured":"Iranzo-Sanchez, J., et al.: Europarl-ST: a multilingual corpus for speech translation of parliamentary debates, pp. 8229\u20138233 (2020)","DOI":"10.1109\/ICASSP40776.2020.9054626"},{"key":"3_CR12","doi-asserted-by":"crossref","unstructured":"Jia, Y., et al.: Leveraging weakly supervised data to improve end-to-end speech-to-text translation. In: ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7180\u20137184 (2019)","DOI":"10.1109\/ICASSP.2019.8683343"},{"key":"3_CR13","doi-asserted-by":"crossref","unstructured":"Klein, G., Kim, Y., Deng, Y., Senellart, J., Rush, A.: OpenNMT: open-source toolkit for neural machine translation. In: Proceedings of ACL 2017, System Demonstrations, pp. 67\u201372. Association for Computational Linguistics, Vancouver (2017)","DOI":"10.18653\/v1\/P17-4012"},{"key":"3_CR14","doi-asserted-by":"publisher","unstructured":"Kudo, T., Richardson, J.: SentencePiece: a simple and language independent subword tokenizer and detokenizer for neural text processing. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, pp. 66\u201371. Association for Computational Linguistics, Brussels (2018). https:\/\/doi.org\/10.18653\/v1\/D18-2012, https:\/\/aclanthology.org\/D18-2012","DOI":"10.18653\/v1\/D18-2012"},{"key":"3_CR15","unstructured":"LeCun, Y., Bengio, Y.: Convolutional networks for images, speech, and time series, pp. 255\u2013258. MIT Press, Cambridge (1998)"},{"key":"3_CR16","doi-asserted-by":"publisher","unstructured":"Ott, M., et al.: fairseq: a fast, extensible toolkit for sequence modeling. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics (Demonstrations), pp. 48\u201353. Association for Computational Linguistics, Minneapolis (2019). https:\/\/doi.org\/10.18653\/v1\/N19-4009, https:\/\/aclanthology.org\/N19-4009","DOI":"10.18653\/v1\/N19-4009"},{"key":"3_CR17","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318. Association for Computational Linguistics, Philadelphia (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"Post, M.: A call for clarity in reporting BLEU scores. In: Proceedings of the Third Conference on Machine Translation: Research Papers, pp. 186\u2013191. Association for Computational Linguistics, Brussels (2018)","DOI":"10.18653\/v1\/W18-6319"},{"key":"3_CR19","unstructured":"Post, M., Kumar, G.S., Lopez, A., Karakos, D.G., Callison-Burch, C., Khudanpur, S.: Improved speech-to-text translation with the fisher and Callhome Spanish-English speech translation corpus. In: International Workshop on Spoken Language Translation (2013)"},{"key":"3_CR20","doi-asserted-by":"crossref","unstructured":"Rix, A., Beerends, J., Hollier, M., Hekstra, A.: Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs. In: 2001 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No.01CH37221), vol. 2, pp. 749\u2013752 (2001)","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"3_CR21","doi-asserted-by":"crossref","unstructured":"Salesky, E., et al.: The multilingual TEDx corpus for speech recognition and translation, pp. 3655\u20133659 (2021)","DOI":"10.21437\/Interspeech.2021-11"},{"key":"3_CR22","unstructured":"Sandhan, J., Daksh, A., Paranjay, O.A., Behera, L., Goyal, P.: Prabhupadavani: a code-mixed speech translation data for 25 languages. arXiv preprint arXiv:2201.11391 (2022)"},{"key":"3_CR23","doi-asserted-by":"crossref","unstructured":"Shen, J., et al.: Natural TTS synthesis by conditioning wavenet on MEL spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783 (2018)","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"3_CR24","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Guyon, I., et al. (eds.) Advances in Neural Information Processing Systems, vol. 30. Curran Associates, Inc. (2017)"},{"issue":"3","key":"3_CR25","doi-asserted-by":"publisher","first-page":"70","DOI":"10.1109\/MSP.2008.918415","volume":"25","author":"A Waibel","year":"2008","unstructured":"Waibel, A., Fugen, C.: Spoken language translation. IEEE Signal Process. Mag. 25(3), 70\u201379 (2008)","journal-title":"IEEE Signal Process. Mag."},{"key":"3_CR26","unstructured":"Wang, C., Pino, J., Wu, A., Gu, J.: CoVoST: a diverse multilingual speech-to-text translation corpus. In: Proceedings of the Twelfth Language Resources and Evaluation Conference, pp. 4197\u20134203. European Language Resources Association, Marseille (2020)"},{"key":"3_CR27","doi-asserted-by":"crossref","unstructured":"Yang, J., Hussein, A., Wiesner, M., Khudanpur, S.: JHU IWSLT 2022 dialect speech translation system description. In: Proceedings of the 19th International Conference on Spoken Language Translation (IWSLT 2022), pp. 319\u2013326. Association for Computational Linguistics, Dublin (in-person and online) (2022)","DOI":"10.18653\/v1\/2022.iwslt-1.29"},{"key":"3_CR28","doi-asserted-by":"publisher","unstructured":"Zhang, S., Feng, Y.: End-to-end simultaneous speech translation with differentiable segmentation. In: Findings of the Association for Computational Linguistics: ACL 2023, pp. 7659\u20137680. Association for Computational Linguistics, Toronto (2023). https:\/\/doi.org\/10.18653\/v1\/2023.findings-acl.485, https:\/\/aclanthology.org\/2023.findings-acl.485","DOI":"10.18653\/v1\/2023.findings-acl.485"}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-48309-7_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T14:48:00Z","timestamp":1730558880000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-48309-7_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031483080","9783031483097"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-48309-7_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"22 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SPECOM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Speech and Computer","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dharwad","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"specom2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.iitdh.ac.in\/specom-2023\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"174","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"94","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"54% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}