{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T20:30:41Z","timestamp":1742934641822,"version":"3.40.3"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031483080"},{"type":"electronic","value":"9783031483097"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-48309-7_14","type":"book-chapter","created":{"date-parts":[[2023,11,21]],"date-time":"2023-11-21T20:03:21Z","timestamp":1700597001000},"page":"169-176","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Learning to\u00a0Predict Speech Intelligibility from\u00a0Speech Distortions"],"prefix":"10.1007","author":[{"given":"Punnoose","family":"Kuriakose","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,22]]},"reference":[{"key":"14_CR1","unstructured":"audiomentations: A Python library for audio data augmentation. https:\/\/github.com\/iver56\/audiomentations"},{"issue":"10","key":"14_CR2","doi-asserted-by":"publisher","first-page":"1925","DOI":"10.1109\/TASLP.2018.2847459","volume":"26","author":"AH Andersen","year":"2018","unstructured":"Andersen, A.H., de Haan, J.M., Tan, Z.H., Jensen, J.: Nonintrusive speech intelligibility prediction using convolutional neural networks. IEEE\/ACM Trans. Audio Speech Lang. Process. 26(10), 1925\u20131939 (2018). https:\/\/doi.org\/10.1109\/TASLP.2018.2847459","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"14_CR3","doi-asserted-by":"crossref","unstructured":"Avila, A.R., Gamper, H., Reddy, C., Cutler, R., Tashev, I., Gehrke, J.: Non-intrusive speech quality assessment using neural networks. arXiv (2019)","DOI":"10.1109\/ICASSP.2019.8683175"},{"key":"14_CR4","unstructured":"Beerends, J., et al.: Perceptual objective listening quality assessment (polqa), the third generation itu-t standard for end-to-end speech quality measurement part i-temporal alignment. AES: J. Audio Eng. Soc. 61, 366\u2013384 (2013)"},{"key":"14_CR5","doi-asserted-by":"crossref","unstructured":"Dean, D., Sridharan, S., Vogt, R., Mason, M.: The qut-noise-timit corpus for evaluation of voice activity detection algorithms. In: Hirose, K., Nakamura, S., Kaboyashi, T. (eds.) Proceedings of the 11th Annual Conference of the International Speech Communication Association, pp. 3110\u20133113. International Speech Communication Association, CD Rom (2010)","DOI":"10.21437\/Interspeech.2010-774"},{"key":"14_CR6","doi-asserted-by":"publisher","unstructured":"Dong, X., Williamson, D.S.: An attention enhanced multi-task model for objective speech assessment in real-world environments. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 911\u2013915 (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9053366","DOI":"10.1109\/ICASSP40776.2020.9053366"},{"key":"14_CR7","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1121\/1.1916407","volume":"19","author":"NR French","year":"1945","unstructured":"French, N.R., Steinberg, J.C.: Factors governing the intelligibility of speech sounds. J. Acoust. Soc. Am. 19, 90\u2013119 (1945)","journal-title":"J. Acoust. Soc. Am."},{"key":"14_CR8","doi-asserted-by":"crossref","unstructured":"Fu, S.W., Tsao, Y., Hwang, H.T., Wang, H.: Quality-net: an end-to-end non-intrusive speech quality assessment model based on BLSTM. ArXiv abs\/1808.05344 (2018)","DOI":"10.21437\/Interspeech.2018-1802"},{"issue":"13","key":"14_CR9","first-page":"1","volume":"2015","author":"A Hines","year":"2015","unstructured":"Hines, A., Skoglund, J., Kokaram, A., Harte, N.: Visqol: an objective speech quality model. EURASIP J. Audio Speech Music Process. 2015(13), 1\u201318 (2015)","journal-title":"EURASIP J. Audio Speech Music Process."},{"issue":"11","key":"14_CR10","doi-asserted-by":"publisher","first-page":"2009","DOI":"10.1109\/TASLP.2016.2585878","volume":"24","author":"J Jensen","year":"2016","unstructured":"Jensen, J., Taal, C.H.: An algorithm for predicting the intelligibility of speech masked by modulated noise maskers. IEEE\/ACM Trans. Audio Speech Lang. Process. 24(11), 2009\u20132022 (2016). https:\/\/doi.org\/10.1109\/TASLP.2016.2585878","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"14_CR11","doi-asserted-by":"publisher","unstructured":"J\u00fcrgens, T., Brand, T.: Microscopic prediction of speech recognition for listeners with normal hearing in noise using an auditory modela). J. Acoust. Soc. Am. 126(5), 2635\u20132648 (2009). https:\/\/doi.org\/10.1121\/1.3224721","DOI":"10.1121\/1.3224721"},{"key":"14_CR12","doi-asserted-by":"crossref","unstructured":"J\u00fcrgens, T., Fredelake, S., Meyer, R., Kollmeier, B., Brand, T.: Challenging the speech intelligibility index: macroscopic vs. microscopic prediction of sentence recognition in normal and hearing-impaired listeners, pp. 2478\u20132481 (2010). https:\/\/doi.org\/10.21437\/Interspeech.2010--666","DOI":"10.21437\/Interspeech.2010-666"},{"key":"14_CR13","doi-asserted-by":"publisher","unstructured":"J\u00fcrgensen, S., Dau, T.: Predicting speech intelligibility based on the signal-to-noise envelope power ratio after modulation-frequency selective processing. J. Acoust. Soc. Am. 130(3), 1475\u20131487 (2011). https:\/\/doi.org\/10.1121\/1.3621502","DOI":"10.1121\/1.3621502"},{"key":"14_CR14","doi-asserted-by":"publisher","unstructured":"Karbasi, M., Abdelaziz, A.H., Kolossa, D.: Twin-hmm-based non-intrusive speech intelligibility prediction. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 624\u2013628 (2016). https:\/\/doi.org\/10.1109\/ICASSP.2016.7471750","DOI":"10.1109\/ICASSP.2016.7471750"},{"key":"14_CR15","unstructured":"Karbasi, M., Bleeck, S., Kolossa, D.: Non-intrusive speech intelligibility prediction using automatic speech recognition derived measures. arXiv: Audio and Speech Processing (2020)"},{"key":"14_CR16","doi-asserted-by":"crossref","unstructured":"Lo, C., et al.: Mosnet: deep learning based objective assessment for voice conversion. CoRR abs\/1904.08352 (2019). http:\/\/arxiv.org\/abs\/1904.08352","DOI":"10.21437\/Interspeech.2019-2003"},{"key":"14_CR17","doi-asserted-by":"crossref","unstructured":"Manocha, P., Xu, B., Kumar, A.: Noresqa - a framework for speech quality assessment using non-matching references. In: Neural Information Processing Systems (2021)","DOI":"10.21437\/Interspeech.2022-407"},{"key":"14_CR18","doi-asserted-by":"publisher","unstructured":"Martinez, A.M.C., Spille, C., Ro\u00dfbach, J., Kollmeier, B., Meyer, B.T.: Prediction of speech intelligibility with DNN-based performance measures. Comput. Speech Lang. 74, 101329 (2022). https:\/\/doi.org\/10.1016\/j.csl.2021.101329","DOI":"10.1016\/j.csl.2021.101329"},{"key":"14_CR19","unstructured":"Patton, B., Agiomyrgiannakis, Y., Terry, M., Wilson, K.W., Saurous, R.A., Sculley, D.: Automos: Learning a non-intrusive assessor of naturalness-of-speech. CoRR abs\/1611.09207 (2016). http:\/\/arxiv.org\/abs\/1611.09207"},{"key":"14_CR20","doi-asserted-by":"publisher","unstructured":"Pavlovic, C.: Sii\u2013speech intelligibility index standard: Ansi s3.5 1997. J. Acoust. Soc. Am. 143(3_Supplement), 1906\u20131906 (2018). https:\/\/doi.org\/10.1121\/1.5036206","DOI":"10.1121\/1.5036206"},{"key":"14_CR21","doi-asserted-by":"publisher","unstructured":"Reddy, C.K.A., Gopal, V., Cutler, R.: Dnsmos p. 835: a non-intrusive perceptual objective speech quality metric to evaluate noise suppressors. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 886\u2013890 (2022). https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9746108","DOI":"10.1109\/ICASSP43922.2022.9746108"},{"key":"14_CR22","doi-asserted-by":"publisher","unstructured":"Rix, A., Beerends, J., Hollier, M., Hekstra, A.: Perceptual evaluation of speech quality (pesq)-a new method for speech quality assessment of telephone networks and codecs. In: 2001 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No.01CH37221), vol. 2, pp. 749\u2013752 (2001). https:\/\/doi.org\/10.1109\/ICASSP.2001.941023","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"14_CR23","doi-asserted-by":"crossref","unstructured":"Ro\u00dfbach, J., R\u00f6ttges, S., Hauth, C.F., Brand, T., Meyer, B.T.: Non-intrusive binaural prediction of speech intelligibility based on phoneme classification. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 396\u2013400 (2021)","DOI":"10.1109\/ICASSP39728.2021.9413874"},{"key":"14_CR24","doi-asserted-by":"crossref","unstructured":"Roux, J.L., Wisdom, S., Erdogan, H., Hershey, J.R.: SDR - half-baked or well done? ICASSP 2019\u20132019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 626\u2013630 (2018)","DOI":"10.1109\/ICASSP.2019.8683855"},{"key":"14_CR25","doi-asserted-by":"crossref","unstructured":"Sch\u00e4dler, M., Warzybok, A., Hochmuth, S., Kollmeier, B.: Matrix sentence intelligibility prediction using an automatic speech recognition system. Int. J. Audiol. 54, 1\u20138 (2015)","DOI":"10.3109\/14992027.2015.1061708"},{"key":"14_CR26","doi-asserted-by":"crossref","unstructured":"Serr\u00e0, J., Pons, J., Pascual, S.: SESQA: semi-supervised learning for speech quality assessment. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 381\u2013385 (2021)","DOI":"10.1109\/ICASSP39728.2021.9414052"},{"key":"14_CR27","doi-asserted-by":"publisher","unstructured":"Soni, M.H., Patil, H.A.: Novel deep autoencoder features for non-intrusive speech quality assessment. In: 2016 24th European Signal Processing Conference (EUSIPCO), pp. 2315\u20132319 (2016). https:\/\/doi.org\/10.1109\/EUSIPCO.2016.7760662","DOI":"10.1109\/EUSIPCO.2016.7760662"},{"issue":"1","key":"14_CR28","doi-asserted-by":"publisher","first-page":"318","DOI":"10.1121\/1.384464","volume":"67","author":"HJM Steeneken","year":"1980","unstructured":"Steeneken, H.J.M., Houtgast, T.: A physical method for measuring speech-transmission quality. J. Acoust. Soc. Am. 67(1), 318\u201326 (1980)","journal-title":"J. Acoust. Soc. Am."},{"key":"14_CR29","doi-asserted-by":"publisher","unstructured":"Szegedy, C., et al.: Going deeper with convolutions. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1\u20139 (2015). https:\/\/doi.org\/10.1109\/CVPR.2015.7298594","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"14_CR30","doi-asserted-by":"publisher","unstructured":"Taal, C.H., Hendriks, R.C., Heusdens, R., Jensen, J.: A short-time objective intelligibility measure for time-frequency weighted noisy speech. In: 2010 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 4214\u20134217 (2010). https:\/\/doi.org\/10.1109\/ICASSP.2010.5495701","DOI":"10.1109\/ICASSP.2010.5495701"},{"issue":"4","key":"14_CR31","doi-asserted-by":"publisher","first-page":"1462","DOI":"10.1109\/TSA.2005.858005","volume":"14","author":"E Vincent","year":"2006","unstructured":"Vincent, E., Gribonval, R., Fevotte, C.: Performance measurement in blind audio source separation. IEEE Trans. Audio Speech Lang. Process. 14(4), 1462\u20131469 (2006). https:\/\/doi.org\/10.1109\/TSA.2005.858005","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"14_CR32","doi-asserted-by":"publisher","unstructured":"Zhang, Z., Vyas, P., Dong, X., Williamson, D.S.: An end-to-end non-intrusive model for subjective and objective real-world speech assessment using a multi-task framework. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 316\u2013320 (2021). https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9414182","DOI":"10.1109\/ICASSP39728.2021.9414182"}],"container-title":["Lecture Notes in Computer Science","Speech and Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-48309-7_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,21]],"date-time":"2023-11-21T20:11:06Z","timestamp":1700597466000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-48309-7_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031483080","9783031483097"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-48309-7_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"22 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"SPECOM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Speech and Computer","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Dharwad","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"specom2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.iitdh.ac.in\/specom-2023\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easychair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"174","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"94","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"54% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}