{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,31]],"date-time":"2025-08-31T10:08:26Z","timestamp":1756634906430,"version":"3.37.3"},"reference-count":69,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"DOI":"10.13039\/100010661","name":"European Union\u2019s Horizon 2020 Research and Innovation Framework Programme under the Global Response Against Child Exploitation (GRACE) Project","doi-asserted-by":"publisher","award":["883341"],"award-info":[{"award-number":["883341"]}],"id":[{"id":"10.13039\/100010661","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014180","name":"\u201cAyudas para financiar la contrataci\u00f3n predoctoral de personal investigador\u201d grant of the Government of Castilla y Le\u00f3n, Spain","doi-asserted-by":"publisher","award":["EDU\/875\/2021"],"award-info":[{"award-number":["EDU\/875\/2021"]}],"id":[{"id":"10.13039\/501100014180","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/access.2023.3300973","type":"journal-article","created":{"date-parts":[[2023,8,1]],"date-time":"2023-08-01T18:18:27Z","timestamp":1690913907000},"page":"80089-80104","source":"Crossref","is-referenced-by-count":4,"title":["MeWEHV: Mel and Wave Embeddings for Human Voice Tasks"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9446-0152","authenticated-orcid":false,"given":"Andr\u00e9s","family":"Carofilis","sequence":"first","affiliation":[{"name":"Department of Electrical, Systems, and Automation Engineering, School of Industrial, Computer and Aerospace Engineering, Universidad de Le&#x00F3;n, Campus de Vegazana, Le&#x00F3;n, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6573-8477","authenticated-orcid":false,"given":"Laura","family":"Fern\u00e1ndez-Robles","sequence":"additional","affiliation":[{"name":"Department of Mechanical, Computer, and Aerospace Engineering, Universidad de Le&#x00F3;n, Campus de Vegazana, Le&#x00F3;n, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2081-774X","authenticated-orcid":false,"given":"Enrique","family":"Alegre","sequence":"additional","affiliation":[{"name":"Department of Electrical, Systems, and Automation Engineering, School of Industrial, Computer and Aerospace Engineering, Universidad de Le&#x00F3;n, Campus de Vegazana, Le&#x00F3;n, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1202-5232","authenticated-orcid":false,"given":"Eduardo","family":"Fidalgo","sequence":"additional","affiliation":[{"name":"Department of Electrical, Systems, and Automation Engineering, School of Industrial, Computer and Aerospace Engineering, Universidad de Le&#x00F3;n, Campus de Vegazana, Le&#x00F3;n, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2022-36"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2826"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404779"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-374"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3008832"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.116469"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-021-11610-8"},{"key":"ref53","first-page":"1","article-title":"Resources for Indian languages","author":"baby","year":"2016","journal-title":"Proc Text Speech Dialogue"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2012-659"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-950"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1994.389377"},{"key":"ref10","first-page":"6504","article-title":"Crowdsourcing Latin American Spanish for low-resource text-to-speech","author":"guevara-rukoz","year":"2020","journal-title":"Proc 12th Lang Resour Eval Conf (LREC)"},{"key":"ref54","first-page":"6486","article-title":"Mass: A large and clean multilingual corpus of sentence-aligned spoken utterances extracted from the Bible","author":"boito","year":"2020","journal-title":"Proc 12th Lang Resour Eval Conf (LREC)"},{"key":"ref17","first-page":"12449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","author":"baevski","year":"2020","journal-title":"Proc Annu Conf Neural Inf Process Syst"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1242"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2020.114416"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383491"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-017-5539-3"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TENCON.2018.8650444"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-00810-9_6"},{"key":"ref47","doi-asserted-by":"crossref","first-page":"252","DOI":"10.1007\/978-3-030-34255-5_17","article-title":"Spoken language identification using ConvNets","volume":"11912","author":"shukla","year":"2019","journal-title":"Proc 15th Eur Conf Ambient Intell"},{"key":"ref42","first-page":"9","article-title":"A multi-device dataset for urban acoustic scene classification","author":"mesaros","year":"2018","journal-title":"Proc Scenes Events Workshop (DCASE)"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683292"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2002.800560"},{"key":"ref43","first-page":"69","article-title":"General-purpose tagging of freesound audio with audioset labels: Task description, dataset, and baseline","author":"fonseca","year":"2018","journal-title":"Proc Scenes Events Workshop (DCASE)"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICAPR.2015.7050669"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052942"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638969"},{"journal-title":"TIMIT Acoust -Phonet Continuous Speech Corpus","year":"1993","author":"garofolo","key":"ref4"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP57327.2022.10038152"},{"key":"ref6","first-page":"4218","article-title":"Common voice: A massively-multilingual speech corpus","author":"ardila","year":"2020","journal-title":"Proc 12th Lang Resour Eval Conf"},{"journal-title":"VoxForge","year":"2018","author":"maclean","key":"ref5"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806390"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2020.3004555"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413386"},{"key":"ref37","article-title":"Speech commands: A dataset for limited-vocabulary speech recognition","author":"warden","year":"2018","journal-title":"arXiv 1804 03209"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-018-3760-2"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Eurospeech.1997-494"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2396"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414292"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2021.107141"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1775"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.11259"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-020-09825-6"},{"key":"ref23","first-page":"3581","article-title":"Semi-supervised learning with deep generative models","author":"kingma","year":"2014","journal-title":"Proc Annu Conf Neural Inf Process Syst"},{"key":"ref67","article-title":"A dataset for voice-based human identity recognition","volume":"42","author":"baha\u2019a","year":"2022","journal-title":"Data Brief"},{"key":"ref26","first-page":"374","article-title":"Distance measures for speech recognition, psychological and instrumental","volume":"116","author":"mermelstein","year":"1976","journal-title":"Pattern Recognit Artif Intell"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1005"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2009-5"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2773081"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-69"},{"key":"ref63","first-page":"4218","article-title":"Common voice: A massively-multilingual speech corpus","author":"ardila","year":"2020","journal-title":"Proc 12th Lang Resour Eval Conf (LREC)"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10796"},{"key":"ref66","first-page":"5353","article-title":"AccentDB: A database of non-native English accents to assist neural speech recognition","author":"ahamad","year":"2020","journal-title":"Proc 12th Lang Resour Eval Conf"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-329"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414349"},{"key":"ref28","first-page":"550","article-title":"AP20-OLR challenge: Three tasks and their baselines","author":"li","year":"2020","journal-title":"Proc Asia&#x2013;Pacific Signal Inf Process Assoc Annu Summit Conf (APSIPA ASC)"},{"key":"ref27","first-page":"946","article-title":"Heart sound analysis using MFCC and time frequency distribution","author":"kamarulafizam","year":"2007","journal-title":"Proc World Congr Med Phys Biomed Eng"},{"key":"ref29","article-title":"THCHS-30: A free Chinese speech corpus","author":"wang","year":"2015","journal-title":"arXiv 1512 01882"},{"key":"ref60","first-page":"499","article-title":"A discriminative feature learning approach for deep face recognition","volume":"9911","author":"wen","year":"2016","journal-title":"Proc 14th Eur Conf Comput Vis (ECCV)"},{"journal-title":"CommonLanguage","year":"2021","author":"sinisetty","key":"ref62"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-32520-6_22"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10005208\/10198451.pdf?arnumber=10198451","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,21]],"date-time":"2023-08-21T18:03:52Z","timestamp":1692641032000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10198451\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":69,"URL":"https:\/\/doi.org\/10.1109\/access.2023.3300973","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2023]]}}}