{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T23:26:33Z","timestamp":1784935593308,"version":"3.55.0"},"reference-count":71,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2019,4,1]],"date-time":"2019-04-01T00:00:00Z","timestamp":1554076800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,4,1]],"date-time":"2019-04-01T00:00:00Z","timestamp":1554076800000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,4,1]],"date-time":"2019-04-01T00:00:00Z","timestamp":1554076800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,4,1]],"date-time":"2019-04-01T00:00:00Z","timestamp":1554076800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["IIS-1453781"],"award-info":[{"award-number":["IIS-1453781"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2019,4]]},"DOI":"10.1109\/taslp.2019.2898816","type":"journal-article","created":{"date-parts":[[2019,2,11]],"date-time":"2019-02-11T19:40:52Z","timestamp":1549914052000},"page":"815-826","source":"Crossref","is-referenced-by-count":84,"title":["Curriculum Learning for Speech Emotion Recognition From Crowdsourced Labels"],"prefix":"10.1109","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1891-0267","authenticated-orcid":false,"given":"Reza","family":"Lotfian","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4075-4072","authenticated-orcid":false,"given":"Carlos","family":"Busso","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953140"},{"key":"ref70","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"van der maaten","year":"2008","journal-title":"J Mach Learn Res"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2017.8273633"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2015.2493525"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2005.03.007"},{"key":"ref32","first-page":"195","article-title":"Desperately seeking emotions or: Actors, wizards and human beings","author":"batliner","year":"0","journal-title":"Proc ISCA Tutorial and Research Workshop Speech and Emotion"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/1027933.1027968"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICSLP.1996.608022"},{"key":"ref37","first-page":"19","article-title":"&#x2018;FEELTRACE&#x2019;: An instrument for recording perceived emotion in real time","author":"cowie","year":"0","journal-title":"Proc ISCA Tutorial and Research Workshop Speech and Emotion"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2008.4607572"},{"key":"ref35","first-page":"801","article-title":"Real-life emotions detection with lexical and paralinguistic cues on human&#x2013;human call center dialogs","author":"devillers","year":"0","journal-title":"Proc Interspeech Int Conf Spoken Lang Process"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6289072"},{"key":"ref62","author":"lord","year":"1980","journal-title":"Applications of Item Response Theory to Practical Testing Problems"},{"key":"ref61","first-page":"4878","article-title":"Selective classification for deep neural networks","author":"geifman","year":"0","journal-title":"Proc Adv Neu Inf Proc Sys"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2736999"},{"key":"ref28","article-title":"Regularized minimax conditional entropy for crowdsourcing","author":"zhou","year":"2015","journal-title":"arXiv 1503 07240"},{"key":"ref64","first-page":"238","article-title":"Building a naturalistic emotional speech corpus by retrieving expressive behaviors from existing speech corpora","author":"mariooryad","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref27","first-page":"262","article-title":"Aggregating ordinal labels from crowds by minimax conditional entropy","author":"zhou","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref65","first-page":"148","article-title":"The INTERSPEECH 2013 computational paralinguistics challenge: Social signals, conflict, emotion, autism","author":"schuller","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1874246"},{"key":"ref29","article-title":"Emotion in speech: Recognition and application to call centers","author":"petrushin","year":"0","journal-title":"Proc Artif Neural Netw Eng"},{"key":"ref67","author":"eyben","year":"2017","journal-title":"Real-Time Speech and Music Classification by Large Audio Feature Space Extraction"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461866"},{"key":"ref69","first-page":"1","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10662-5_28"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2003.10057"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.cognition.2008.11.014"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2012.05.008"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/70.294207"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553380"},{"key":"ref23","article-title":"Learning to execute","author":"zaremba","year":"2014","journal-title":"arXiv 1410 4615"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123383"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2016.2515617"},{"key":"ref50","first-page":"21","article-title":"Computational models of emotion","author":"marsella","year":"0","journal-title":"A Blueprint for Affective Computing A Sourcebook and Manual"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1159\/000068580"},{"key":"ref59","first-page":"1595","article-title":"Data-driven clustering in emotional space for affect recognition using discriminatively trained LSTM networks","author":"w\u00f6llmer","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-24571-8_53"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2015.2508454"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1145\/2808196.2811642"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1016\/j.patrec.2014.11.007"},{"key":"ref54","first-page":"2196","article-title":"Discriminatively trained recurrent neural networks for continuous dimensional emotion recognition from audio","author":"weninger","year":"0","journal-title":"Proc Int Joint Conf Artif Intell"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1016\/S0160-2896(99)00016-1"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1037\/0893-3200.16.4.447"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2867099"},{"key":"ref11","first-page":"1123","article-title":"Improving automatic emotion recognition from speech via gender differentiation","author":"vogt","year":"0","journal-title":"Proc Int Conf Lang Resour Eval"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-868"},{"key":"ref12","first-page":"805","article-title":"Speaker independent emotion recognition by early fusion of acoustic and linguistic features within ensembles","author":"schuller","year":"0","journal-title":"Proc 9th Eur Conf Speech Commun Technol"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/T-AFFC.2013.26"},{"key":"ref14","first-page":"941","article-title":"Role of regularization in the prediction of valence from speech","author":"parthasarathy","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-998"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(02)00084-5"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2009.5349500"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1016\/0010-0277(93)90058-4"},{"key":"ref19","first-page":"384","article-title":"Word representations: A simple and general method for semi-supervised learning","author":"turian","year":"0","journal-title":"Proc Assoc Comput Linguist"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.3115\/1218955.1219000"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/1514402.1514408"},{"key":"ref6","first-page":"1054","article-title":"Understanding user needs for serious games for teaching children with autism spectrum disorders emotions","author":"abirached","year":"0","journal-title":"Proc World Conf Educ Media Technol"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2005.09.008"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/T-AFFC.2010.8"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780195387643.003.0008"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.2307\/2346806"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2011.6163986"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2765832"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO.2017.8081267"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2005.1415114"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2871949"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2011.06.004"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2011.06.004"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2017.68"},{"key":"ref43","first-page":"1","article-title":"Automated curriculum learning for neural networks","author":"graves","year":"0","journal-title":"Proc Int Conf Mach Learn"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielaam\/6570655\/8643854\/8638999-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8643854\/08638999.pdf?arnumber=8638999","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,13]],"date-time":"2022-07-13T21:05:12Z","timestamp":1657746312000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8638999\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,4]]},"references-count":71,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2019.2898816","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,4]]}}}