{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T07:32:35Z","timestamp":1780471955877,"version":"3.54.1"},"reference-count":65,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"funder":[{"name":"BMBF IKT2020-Grant","award":["16SV7213"],"award-info":[{"award-number":["16SV7213"]}]},{"DOI":"10.13039\/501100004963","name":"European Communitys Seventh Framework Programme","doi-asserted-by":"crossref","award":["338164"],"award-info":[{"award-number":["338164"]}],"id":[{"id":"10.13039\/501100004963","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100004543","name":"China Scholarship Council","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004543","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001711","name":"Swiss National Science Foundation","doi-asserted-by":"publisher","award":["(SNSF PP00P1_157409\/1)"],"award-info":[{"award-number":["(SNSF PP00P1_157409\/1)"]}],"id":[{"id":"10.13039\/501100001711","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2018,1]]},"DOI":"10.1109\/taslp.2017.2759338","type":"journal-article","created":{"date-parts":[[2017,10,5]],"date-time":"2017-10-05T18:15:40Z","timestamp":1507227340000},"page":"31-43","source":"Crossref","is-referenced-by-count":128,"title":["Semisupervised Autoencoders for Speech Emotion Recognition"],"prefix":"10.1109","volume":"26","author":[{"given":"Jun","family":"Deng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinzhou","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zixing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sascha","family":"Fruhholz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bjorn","family":"Schuller","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2014.2360798"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR.2014.141"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2015.7344564"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2015.7344553"},{"key":"ref31","first-page":"19","article-title":"Studying\n self-and active-training methods for multi-feature set emotion recognition","author":"esparza","year":"2011","journal-title":"Partially Supervised Learning"},{"key":"ref30","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","author":"glorot","year":"0","journal-title":"Proc 13th Int Conf Artif Intell Statist"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2009.5202789"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/1281192.1281225"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639325"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/279943.279962"},{"key":"ref60","first-page":"321","article-title":"Learning with local and global consistency","author":"zhou","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref62","first-page":"2794","article-title":"The INTERSPEECH 2010 Paralinguistic Challenge","author":"schuller","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref61","first-page":"2791","article-title":"Winner-take-all autoencoders","author":"makhzani","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref63","first-page":"3201","article-title":"The INTERSPEECH 2011 speaker state challenge","author":"schuller","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref28","first-page":"3581","article-title":"Semi-supervised learning\n with deep generative models","author":"kingma","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref64","first-page":"254","article-title":"The INTERSPEECH 2012 speaker trait challenge","author":"schuller","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-35289-8_34"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1561\/2200000006"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1177\/089443930101900307"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/79.911197"},{"key":"ref20","first-page":"3061","article-title":"Semi-supervised sequence learning","author":"dai","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1126\/science.1127647"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2015.2403573"},{"key":"ref24","first-page":"448","article-title":"Deep Boltzmann machines","author":"salakhutdinov","year":"0","journal-title":"Proc Int Conf Artif Intell Statist"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390294"},{"key":"ref25","first-page":"548","article-title":"Multi-prediction deep Boltzmann machines","author":"goodfellow","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2007.366340"},{"key":"ref51","first-page":"1517","article-title":"A database of German emotional speech","author":"burkhardt","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref59","article-title":"Learning from labeled and unlabeled data with label propagation","author":"zhu","year":"2002"},{"key":"ref58","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc Int Conf Pattern Recognit San Diego CA USA"},{"key":"ref57","first-page":"358","article-title":"Combining ranking and\n classification to improve emotion recognition in spontaneous speech","author":"cao","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2013.2255278"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2011.06.004"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/2502081.2502224"},{"key":"ref53","first-page":"1459","article-title":"openSMILE&#x2014;The Munich versatile and fast open-source audio feature extractor","author":"eyben","year":"0","journal-title":"Proc ACM Int Conf on Multimedia MM'10"},{"key":"ref52","first-page":"1743","article-title":"Getting started with SUSAS: A speech under simulated and actual stress database","author":"hansen","year":"0","journal-title":"Proc EUROSPEECH"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref11","first-page":"216","article-title":"Emotion recognition from spontaneous speech using hidden\n Markov models with deep belief networks","author":"provost","year":"0","journal-title":"Proc Automatic Speech Recognition and Understanding"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390224"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2015.7344598"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2016.7727636"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472669"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/9780262033589.001.0001"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2013.90"},{"key":"ref17","first-page":"115","article-title":"Cooperative learning and its\n application to emotion recognition from speech","volume":"23","author":"zhang","year":"2015","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"ref18","first-page":"2856","article-title":"Active learning by label uncertainty\n for acoustic emotion recognition","author":"zhang","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref19","first-page":"3546","article-title":"Semi-supervised learning with ladder networks","author":"rasmus","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/T-AFFC.2010.1"},{"key":"ref3","first-page":"1","article-title":"Features and classifiers for emotion recognition from speech: A survey from 2000 to 2011","author":"anagnostopoulos","year":"2012","journal-title":"Artif Intell Rev"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(03)00099-2"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1002\/9781118706664"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2014.2324759"},{"key":"ref7","first-page":"957","article-title":"Speech\n emotion recognition using Gaussian mixture vector autoregressive models","author":"ayadi","year":"0","journal-title":"Proc Int Conf Acoust Speech Signal Process"},{"key":"ref49","first-page":"148","article-title":"The INTERSPEECH 2013 Computational\n Paralinguistics Challenge: Social signals, conflict, emotion, autism","author":"schuller","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref46","first-page":"312","article-title":"The\n INTERSPEECH 2009 Emotion Challenge","author":"schuller","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-10-2585-3_15"},{"key":"ref48","article-title":"Fisher kernels on phase-based features for speech emotion recognition","author":"deng","year":"0","journal-title":"Proc 1st Int Workshop Spoken Dialogue Syst"},{"key":"ref47","first-page":"348","article-title":"Brno university of technology system for Interspeech 2009 emotion challenge","author":"kockmann","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref42","first-page":"807","article-title":"Rectified linear units improve restricted Boltzmann machines","author":"nair","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref41","first-page":"643","article-title":"Learning algorithms for the\n classification restricted boltzmann machine","volume":"13","author":"larochelle","year":"2012","journal-title":"J Mach Learn Res"},{"key":"ref44","first-page":"2377","article-title":"Training very deep networks","author":"greff","year":"0","journal-title":"Proc Int Conf Neural Inf Process"},{"key":"ref43","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"0","journal-title":"Proc Int Conf Mach Learn"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8124117\/08059872.pdf?arnumber=8059872","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,12]],"date-time":"2022-01-12T16:24:16Z","timestamp":1642004656000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/8059872\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,1]]},"references-count":65,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2017.2759338","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,1]]}}}