{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T16:09:59Z","timestamp":1775578199834,"version":"3.50.1"},"reference-count":70,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"1","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100004358","name":"Samsung","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100004358","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Affective Comput."],"published-print":{"date-parts":[[2023,1,1]]},"DOI":"10.1109\/taffc.2021.3062406","type":"journal-article","created":{"date-parts":[[2021,2,26]],"date-time":"2021-02-26T21:04:51Z","timestamp":1614373491000},"page":"406-420","source":"Crossref","is-referenced-by-count":24,"title":["Does Visual Self-Supervision Improve Learning of Speech Representations for Emotion Recognition?"],"prefix":"10.1109","volume":"14","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4030-962X","authenticated-orcid":false,"given":"Abhinav","family":"Shukla","sequence":"first","affiliation":[{"name":"iBUG Group, Department of Computing, Imperial College London, London, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stavros","family":"Petridis","sequence":"additional","affiliation":[{"name":"iBUG Group, Department of Computing, Imperial College London, London, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Maja","family":"Pantic","sequence":"additional","affiliation":[{"name":"iBUG Group, Department of Computing, Imperial College London, London, U.K."}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Self-supervised learning by cross-modal audio-video clustering","author":"Alwassel","year":"2019"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1202"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/vl\/N19-142"},{"key":"ref4","article-title":"Unsupervised representation learning by predicting image rotations","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Gidaris"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.167"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.607"},{"key":"ref7","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","author":"Caron","year":"2020"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00674"},{"key":"ref10","first-page":"10 541","article-title":"Large scale adversarial representation learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Donahue"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00267"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01070"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00156"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00202"},{"key":"ref16","article-title":"Representation learning with contrastive predictive coding","author":"Oord","year":"2018"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1473"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2380"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref20","article-title":"Self-supervised audio representation learning for mobile devices","author":"Tagliasacchi","year":"2019"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2605"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053613"},{"key":"ref23","article-title":"Learning audio representations via phase prediction","author":"Quitry","year":"2019"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2938863"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054548"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-54184-6_6"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33016892"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1083-5"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2015.2446462"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00021"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"ref33","article-title":"Multi-modal self-supervision from generalized data transformations","author":"Patrick","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01229"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-021-1293-0"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/258734.258880"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/566654.566594"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073640"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1038\/264746a0"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461326"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2009.5349358"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3136755.3136796"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1145\/3242969.3242988"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2020.2964549"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01158-4"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1208"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123444"},{"key":"ref48","article-title":"End-to-end speech-driven facial animation with temporal GANs","volume-title":"Proc. Brit. Conf. Mach. Vis.","author":"Vougioukas"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053415"},{"key":"ref50","article-title":"Visual self-supervision by facial reconstruction for speech representation learning","volume-title":"Proc. Sight Sound Workshop CVPR","author":"Shukla"},{"key":"ref51","article-title":"Learning speech representations from raw audio by joint audiovisual self-supervision","volume-title":"Proc. Workshop Self-Supervision Audio Speech","author":"Shukla"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01251-8"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3455008"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2014.2336244"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2013.6553805"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2944808"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-008-9076-6"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2018.07.041"},{"key":"ref61","article-title":"Speech commands: A dataset for limited-vocabulary speech recognition","author":"Warden","year":"2018"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"ref63","first-page":"7763","article-title":"Cooperative learning of audio and video models from self-supervised synchronization","volume-title":"Proc. 32nd Int. Conf. Neural Inf. Process. Syst.","author":"Korbar"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1145\/2502081.2502224"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2017.2762739"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(93)90095-3"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201357"},{"key":"ref68","article-title":"Interpretable convolutional filters with SincNet","volume-title":"Proc. IEEE SLT Workshop","author":"Ravanelli"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1445"},{"key":"ref70","article-title":"Learning audio-visual representations with active contrastive coding","author":"Ma","year":"2020"}],"container-title":["IEEE Transactions on Affective Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5165369\/10056372\/09364716.pdf?arnumber=9364716","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,9]],"date-time":"2024-01-09T23:51:44Z","timestamp":1704844304000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9364716\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,1,1]]},"references-count":70,"journal-issue":{"issue":"1"},"URL":"https:\/\/doi.org\/10.1109\/taffc.2021.3062406","relation":{},"ISSN":["1949-3045","2371-9850"],"issn-type":[{"value":"1949-3045","type":"electronic"},{"value":"2371-9850","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,1,1]]}}}