{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T16:19:09Z","timestamp":1778084349729,"version":"3.51.4"},"reference-count":15,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,10,12]],"date-time":"2022-10-12T00:00:00Z","timestamp":1665532800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,12]],"date-time":"2022-10-12T00:00:00Z","timestamp":1665532800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,10,12]]},"DOI":"10.1109\/sdf55338.2022.9931697","type":"proceedings-article","created":{"date-parts":[[2022,11,9]],"date-time":"2022-11-09T15:44:42Z","timestamp":1668008682000},"page":"1-6","source":"Crossref","is-referenced-by-count":3,"title":["Audio-Visual Active Speaker Identification: A comparison of dense image-based features and sparse facial landmark-based features"],"prefix":"10.1109","author":[{"given":"Warre","family":"Geeroms","sequence":"first","affiliation":[{"name":"Ghent University - imec,TELIN-IPI,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gianni","family":"Allebosch","sequence":"additional","affiliation":[{"name":"Ghent University - imec,TELIN-IPI,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stijn","family":"Kindt","sequence":"additional","affiliation":[{"name":"Ghent University - imec,IDLab-ELIS,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Loubna","family":"Kadri","sequence":"additional","affiliation":[{"name":"Ghent University - imec,IDLab-ELIS,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peter","family":"Veelaert","sequence":"additional","affiliation":[{"name":"Ghent University - imec,TELIN-IPI,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nilesh","family":"Madhu","sequence":"additional","affiliation":[{"name":"Ghent University - imec,IDLab-ELIS,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","article-title":"Real-time facial surface geometry from monocular video on mobile gpus","author":"kartynnik","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref11","article-title":"Lrs3-ted: a large-scale dataset for visual speech recognition","author":"afouras","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref12","article-title":"Voxceleb 2: Deep speaker recognition","author":"chung","year":"2018","journal-title":"ArXiv Preprint"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_18"},{"key":"ref14","article-title":"Semi-supervised classification with graph convolutional networks","author":"kipf","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref15","first-page":"192","article-title":"S3fd: Single shot scale-invariant face detector","author":"zhang","year":"2017","journal-title":"Proceedings of the IEEE International Conference on Computer Vision"},{"key":"ref4","first-page":"251","article-title":"Out of time: automated lip sync in the wild","author":"chung","year":"2016","journal-title":"Asian Conference on Computer Vision"},{"key":"ref3","first-page":"10","year":"2021","journal-title":"Mel Frequency Cepstral Coefficient (MFCC) Tutorial"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475587"},{"key":"ref5","article-title":"Naver at activitynet challenge 2019-task b active speaker detection (ava)","year":"2019","journal-title":"ArXiv Preprint"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2017.296"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2018.00019"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2000.871073"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1002\/9780470727188.ch6"},{"key":"ref9","first-page":"354","article-title":"Constrained local neural fields for robust facial landmark detection in the wild","author":"baltrusaitis","year":"2013","journal-title":"Proceedings of the IEEE International Conference on Computer Vision Workshops"}],"event":{"name":"2022 Sensor Data Fusion: Trends, Solutions, Applications (SDF)","location":"Bonn, Germany","start":{"date-parts":[[2022,10,12]]},"end":{"date-parts":[[2022,10,14]]}},"container-title":["2022 Sensor Data Fusion: Trends, Solutions, Applications (SDF)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9931694\/9931695\/09931697.pdf?arnumber=9931697","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T17:43:12Z","timestamp":1758562992000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9931697\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,12]]},"references-count":15,"URL":"https:\/\/doi.org\/10.1109\/sdf55338.2022.9931697","relation":{},"subject":[],"published":{"date-parts":[[2022,10,12]]}}}