{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,27]],"date-time":"2025-05-27T15:21:23Z","timestamp":1748359283160,"version":"3.37.3"},"reference-count":20,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,5,23]],"date-time":"2022-05-23T00:00:00Z","timestamp":1653264000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,5,23]],"date-time":"2022-05-23T00:00:00Z","timestamp":1653264000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100015803","name":"Tencent","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100015803","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,5,23]]},"DOI":"10.1109\/icassp43922.2022.9747595","type":"proceedings-article","created":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T19:50:34Z","timestamp":1651089034000},"page":"5068-5072","source":"Crossref","is-referenced-by-count":8,"title":["Audio-Visual Tracking of Multiple Speakers Via a PMBM Filter"],"prefix":"10.1109","author":[{"given":"Jinzheng","family":"Zhao","sequence":"first","affiliation":[{"name":"University of Surrey,Centre for Vision, Speech and Signal Processing (CVSSP),UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peipei","family":"Wu","sequence":"additional","affiliation":[{"name":"University of Surrey,Centre for Vision, Speech and Signal Processing (CVSSP),UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xubo","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Surrey,Centre for Vision, Speech and Signal Processing (CVSSP),UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yong","family":"Xu","sequence":"additional","affiliation":[{"name":"Tencent AI Lab,Bellevue,WA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lyudmila","family":"Mihaylova","sequence":"additional","affiliation":[{"name":"University of Sheffield,Department of Automatic Control and Systems Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Simon","family":"Godsill","sequence":"additional","affiliation":[{"name":"University of Cambridge,Department of Engineering,UK"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenwu","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Surrey,Centre for Vision, Speech and Signal Processing (CVSSP),UK"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.23919\/ICIF.2017.8009710"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/IVS.2018.8500454"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9415072"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2902489"},{"key":"ref14","article-title":"A lightened cnn for deep face representation","volume":"4","author":"wu","year":"2015"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00520"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4518618"},{"article-title":"Yolov3: An incremental improvement","year":"2018","author":"redmon","key":"ref17"},{"key":"ref18","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"lin","year":"2014","journal-title":"European Conference on Computer Vision"},{"key":"ref19","first-page":"182","article-title":"Av16. 3: An audio-visual corpus for speaker localization and tracking","author":"lathoud","year":"2004","journal-title":"Int Workshop Mach Learn Multimodal Interact"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2005.1406476"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2648793"},{"key":"ref6","doi-asserted-by":"crossref","first-page":"2417","DOI":"10.1109\/TMM.2016.2599150","article-title":"Mean-shift and sparse sampling-based smc-phd filtering for audio informed visual speaker tracking","volume":"18","author":"k?l?c\u00b8","year":"2016","journal-title":"IEEE Transactions on Multimedia"},{"key":"ref5","first-page":"186","article-title":"Audio assisted robust visual tracking with adaptive particle filtering","volume":"17","author":"k?l?c\u00b8","year":"2014","journal-title":"IEEE Transactions on Multimedia"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TAES.2015.130550"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1969"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2010.2057890"},{"key":"ref1","article-title":"Joint audiovisual speech processing for recognition and enhancement","author":"potamianos","year":"2003","journal-title":"AVSP 2003-International Conference on Audio-Visual Speech Processing"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TAES.2018.2805153"},{"key":"ref20","first-page":"1","article-title":"On performance evaluation of multi-object filters","author":"schuhmacher","year":"2008","journal-title":"2008 11th International Conference on Information Fusion FUSION"}],"event":{"name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2022,5,23]]},"location":"Singapore, Singapore","end":{"date-parts":[[2022,5,27]]}},"container-title":["ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9745891\/9746004\/09747595.pdf?arnumber=9747595","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,15]],"date-time":"2022-08-15T20:10:10Z","timestamp":1660594210000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9747595\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,5,23]]},"references-count":20,"URL":"https:\/\/doi.org\/10.1109\/icassp43922.2022.9747595","relation":{},"subject":[],"published":{"date-parts":[[2022,5,23]]}}}