{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,8]],"date-time":"2026-04-08T16:50:08Z","timestamp":1775667008238,"version":"3.50.1"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2023,2,21]],"date-time":"2023-02-21T00:00:00Z","timestamp":1676937600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,2,21]],"date-time":"2023-02-21T00:00:00Z","timestamp":1676937600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Process Lett"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s11063-023-11183-7","type":"journal-article","created":{"date-parts":[[2023,2,21]],"date-time":"2023-02-21T20:57:38Z","timestamp":1677013058000},"page":"8887-8901","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Multi-speaker DoA Estimation Using Audio and Visual Modality"],"prefix":"10.1007","volume":"55","author":[{"given":"Yulin","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruimin","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaochen","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shanfa","family":"Ke","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,2,21]]},"reference":[{"key":"11183_CR1","doi-asserted-by":"crossref","unstructured":"Adavanne S, Politis A, Virtanen T (2018) Direction of arrival estimation for multiple sound sources using convolutional recurrent neural network. In: 26th european signal processing conference (EUSIPCO), pp 1462\u20131466","DOI":"10.23919\/EUSIPCO.2018.8553182"},{"key":"11183_CR2","doi-asserted-by":"crossref","unstructured":"Adavanne S, Politis A, Nikunen J, Virtanen T (2019) Sound event localization and detection of overlapping sources using convolutional recurrent neural networks. IEEE J Sel Top Signal Process 13(1):34\u201348","DOI":"10.1109\/JSTSP.2018.2885636"},{"key":"11183_CR3","doi-asserted-by":"crossref","unstructured":"Adavanne S, Politis A, Virtanen T (2019b) Localization, detection and tracking of multiple moving sound sources with a convolutional recurrent neural network. In: Proceedings of the workshop on detection and classification of acoustic scenes and events (DCASE)","DOI":"10.33682\/xb0q-a335"},{"key":"11183_CR4","doi-asserted-by":"crossref","unstructured":"Adavanne S, Politis A, Virtanen T (2021) Differentiable tracking-based training of deep learning sound source localizers. In: IEEE workshop on applications of signal processing to audio and acoustics (WASPAA), pp 211\u2013215","DOI":"10.1109\/WASPAA52581.2021.9632773"},{"issue":"1","key":"11183_CR5","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1016\/j.csl.2015.03.003","volume":"34","author":"S Argentieri","year":"2015","unstructured":"Argentieri S, Dan\u00e8s P, Sou\u00e8res P (2015) A survey on sound source localization in robotics: from binaural to array processing methods. Comput Speech Lang 34(1):87\u2013112","journal-title":"Comput Speech Lang"},{"key":"11183_CR6","unstructured":"Brandstein MS, Silverman HF (1997) A robust method for speech signal time-delay estimation in reverberant rooms. In: IEEE international conference on acoustics, speech, and signal processing (ICASSP), vol\u00a01, pp 375\u2013378"},{"key":"11183_CR7","doi-asserted-by":"crossref","unstructured":"Chakrabarty S, Habets EA (2017a) Broadband doa estimation using convolutional neural networks trained with noise signals. In: IEEE workshop on applications of signal processing to audio and acoustics (WASPAA), pp 136\u2013140","DOI":"10.1109\/WASPAA.2017.8170010"},{"key":"11183_CR8","unstructured":"Chakrabarty S, Habets EA (2017b) Multi-speaker localization using convolutional neural network trained with noise. arXiv preprint arXiv:1712.04276"},{"key":"11183_CR9","doi-asserted-by":"crossref","unstructured":"Chakrabarty S, Habets EA (2019) Multi-speaker DOA estimation using deep convolutional networks trained with noise signals. IEEE J Sel Top Signal Process 13(1):8\u201321","DOI":"10.1109\/JSTSP.2019.2901664"},{"key":"11183_CR10","doi-asserted-by":"crossref","unstructured":"Deng J, Guo J, Ververas E, Kotsia I, Zafeiriou S (2020) Retinaface: Single-shot multi-level face localisation in the wild. In: IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 5203\u20135212","DOI":"10.1109\/CVPR42600.2020.00525"},{"key":"11183_CR11","doi-asserted-by":"crossref","unstructured":"DiBiase JH, Silverman HF, Brandstein MS (2001) Robust localization in reverberant rooms. In: Microphone arrays, Springer, pp 157\u2013180","DOI":"10.1007\/978-3-662-04619-7_8"},{"issue":"8","key":"11183_CR12","doi-asserted-by":"publisher","first-page":"2510","DOI":"10.1109\/TASL.2007.906694","volume":"15","author":"JP Dmochowski","year":"2007","unstructured":"Dmochowski JP, Benesty J, Affes S (2007) A generalized steered response power method for computationally viable source localization. IEEE Trans Audio Speech Lang Process 15(8):2510\u20132526","journal-title":"IEEE Trans Audio Speech Lang Process"},{"issue":"4","key":"11183_CR13","doi-asserted-by":"publisher","first-page":"109:1","DOI":"10.1145\/3197517.3201357","volume":"37","author":"A Ephrat","year":"2018","unstructured":"Ephrat A, Mosseri I, Lang O, Dekel T, Wilson K, Hassidim A, Freeman WT, Rubinstein M (2018) Looking to listen at the cocktail party: a speaker-independent audio-visual model for speech separation. ACM Trans Graph 37(4):109:1-109:11","journal-title":"ACM Trans Graph"},{"issue":"1","key":"11183_CR14","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1121\/10.0011809","volume":"152","author":"PA Grumiaux","year":"2022","unstructured":"Grumiaux PA, Kiti\u0107 S, Girin L, Gu\u00e9rin A (2022) A survey of sound source localization with deep learning methods. J Acoust Soc Am 152(1):107\u2013151","journal-title":"J Acoust Soc Am"},{"key":"11183_CR15","doi-asserted-by":"crossref","unstructured":"Hartley R, Zisserman A (2003) Multiple view geometry in computer vision. Cambridge University Press, Cambridge","DOI":"10.1017\/CBO9780511811685"},{"key":"11183_CR16","doi-asserted-by":"crossref","unstructured":"He W, Motl\u00edcek P, Odobez J (2018) Deep neural networks for multiple speaker detection and localization. In: IEEE international conference on robotics and automation (ICRA), pp 74\u201379","DOI":"10.1109\/ICRA.2018.8461267"},{"key":"11183_CR17","unstructured":"Hirvonen T (2015) Classification of spatial audio location and content using convolutional neural networks. Audio Eng Soc Conv 138:1\u201310"},{"key":"11183_CR18","doi-asserted-by":"crossref","unstructured":"Jarrett DP, Habets EA, Naylor PA (2017) Theory and applications of spherical microphone array processing, vol 9. Springer, New York","DOI":"10.1007\/978-3-319-42211-4"},{"issue":"3","key":"11183_CR19","doi-asserted-by":"publisher","first-page":"241","DOI":"10.3758\/BF03203206","volume":"17","author":"B Jones","year":"1975","unstructured":"Jones B, Kabanoff B (1975) Eye movements in auditory space perception. Percept Psychophys 17(3):241\u2013245","journal-title":"Percept Psychophys"},{"key":"11183_CR20","doi-asserted-by":"crossref","unstructured":"Kim Y, Ling H (2011) Direction of arrival estimation of humans with a small sensor array using an artificial neural network. Prog Electromagn Res B 27:127\u2013149","DOI":"10.2528\/PIERB10100510"},{"key":"11183_CR21","unstructured":"Kingma DP, Ba JL (2015) Adam: a method for stochastic optimization. In: International conference on learning representations (ICLR)"},{"issue":"4","key":"11183_CR22","doi-asserted-by":"publisher","first-page":"320","DOI":"10.1109\/TASSP.1976.1162830","volume":"24","author":"CH Knapp","year":"1976","unstructured":"Knapp CH, Carter GC (1976) The generalized correlation method for estimation of time delay. IEEE Trans Acoust Speech Signal Process 24(4):320\u2013327","journal-title":"IEEE Trans Acoust Speech Signal Process"},{"issue":"1","key":"11183_CR23","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1121\/1.381498","volume":"62","author":"GF Kuhn","year":"1977","unstructured":"Kuhn GF (1977) Model for the interaural time differences in the azimuthal plane. J Acoust Soc Am 62(1):157\u2013167","journal-title":"J Acoust Soc Am"},{"key":"11183_CR24","doi-asserted-by":"crossref","unstructured":"Liaquat MU, Munawar HS, Rahman A, Qadir Z, Kouzani AZ, Mahmud MAP (2021) Localization of sound sources: a systematic review. Energies 14(13):1\u201317","DOI":"10.3390\/en14133910"},{"key":"11183_CR25","doi-asserted-by":"crossref","unstructured":"Nguyen TNT, Nguyen NK, Phan H, Pham L, Ooi K, Jones DL, Gan WS (2021) A general network architecture for sound event localization and detection using transfer learning and recurrent neural network. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 935\u2013939","DOI":"10.1109\/ICASSP39728.2021.9414602"},{"key":"11183_CR26","doi-asserted-by":"crossref","unstructured":"Politis A, Mesaros A, Adavanne S, Heittola T, Virtanen T (2021) Overview and evaluation of sound event localization and detection in DCASE 2019. IEEE\/ACM Trans Audio Speech Lang Process 29:684\u2013698","DOI":"10.1109\/TASLP.2020.3047233"},{"key":"11183_CR27","doi-asserted-by":"crossref","unstructured":"Pulkki V, Delikaris-Manias S, Politis A (2017) Parametric time-frequency domain spatial audio. Wiley, Hoboken","DOI":"10.1002\/9781119252634"},{"key":"11183_CR28","doi-asserted-by":"crossref","unstructured":"Qian X, Xompero A, Brutti A, Lanz O, Omologo M, Cavallaro A (2018) 3d mouth tracking from a compact microphone array co-located with a camera. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 3071\u20133075","DOI":"10.1109\/ICASSP.2018.8461323"},{"key":"11183_CR29","doi-asserted-by":"publisher","first-page":"1405","DOI":"10.1109\/LSP.2021.3092959","volume":"28","author":"X Qian","year":"2021","unstructured":"Qian X, Liu Q, Wang J, Li H (2021) Three-dimensional speaker localization: audio-refined visual scaling factor estimation. IEEE Signal Process Lett 28:1405\u20131409","journal-title":"IEEE Signal Process Lett"},{"key":"11183_CR30","doi-asserted-by":"crossref","unstructured":"Qian X, Madhavi M, Pan Z, Wang J, Li H (2021b) Multi-target DoA estimation with an audio-visual fusion mechanism. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4280\u20134284","DOI":"10.1109\/ICASSP39728.2021.9413776"},{"key":"11183_CR31","doi-asserted-by":"crossref","unstructured":"Rascon C, Meza I (2017) Localization of sound sources in robotics: a review. Robot Auton Syst 96:184\u2013210","DOI":"10.1016\/j.robot.2017.07.011"},{"issue":"3","key":"11183_CR32","doi-asserted-by":"publisher","first-page":"276","DOI":"10.1109\/TAP.1986.1143830","volume":"34","author":"RO Schmidt","year":"1986","unstructured":"Schmidt RO (1986) Multiple emitter location and signal parameter estimation. IEEE Trans Antennas Propag 34(3):276\u2013280","journal-title":"IEEE Trans Antennas Propag"},{"key":"11183_CR33","doi-asserted-by":"crossref","unstructured":"Senocak A, Oh TH, Kim J, Yang MH, Kweon IS (2018) Learning to localize sound source in visual scenes. In: IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 4358\u20134366","DOI":"10.1109\/CVPR.2018.00458"},{"key":"11183_CR34","doi-asserted-by":"crossref","unstructured":"Thomas F, Ros L (2005) Revisiting trilateration for robot localization. IEEE Trans Rob 21(1):93\u2013101","DOI":"10.1109\/TRO.2004.833793"},{"key":"11183_CR35","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Ukaszkaiser L, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30:1\u201311"},{"issue":"1","key":"11183_CR36","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1109\/TASLP.2018.2876169","volume":"27","author":"ZQ Wang","year":"2018","unstructured":"Wang ZQ, Zhang X, Wang D (2018) Robust speaker localization guided by deep learning-based time-frequency masking. IEEE\/ACM Trans Audio Speech Lang Process 27(1):178\u2013188","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"issue":"3","key":"11183_CR37","doi-asserted-by":"publisher","first-page":"1648","DOI":"10.1121\/1.402445","volume":"91","author":"FL Wightman","year":"1992","unstructured":"Wightman FL, Kistler DJ (1992) The dominant role of low-frequency interaural time differences in sound localization. J Acoust Soc Am 91(3):1648\u20131661","journal-title":"J Acoust Soc Am"},{"issue":"6","key":"11183_CR38","doi-asserted-by":"publisher","first-page":"3912","DOI":"10.1121\/1.5042222","volume":"143","author":"A Xenaki","year":"2018","unstructured":"Xenaki A, Boldt JB, Christensen MG (2018) Sound source localization and speech enhancement with sparse Bayesian learning beamforming. J Acoust Soc Am 143(6):3912\u20133921","journal-title":"J Acoust Soc Am"},{"key":"11183_CR39","doi-asserted-by":"crossref","unstructured":"Xiao X, Zhao S, Zhong X, Jones DL, Chng ES, Li H (2015) A learning-based approach to direction of arrival estimation in noisy and reverberant environments. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 2814\u20132818","DOI":"10.1109\/ICASSP.2015.7178484"},{"key":"11183_CR40","doi-asserted-by":"crossref","unstructured":"Zotter F, Frank M (2019) Ambisonics: a practical 3D audio theory for recording, studio production, sound reinforcement, and virtual reality, vol 19. Springer, New York","DOI":"10.1007\/978-3-030-17207-7"}],"container-title":["Neural Processing Letters"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11063-023-11183-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11063-023-11183-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11063-023-11183-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,11]],"date-time":"2023-11-11T17:05:51Z","timestamp":1699722351000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11063-023-11183-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,21]]},"references-count":40,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["11183"],"URL":"https:\/\/doi.org\/10.1007\/s11063-023-11183-7","relation":{},"ISSN":["1370-4621","1573-773X"],"issn-type":[{"value":"1370-4621","type":"print"},{"value":"1573-773X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,2,21]]},"assertion":[{"value":"5 February 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}