{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T17:19:41Z","timestamp":1783185581157,"version":"3.54.6"},"reference-count":26,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100002367","name":"Chinese Academy of Sciences","doi-asserted-by":"publisher","award":["2021233"],"award-info":[{"award-number":["2021233"]}],"id":[{"id":"10.13039\/501100002367","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004739","name":"Chinese Academy of Sciences Youth Innovation Promotion Association","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004739","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012492","name":"Youth Innovation Promotion Association","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012492","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62303441"],"award-info":[{"award-number":["62303441"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition Letters"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.patrec.2026.05.012","type":"journal-article","created":{"date-parts":[[2026,5,16]],"date-time":"2026-05-16T15:24:21Z","timestamp":1778945061000},"page":"69-74","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Learning to chase: Adaptive audio-visual navigation for moving sounds in complex environments"],"prefix":"10.1016","volume":"206","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-3841-4913","authenticated-orcid":false,"given":"Yuanzheng","family":"He","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5927-0000","authenticated-orcid":false,"given":"Yuyi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8960-8436","authenticated-orcid":false,"given":"Chenfan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1579-3942","authenticated-orcid":false,"given":"Dongchen","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6655-188X","authenticated-orcid":false,"given":"Lei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patrec.2026.05.012_bib0001","series-title":"Proc. ECCV","first-page":"17","article-title":"Soundspaces: audio-visual navigation in 3d environments","author":"Chen","year":"2020"},{"key":"10.1016\/j.patrec.2026.05.012_bib0002","series-title":"Proc. ICRA","first-page":"9701","article-title":"Look, listen, and act: towards audio-visual embodied navigation","author":"Gan","year":"2020"},{"key":"10.1016\/j.patrec.2026.05.012_bib0003","series-title":"Proc. ICCV","first-page":"9339","article-title":"Habitat: a platform for embodied ai research","author":"Savva","year":"2019"},{"key":"10.1016\/j.patrec.2026.05.012_bib0004","doi-asserted-by":"crossref","unstructured":"C. Chen, C. Schissler, S. Garg, et al., Soundspaces 2.0: a simulation platform for visual-acoustic learning, Adv. NeurIPS2022, 35, 8896\u20138911. 10.52202\/068431-0647.","DOI":"10.52202\/068431-0647"},{"key":"10.1016\/j.patrec.2026.05.012_bib0005","unstructured":"C. Chen, S. Majumder, Z. Al-Halah, et al., Learning to set waypoints for audio-visual navigation, (2020) arXiv preprint arXiv: 2008.09622."},{"key":"10.1016\/j.patrec.2026.05.012_bib0006","first-page":"928","article-title":"Catch me if you hear me: audio-visual navigation in complex unmapped environments with moving sounds","author":"Younes","year":"2023","journal-title":"IEEE RAL"},{"key":"10.1016\/j.patrec.2026.05.012_bib0007","unstructured":"Y. Yu, L. Cao, F. Sun, et al., Pay self-attention to audio-visual navigation, (2022). 10.48550\/arXiv.2210.01353."},{"key":"10.1016\/j.patrec.2026.05.012_bib0008","doi-asserted-by":"crossref","first-page":"184","DOI":"10.1016\/j.robot.2017.07.011","article-title":"Localization of sound sources in robotics: a review","author":"Rascon","year":"2017","journal-title":"Robot. Auton. Syst."},{"key":"10.1016\/j.patrec.2026.05.012_bib0009","series-title":"Proc. AAAI","first-page":"832","article-title":"Active audition for humanoid","author":"Nakadai","year":"2000"},{"key":"10.1016\/j.patrec.2026.05.012_bib0010","doi-asserted-by":"crossref","DOI":"10.3390\/s22031011","article-title":"3D Multiple sound source localization by proposed T-shaped circular distributed microphone arrays in combination with GEVD and adaptive GCC-PHAT\/ML algorithms","volume":"22","author":"Dehghan Firoozabadi","year":"2022","journal-title":"Sensors"},{"key":"10.1016\/j.patrec.2026.05.012_bib0011","series-title":"Proc. EUSIPCO","first-page":"1","article-title":"Dereverberation for reverberation-robust microphone arrays","author":"Yoshioka","year":"2013"},{"key":"10.1016\/j.patrec.2026.05.012_bib0012","series-title":"Proc. ICRA","first-page":"3699","article-title":"Learning to listen and move: an implementation of audio-aware mobile robot navigation in complex indoor environment","author":"Rao","year":"2022"},{"key":"10.1016\/j.patrec.2026.05.012_bib0013","series-title":"Proc. CVPR","first-page":"1029","article-title":"A proposal-Based paradigm for self-Supervised sound source localization in videos","author":"Xuan","year":"2022"},{"issue":"7","key":"10.1016\/j.patrec.2026.05.012_bib0014","doi-asserted-by":"crossref","first-page":"4896","DOI":"10.1109\/TPAMI.2024.3363508","article-title":"Robust audio-Visual contrastive learning for proposal-Based self-Supervised sound source localization in videos","volume":"46","author":"Xuan","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patrec.2026.05.012_bib0015","series-title":"Proc. IJCAI","first-page":"3643","article-title":"Active contrastive set mining for robust audio-visual instance discrimination","author":"Xuan","year":"2022"},{"key":"10.1016\/j.patrec.2026.05.012_bib0016","series-title":"Proc. CVPR","first-page":"15516","article-title":"Semantic audio-visual navigation","author":"Chen","year":"2021"},{"key":"10.1016\/j.patrec.2026.05.012_bib0017","unstructured":"G. Tatiya, J. Francis, L. Bondi, et al., Knowledge-driven scene priors for semantic audio-visual embodied navigation, (2022) arXiv preprint arXiv: 2212.11345."},{"key":"10.1016\/j.patrec.2026.05.012_bib0018","series-title":"Proc. AAAI","first-page":"3765","article-title":"Caven: an embodied conversational agent for efficient audio-visual navigation in noisy environments","author":"Liu","year":"2024"},{"key":"10.1016\/j.patrec.2026.05.012_bib0019","series-title":"Proc. AAAI","first-page":"14673","article-title":"Towards audio-Visual navigation in noisy environments: a large-Scale benchmark dataset and an architecture considering multiple sound-Sources","author":"Shi","year":"2025"},{"key":"10.1016\/j.patrec.2026.05.012_bib0020","series-title":"Proc. ICRA","first-page":"704","article-title":"Sonicverse: a multisensory simulation platform for embodied household agents that see and hear","author":"Gao","year":"2023"},{"key":"10.1016\/j.patrec.2026.05.012_bib0021","unstructured":"J. Chung, K. Kastner, L. Dinh, et al., A recurrent latent variable model for sequential data, Adv. NeurIPS, 2015, 28."},{"key":"10.1016\/j.patrec.2026.05.012_bib0022","unstructured":"J. Straub, T. Whelan, L. Ma, et al., The replica dataset: a digital replica of indoor spaces, (2019) arXiv preprint arXiv: 1906.05797."},{"key":"10.1016\/j.patrec.2026.05.012_bib0023","doi-asserted-by":"crossref","unstructured":"A. Chang, A. Dai, T. Funkhouser, et al., Matterport3d: learning from rgb-d data in indoor environments, (2017) arXiv preprint arXiv: 1709.06158. 10.1109\/3DV.2017.00081.","DOI":"10.1109\/3DV.2017.00081"},{"key":"10.1016\/j.patrec.2026.05.012_bib0024","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, et al., Attention is all you need, Adv. NeurIPS, 2017, 30."},{"key":"10.1016\/j.patrec.2026.05.012_bib0025","unstructured":"J. Schulman, F. Wolski, P. Dhariwal, et al., Proximal policy optimization algorithms, (2017) arXiv preprint arXiv: 1707.06347."},{"key":"10.1016\/j.patrec.2026.05.012_bib0026","series-title":"Proc. ICLR","article-title":"A method for stochastic optimization","author":"Kinga","year":"2015"}],"container-title":["Pattern Recognition Letters"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001790?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001790?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T17:12:04Z","timestamp":1783185124000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167865526001790"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":26,"alternative-id":["S0167865526001790"],"URL":"https:\/\/doi.org\/10.1016\/j.patrec.2026.05.012","relation":{},"ISSN":["0167-8655"],"issn-type":[{"value":"0167-8655","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Learning to chase: Adaptive audio-visual navigation for moving sounds in complex environments","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition Letters","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patrec.2026.05.012","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}]}}