{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T01:04:59Z","timestamp":1759971899869,"version":"build-2065373602"},"reference-count":47,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/access.2025.3609479","type":"journal-article","created":{"date-parts":[[2025,9,12]],"date-time":"2025-09-12T17:31:22Z","timestamp":1757698282000},"page":"171613-171625","source":"Crossref","is-referenced-by-count":0,"title":["Spatial Annotation-Free Sound Event Localization and Detection via Spatial Instance Classification"],"prefix":"10.1109","volume":"13","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2776-9701","authenticated-orcid":false,"given":"Masahiro","family":"Yasuda","sequence":"first","affiliation":[{"name":"NTT, Inc., Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1759-4533","authenticated-orcid":false,"given":"Noboru","family":"Harada","sequence":"additional","affiliation":[{"name":"NTT, Inc., Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shoichiro","family":"Saito","sequence":"additional","affiliation":[{"name":"NTT, Inc., Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4242-2773","authenticated-orcid":false,"given":"Nobutaka","family":"Ono","sequence":"additional","affiliation":[{"name":"Tokyo Metropolitan University, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746544"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446749"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/VTCSpring.2018.8417601"},{"article-title":"Surrey-CVSSP system for DCASE2017 challenge task4","year":"2017","author":"Xu","key":"ref4"},{"article-title":"Ensemble of convolutional neural networks for weakly-supervised sound event detection using multiple scale input","year":"2017","author":"Lee","key":"ref5"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/icassp39728.2021.9413609"},{"article-title":"The NERC-slip system for sound event localization and detection of DCASE2023 challenge","year":"2023","author":"Wang","key":"ref7"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746384"},{"key":"ref9","article-title":"A dataset of dynamic reverberant sound scenes with directional interferers for sound event localization and detection","author":"Politis","year":"2021","journal-title":"arXiv:2106.06999"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446118"},{"key":"ref11","article-title":"STARSS22: A dataset of spatial recordings of real scenes with spatiotemporal annotations of sound events","author":"Politis","year":"2022","journal-title":"arXiv:2206.01948"},{"key":"ref12","first-page":"72931","article-title":"STARSS23: An audio-visual dataset of spatial recordings of real scenes with spatiotemporal annotations of sound events","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shimada"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095504"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448497"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.10.009"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10889696"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3510453"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC58517.2023.10317433"},{"key":"ref19","first-page":"48639","article-title":"Dual mean-teacher: An unbiased semi-supervised framework for audio-visual source localization","volume-title":"Proc. 37th Int. Conf. Neural Inf. Process. Syst.","author":"Guo"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW63481.2024.10645397"},{"key":"ref21","first-page":"12449","article-title":"Wav2Vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst.","author":"Baevski"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_35"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/EUSIPCO.2015.7362645"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1976.1162830"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TAP.1986.1143830"},{"article-title":"DCASE 2019 TASK 3: A two-step system for sound event localization and detection","year":"2019","author":"Nguyen","key":"ref26"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054462"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.33682\/9f2t-ab23"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3173054"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746132"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1155\/S1110865703212014"},{"key":"ref32","first-page":"240","article-title":"Robust acoustic speaker localization with distributed microphones","volume-title":"Proc. 19th Eur. Signal Process. Conf.","author":"Hummes"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10097211"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/WIAMIS.2013.6616127"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2690559"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3224204"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3707447"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3256088"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.33682\/3qgs-e216"},{"volume-title":"EM32 Modal Beamformer Datasheet","key":"ref40"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3382294"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3047233"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461310"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3133208"},{"key":"ref45","first-page":"316","article-title":"FMA: A dataset for music analysis","volume-title":"Proc. 18th Int. Soc. Music Inf. Retr. Conf. (ISMIR)","author":"Defferrard"},{"key":"ref46","article-title":"Optimal scalogram for computational complexity reduction in acoustic recognition using deep learning","author":"Phan","year":"2025","journal-title":"arXiv:2505.13017"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2829405"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10820123\/11162519.pdf?arnumber=11162519","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T17:39:01Z","timestamp":1759945141000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11162519\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/access.2025.3609479","relation":{},"ISSN":["2169-3536"],"issn-type":[{"type":"electronic","value":"2169-3536"}],"subject":[],"published":{"date-parts":[[2025]]}}}