{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T12:17:17Z","timestamp":1756383437128,"version":"3.44.0"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2025,8,1]],"date-time":"2025-08-01T00:00:00Z","timestamp":1754006400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,8,13]],"date-time":"2025-08-13T00:00:00Z","timestamp":1755043200000},"content-version":"vor","delay-in-days":12,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62071302"],"award-info":[{"award-number":["62071302"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. King Saud Univ. Comput. Inf. Sci."],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s44443-025-00181-5","type":"journal-article","created":{"date-parts":[[2025,8,13]],"date-time":"2025-08-13T10:09:43Z","timestamp":1755079783000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Revisiting SSL for sound event detection: complementary fusion and adaptive post-processing"],"prefix":"10.1007","volume":"37","author":[{"given":"Hanfang","family":"Cui","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Longfei","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongxing","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanhua","family":"Long","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,13]]},"reference":[{"doi-asserted-by":"publisher","unstructured":"Ashraf K, Elizalde B, Iandola F, Moskewicz M, Bernd J, Friedland G, Keutzer K (2015) Audio-based multimedia event detection with dnns sparse sampling. In: Proceedings of ICMR, pp 611\u2013614. https:\/\/doi.org\/10.1145\/2671188.2749396","key":"181_CR1","DOI":"10.1145\/2671188.2749396"},{"unstructured":"Baevski A, Zhou Y, Mohamed A, Auli M (2020) wav2vec 2.0: A framework for self-supervised learning of speech representations. In: Proceedings of NeurIPS, vol 33, pp 12449\u201312460","key":"181_CR2"},{"doi-asserted-by":"publisher","unstructured":"Bello JP, Mydlarz C, Salamon J (2018) Sound analysis in smart cities. In: Virtanen T, Plumbley MD, Ellis D (eds) Computational Analysis of Sound Scenes and Events, pp. 373\u2013397. Springer, Cham. https:\/\/doi.org\/10.1007\/978-3-319-63450-0_13","key":"181_CR3","DOI":"10.1007\/978-3-319-63450-0_13"},{"issue":"2","key":"181_CR4","doi-asserted-by":"publisher","first-page":"68","DOI":"10.1145\/3224204","volume":"62","author":"JP Bello","year":"2019","unstructured":"Bello JP, Silva C, Nov O, Dubois RL, Arora A, Salamon J, Mydlarz C, Doraiswamy H (2019) Sonyc: a system for monitoring, analyzing, and mitigating urban noise pollution. Commun ACM 62(2):68\u201377. https:\/\/doi.org\/10.1145\/3224204","journal-title":"Commun ACM"},{"doi-asserted-by":"publisher","unstructured":"Bilen \u00c7, Ferroni G, Tuveri F, Azcarreta J, Krstulovi\u0107 S (2020) A framework for the robust evaluation of sound event detection. In: Proceedings of ICASSP, pp 61\u201365. https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9052995","key":"181_CR5","DOI":"10.1109\/ICASSP40776.2020.9052995"},{"doi-asserted-by":"publisher","unstructured":"Cai P, Song Y, Li K, Song H, McLoughlin I (2024) MAT-SED: a masked audio transformer with masked-reconstruction based pre-training for sound event detection. In: Proceedings of interspeech, pp 557\u2013561. https:\/\/doi.org\/10.21437\/Interspeech.2024-714","key":"181_CR6","DOI":"10.21437\/Interspeech.2024-714"},{"doi-asserted-by":"publisher","unstructured":"Chen S, Wang C, Chen Z, Wu Y, Liu S, Chen Z, Li J, Kanda N, Yoshioka T, Xiao X, Wu J, Zhou L, Ren S, Qian Y, Qian Y, Zeng M, Yu X, Wei F (2022) WavLM: large-scale self-supervised pre-training for full stack speech processing. IEEE J Sel Top Signal Process 16(6):1505\u20131518. https:\/\/doi.org\/10.1109\/JSTSP.2022.3188113","key":"181_CR7","DOI":"10.1109\/JSTSP.2022.3188113"},{"unstructured":"Chen S, Wu Y, Wang C, Liu S, Tompkins D, Chen Z, Che W, Yu X, Wei F (2022) BEATs: audio pre-training with acoustic tokenizers. arXiv:2212.09058","key":"181_CR8"},{"key":"181_CR9","first-page":"20","volume":"400","author":"B Clarkson","year":"1998","unstructured":"Clarkson B, Pentl A, Sawhney N (1998) Auditory context awareness via wearable computing. Energy 400:20","journal-title":"Energy"},{"unstructured":"DCASE Community (2023) DCASE Event Detection with Weak Soundscapes. https:\/\/dcase.community\/challenge2023\/task-sound-event-detection-with-weak-labels-and-synthetic-soundscapes. Accessed 12-Oct-2024","key":"181_CR10"},{"issue":"2","key":"181_CR11","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1109\/MSP.2015.2503881","volume":"33","author":"C Debes","year":"2016","unstructured":"Debes C, Merentitis A, Sukhanov S, Niessen M, Frangiadakis N, Bauer A (2016) Monitoring activities of daily living in smart homes: understanding human behavior. IEEE Signal Process Mag 33(2):81\u201394. https:\/\/doi.org\/10.1109\/MSP.2015.2503881","journal-title":"IEEE Signal Process Mag"},{"doi-asserted-by":"publisher","unstructured":"Dinkel H, Yan Z, Wang Y, Zhang J, Wang Y, Wang B (2024) Scaling up masked audio encoder learning for general audio classification. In: Proceedings of interspeech, pp 547\u2013551. https:\/\/doi.org\/10.21437\/Interspeech.2024-246","key":"181_CR12","DOI":"10.21437\/Interspeech.2024-246"},{"unstructured":"Dohi K, Imoto K, Harada N, Niizumi D, Koizumi Y, Nishida T, Purohit H, Tanabe R, Endo T, Kawaguchi Y (2023) Description and discussion on DCASE 2023 Challenge Task 2: first-shot unsupervised anomalous sound detection for machine condition monitoring. arXiv:2305.07828","key":"181_CR13"},{"doi-asserted-by":"publisher","unstructured":"Ebbers J, Germain FG, Wichern G, Le\u00a0Roux J (2024) Sound event bounding boxes. In: Proceedings of interspeech, pp 562\u2013566. https:\/\/doi.org\/10.21437\/Interspeech.2024-2075","key":"181_CR14","DOI":"10.21437\/Interspeech.2024-2075"},{"unstructured":"Ebbers J, Haeb-Umbach R (2022) Pre-training and self-training for sound event detection in domestic environments. Technical report, DCASE2022 Challenge","key":"181_CR15"},{"doi-asserted-by":"publisher","unstructured":"Gong Y, Chung Y-A, Glass J (2021) AST: audio spectrogram transformer. In: Proceedings of Interspeech, pp 571\u2013575. https:\/\/doi.org\/10.21437\/Interspeech.2021-698","key":"181_CR16","DOI":"10.21437\/Interspeech.2021-698"},{"doi-asserted-by":"publisher","unstructured":"He K, Chen X, Xie S, Li Y, Doll\u00e1r P, Girshick R (2022) Masked autoencoders are scalable vision learners. In: Proceedings of CVPR, pp 15979\u201315988. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01553","key":"181_CR17","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"181_CR18","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"W-N Hsu","year":"2021","unstructured":"Hsu W-N, Bolte B, Tsai Y-HH, Lakhotia K, Salakhutdinov R, Mohamed A (2021) HuBERT: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans Audio Speech Lang Process 29:3451\u20133460. https:\/\/doi.org\/10.1109\/TASLP.2021.3122291","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"issue":"1","key":"181_CR19","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1561\/116.00000051","volume":"13","author":"T Khandelwal","year":"2024","unstructured":"Khandelwal T, Das RK, Chng ES (2024) Sound event detection: a journey through dcase challenge series. APSIPA Trans Signal Inf Process 13(1):51. https:\/\/doi.org\/10.1561\/116.00000051","journal-title":"APSIPA Trans Signal Inf Process"},{"unstructured":"Kim JW, Son SW, Song Y, Kim HK, Song IH, Lim JE (2023) Semi-supervised learning-based sound event detection using frequency dynamic convolution with large kernel attention for DCASE Challenge 2023 Task 4. arXiv:2306.06461","key":"181_CR20"},{"key":"181_CR21","doi-asserted-by":"publisher","first-page":"2880","DOI":"10.1109\/TASLP.2020.3030497","volume":"28","author":"Q Kong","year":"2020","unstructured":"Kong Q, Cao Y, Iqbal T, Wang Y, Wang W, Plumbley MD (2020) PANNs: large-scale pretrained audio neural networks for audio pattern recognition. IEEE\/ACM Trans Audio Speech Lang Process 28:2880\u20132894. https:\/\/doi.org\/10.1109\/TASLP.2020.3030497","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"issue":"1","key":"181_CR22","doi-asserted-by":"publisher","first-page":"98","DOI":"10.2307\/3213263","volume":"14","author":"AJ Lawrance","year":"1977","unstructured":"Lawrance AJ, Lewis PAW (1977) An exponential moving-average sequence and point process (ema1). J Appl Prob 14(1):98\u2013113. https:\/\/doi.org\/10.2307\/3213263","journal-title":"J Appl Prob"},{"issue":"11","key":"181_CR23","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun Y, Bottou L, Bengio Y, Haffner P (1998) Gradient-based learning applied to document recognition. Proc IEEE 86(11):2278\u20132324. https:\/\/doi.org\/10.1109\/5.726791","journal-title":"Proc IEEE"},{"doi-asserted-by":"publisher","unstructured":"Li X, Shao N, Li X (2024) Self-supervised audio teacher-student transformer for both clip-level and frame-level tasks. IEEE\/ACM Trans Audio Speech Lang Process 32:1336\u20131351. https:\/\/doi.org\/10.1109\/TASLP.2024.3352248","key":"181_CR24","DOI":"10.1109\/TASLP.2024.3352248"},{"key":"181_CR25","doi-asserted-by":"publisher","first-page":"103446","DOI":"10.1016\/j.dsp.2022.103446","volume":"123","author":"Y Liang","year":"2022","unstructured":"Liang Y, Long Y, Li Y, Liang J, Wang Y (2022) Joint framework with deep feature distillation and adaptive focal loss for weakly supervised audio tagging and acoustic event detection. Digit Signal Process 123:103446. https:\/\/doi.org\/10.1016\/j.dsp.2022.103446","journal-title":"Digit Signal Process"},{"unstructured":"Liu Q, Song L, Xu D, Long Y (2024) ICSD: an open-source dataset for infant cry and snoring detection. arXiv:2408.10561","key":"181_CR26"},{"unstructured":"Lv Z, Han B, Chen Z, Qian Y, Ding J, Liu J (2023) Unsupervised anomalous detection based on unsupervised pretrained models. Technical report, DCASE2023 Challenge","key":"181_CR27"},{"issue":"86","key":"181_CR28","first-page":"2579","volume":"9","author":"L Maaten","year":"2008","unstructured":"Maaten L, Hinton G (2008) Visualizing data using t-SNE. J Mach Learn Res 9(86):2579\u20132605","journal-title":"J Mach Learn Res"},{"doi-asserted-by":"publisher","unstructured":"Nam H, Kim S-H, Ko B-Y, Park Y-H (2022) Frequency dynamic convolution: frequency-adaptive pattern recognition for sound event detection. In: Proceedings of interspeech, pp 2763\u20132767. https:\/\/doi.org\/10.21437\/Interspeech.2022-10127","key":"181_CR29","DOI":"10.21437\/Interspeech.2022-10127"},{"doi-asserted-by":"publisher","unstructured":"Oord A, Li Y, Vinyals O (2018) Representation learning with contrastive predictive coding. https:\/\/doi.org\/10.48550\/arXiv.1807.03748arXiv:1807.03748","key":"181_CR30","DOI":"10.48550\/arXiv.1807.03748"},{"doi-asserted-by":"publisher","unstructured":"Schneider S, Baevski A, Collobert R, Auli M (2019) wav2vec: unsupervised pre-training for speech recognition. In: Proceedings of interspeech, pp 3465\u20133469. https:\/\/doi.org\/10.21437\/Interspeech.2019-1873","key":"181_CR31","DOI":"10.21437\/Interspeech.2019-1873"},{"doi-asserted-by":"publisher","unstructured":"Shao N, Li X, Li X (2024) Fine-tune the pretrained atst model for sound event detection. In: Proceedings of ICASSP, pp 911\u2013915. https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10446159","key":"181_CR32","DOI":"10.1109\/ICASSP48485.2024.10446159"},{"issue":"11","key":"181_CR33","doi-asserted-by":"publisher","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","volume":"39","author":"B Shi","year":"2016","unstructured":"Shi B, Bai X, Yao C (2016) An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. IEEE Trans Pattern Anal Mach Intell 39(11):2298\u20132304. https:\/\/doi.org\/10.1109\/TPAMI.2016.2646371","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"doi-asserted-by":"publisher","unstructured":"Singh U, Dash DD, Sharma M, Mishra S, Malarvizhi S, Tiwari S, Shankarappa RT (2021) Polyphonic sound event detection and classification using convolutional recurrent neural network with mean teacher. In: Proceedings of ICCCNT, pp 1\u20134. https:\/\/doi.org\/10.1109\/ICCCNT51525.2021.9579677","key":"181_CR34","DOI":"10.1109\/ICCCNT51525.2021.9579677"},{"key":"181_CR35","doi-asserted-by":"publisher","first-page":"102460","DOI":"10.1016\/j.inffus.2024.102460","volume":"110","author":"H Tang","year":"2024","unstructured":"Tang H, Hu Y, Wang Y, Zhang S, Xu M, Zhu J, Zheng Q (2024) Listen as you wish: Fusion of audio and text for cross-modal event detection in smart cities. Inf Fusion 110:102460. https:\/\/doi.org\/10.1016\/j.inffus.2024.102460","journal-title":"Inf Fusion"},{"unstructured":"Tarvainen A, Valpola H (2017) Mean teachers are better role models: weight-averaged consistency targets improve semi-supervised deep learning results. In: Proceedings of NeurIPS, vol 30, pp 1195\u20131204","key":"181_CR36"},{"key":"181_CR37","first-page":"125","volume":"176","author":"J Turian","year":"2022","unstructured":"Turian J, Shier J, Khan HR, Raj B, Schuller BW, Steinmetz CJ, Malloy C, Tzanetakis G, Velarde G, McNally K et al (2022) HEAR: holistic evaluation of audio representations. Proc Mach Learn Res 176:125\u2013145","journal-title":"Proc Mach Learn Res"},{"doi-asserted-by":"publisher","unstructured":"Turpault N, Serizel R, Shah AP, Salamon J (2019) Sound event detection in domestic environments with weakly labeled data and soundscape synthesis. In: Workshop on detection and classification of acoustic scenes and events (DCASE), pp 253\u2013257. https:\/\/doi.org\/10.33682\/006b-jx26","key":"181_CR38","DOI":"10.33682\/006b-jx26"},{"unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. In: Proceedings of NeurIPS, pp 6000\u20136010","key":"181_CR39"},{"doi-asserted-by":"publisher","unstructured":"Wang Y, Zheng H, Sun Q, Ma Y, Zhu S, Zhang L, Zhang W-Q (2024) Cross-lingual alzheimer\u2019s disease detection based on scale criteria. In: Proceedings of ISCSLP, pp 491\u2013495. https:\/\/doi.org\/10.1109\/ISCSLP63861.2024.10800047","key":"181_CR40","DOI":"10.1109\/ISCSLP63861.2024.10800047"},{"doi-asserted-by":"publisher","unstructured":"Yin H, Chen J, Bai J, Wang M, Rahardja S, Shi D, Gan W-s (2025) Multi-granularity acoustic information fusion for sound event detection. Signal Process 227:109691. https:\/\/doi.org\/10.1016\/j.sigpro.2024.109691","key":"181_CR41","DOI":"10.1016\/j.sigpro.2024.109691"},{"issue":"2","key":"181_CR42","doi-asserted-by":"publisher","first-page":"294","DOI":"10.23919\/JSEE.2023.000110","volume":"35","author":"D Zhao","year":"2024","unstructured":"Zhao D, Ding K, Qi X, Chen Y, Feng H (2024) Sound event localization and detection based on deep learning. J Syst Eng Electron 35(2):294\u2013301. https:\/\/doi.org\/10.23919\/JSEE.2023.000110","journal-title":"J Syst Eng Electron"},{"doi-asserted-by":"publisher","unstructured":"Zheng Y, Zhang R, Atito S, Yang S, Wang W, Mei Y (2025) ASiT-CRNN: a method for sound event detection with fine-tuning of self-supervised pre-trained asit-based model. Digit Signal Process 160:105055. https:\/\/doi.org\/10.1016\/j.dsp.2025.105055","key":"181_CR43","DOI":"10.1016\/j.dsp.2025.105055"},{"doi-asserted-by":"publisher","unstructured":"Zheng X, Song Y, Dai L-R, McLoughlin I, Liu L (2021) An effective mutual mean teaching based domain adaptation method for sound event detection. In: Proceedings of interspeech, pp 556\u2013560. https:\/\/doi.org\/10.21437\/Interspeech.2021-281","key":"181_CR44","DOI":"10.21437\/Interspeech.2021-281"}],"container-title":["Journal of King Saud University Computer and Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44443-025-00181-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s44443-025-00181-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s44443-025-00181-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,28]],"date-time":"2025-08-28T11:43:38Z","timestamp":1756381418000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s44443-025-00181-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8]]},"references-count":44,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["181"],"URL":"https:\/\/doi.org\/10.1007\/s44443-025-00181-5","relation":{},"ISSN":["1319-1578","2213-1248"],"issn-type":[{"type":"print","value":"1319-1578"},{"type":"electronic","value":"2213-1248"}],"subject":[],"published":{"date-parts":[[2025,8]]},"assertion":[{"value":"14 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 July 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 August 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interest"}}],"article-number":"160"}}