{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T08:37:52Z","timestamp":1777106272129,"version":"3.51.4"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T00:00:00Z","timestamp":1760400000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T00:00:00Z","timestamp":1760400000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s00034-025-03379-0","type":"journal-article","created":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T15:34:03Z","timestamp":1760456043000},"page":"3198-3223","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Weakly Labeled Environmental Sound Event Detection Based on Dynamic Multi-scale Convolution Attention"],"prefix":"10.1007","volume":"45","author":[{"given":"Baojun","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianxin","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,10,14]]},"reference":[{"key":"3379_CR1","doi-asserted-by":"publisher","first-page":"110794","DOI":"10.1016\/j.apacoust.2025.110794","volume":"238","author":"B Chen","year":"2025","unstructured":"B. Chen, J. Peng, Environmental sound classification using two-stream deep neural network with interactive time-frequency attention. Appl. Acoust. 238, 110794 (2025)","journal-title":"Appl. Acoust."},{"key":"3379_CR2","doi-asserted-by":"publisher","first-page":"110593","DOI":"10.1016\/j.apacoust.2025.110593","volume":"232","author":"F Chen","year":"2025","unstructured":"F. Chen, Z. Zhu, C. Sun et al., Evaluating metric and contrastive learning in pretrained models for environmental sound classification. Appl. Acoust. 232, 110593 (2025)","journal-title":"Appl. Acoust."},{"key":"3379_CR3","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1016\/j.apacoust.2018.12.019","volume":"148","author":"Y Chen","year":"2019","unstructured":"Y. Chen, Q. Guo, X. Liang et al., Environmental sound classification with dilated convolutions. Appl. Acoust. 148, 123\u2013132 (2019)","journal-title":"Appl. Acoust."},{"key":"3379_CR4","doi-asserted-by":"crossref","unstructured":"Y. Chen, X. Dai, M. Liu et al., Dynamic convolution: attention over convolution Kernels. arXiv:1912.03458 (2020)","DOI":"10.1109\/CVPR42600.2020.01104"},{"key":"3379_CR5","unstructured":"S. Cornell, J. Ebbers, C. Douwes et al., DCASE 2024 Task 4: sound event detection with heterogeneous data and missing labels. arXiv:2406.08056 (2024)"},{"key":"3379_CR6","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov et al. An image is worth 16x16 words: transformers for image recognition at scale. arXiv:2010.11929 (2021)"},{"issue":"3","key":"3379_CR7","first-page":"39","volume":"164","author":"H Farghaly","year":"2020","unstructured":"H. Farghaly, A. Ali, T. Abd El-Hafeez, Building an effective and accurate associative classifier based on support vector machine. Sylwan 164(3), 39\u201356 (2020)","journal-title":"Sylwan"},{"key":"3379_CR8","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1007\/978-3-030-51971-1_5","volume-title":"Artificial intelligence and bioinspired computational methods","author":"HM Farghaly","year":"2020","unstructured":"H.M. Farghaly, A.A. Ali, T.A. El-Hafeez, Developing an efficient method for automatic threshold detection based on hybrid feature selection approach, in Artificial intelligence and bioinspired computational methods. ed. by R. Silhavy (Springer, Cham, 2020), pp.56\u201372"},{"issue":"1","key":"3379_CR9","doi-asserted-by":"publisher","first-page":"21","DOI":"10.3390\/math7010021","volume":"7","author":"A Farina","year":"2019","unstructured":"A. Farina, Ecoacoustics: a quantitative approach to investigate the ecological role of environmental sounds. Mathematics 7(1), 21 (2019)","journal-title":"Mathematics"},{"issue":"6","key":"3379_CR10","doi-asserted-by":"publisher","first-page":"163","DOI":"10.1007\/s10462-025-11106-z","volume":"58","author":"P Gairi","year":"2025","unstructured":"P. Gairi, T. Palleja, M. Tresanchez, Environmental sound recognition on embedded devices using deep learning: a review. Artif. Intell. Rev. 58(6), 163 (2025)","journal-title":"Artif. Intell. Rev."},{"issue":"1","key":"3379_CR11","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1148\/radiology.143.1.7063747","volume":"143","author":"JA Hanley","year":"1982","unstructured":"J.A. Hanley, B.J. McNeil, The meaning and use of the area under a receiver operating characteristic (ROC) curve. Radiology 143(1), 29\u201336 (1982)","journal-title":"Radiology"},{"issue":"1","key":"3379_CR12","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1186\/s40537-024-00985-8","volume":"11","author":"E Hassan","year":"2024","unstructured":"E. Hassan, S. Elbedwehy, M.Y. Shams et al., Optimizing poultry audio signal classification with deep learning and burn layer fusion. J. Big Data 11(1), 135 (2024)","journal-title":"J. Big Data"},{"key":"3379_CR13","doi-asserted-by":"publisher","first-page":"108128","DOI":"10.1016\/j.bspc.2025.108128","volume":"110","author":"E Hassan","year":"2025","unstructured":"E. Hassan, A. Saber, T. Abd El-Hafeez et al., Enhanced dysarthria detection in cerebral palsy and ALS patients using WaveNet and CNN-BiLSTM models: a comparative study with model interpretability. Biomed. Signal Process. Control 110, 108128 (2025)","journal-title":"Biomed. Signal Process. Control"},{"issue":"1","key":"3379_CR14","doi-asserted-by":"publisher","first-page":"24358","DOI":"10.1038\/s41598-025-09643-2","volume":"15","author":"E Hassan","year":"2025","unstructured":"E. Hassan, M.Y. Shams, T. Abd El-Hafeez et al., A novel model for expanding horizons in sign Language recognition. Sci. Rep. 15(1), 24358 (2025)","journal-title":"Sci. Rep."},{"key":"3379_CR15","doi-asserted-by":"crossref","unstructured":"Q. Hou, D. Zhou, J. Feng, Coordinate attention for efficient mobile network design. arXiv:2103.02907 (2021)","DOI":"10.1109\/CVPR46437.2021.01350"},{"key":"3379_CR16","unstructured":"A.G. Howard, M. Zhu, B. Chen et al., MobileNets: efficient convolutional neural networks for mobile vision applications. arXiv:1704.04861 (2017)"},{"key":"3379_CR17","unstructured":"J. Hu, L. Shen, S. Albanie et al., Squeeze-and-Excitation Networks. arXiv:1709.01507 (2019)"},{"issue":"5","key":"3379_CR18","doi-asserted-by":"publisher","first-page":"1045","DOI":"10.3390\/sym15051045","volume":"15","author":"M Huang","year":"2023","unstructured":"M. Huang, M. Wang, X. Liu et al., Environmental sound classification framework based on L-mHP features and SE-ResNet50 network model. Symmetry-Basel 15(5), 1045 (2023)","journal-title":"Symmetry-Basel"},{"key":"3379_CR19","doi-asserted-by":"publisher","first-page":"695","DOI":"10.1007\/978-3-319-46493-0_42","volume-title":"Computer Vision - ECCV 2016","author":"A Kolesnikov","year":"2016","unstructured":"A. Kolesnikov, C.H. Lampert, Seed, Expand and Constrain: Three Principles for Weakly-Supervised Image Segmentation, in Computer Vision - ECCV 2016. ed. by B. Leibe, J. Matas, N. Sebe et al. (Springer, Cham, 2016), pp.695\u2013711"},{"issue":"4","key":"3379_CR20","doi-asserted-by":"publisher","first-page":"777","DOI":"10.1109\/TASLP.2019.2895254","volume":"27","author":"Q Kong","year":"2019","unstructured":"Q. Kong, Y. Xu, I. Sobieraj et al., Sound event detection and time-frequency segmentation from weakly labelled data. IEEE-ACM Trans. Audio Speech Lang Process. 27(4), 777\u2013787 (2019)","journal-title":"IEEE-ACM Trans. Audio Speech Lang Process."},{"key":"3379_CR21","doi-asserted-by":"crossref","unstructured":"A. Kumar, B. Raj, Audio event detection using weakly labeled data. In: Proceedings of the 24th ACM international conference on Multimedia. ACM, Amsterdam The Netherlands, pp. 1038\u20131047 (2016)","DOI":"10.1145\/2964284.2964310"},{"key":"3379_CR22","unstructured":"A. Kumar, B. Raj, Deep CNN framework for audio event recognition using weakly labeled web data. arXiv:1707.02530 (2022)"},{"issue":"4","key":"3379_CR23","doi-asserted-by":"publisher","first-page":"1885","DOI":"10.3390\/app15041885","volume":"15","author":"W Li","year":"2025","unstructured":"W. Li, D. Lv, Y. Yu et al., Multi-scale deep feature fusion with machine learning classifier for birdsong classification. Appl. Sci. -Basel 15(4), 1885 (2025)","journal-title":"Appl. Sci. -Basel"},{"key":"3379_CR24","unstructured":"M. Lin, Q. Chen, S. Yan, Network in network. arXiv:1312.4400 (2014)"},{"key":"3379_CR25","doi-asserted-by":"crossref","unstructured":"S. Liu, F. Yang, F. Kang et al., A Multi-task learning method for weakly supervised sound event detection, in ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). (IEEE, Singapore, Singapore, 2022), pp.8802\u20138806","DOI":"10.1109\/ICASSP43922.2022.9746947"},{"key":"3379_CR26","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2025.3550979","author":"M Lou","year":"2025","unstructured":"M. Lou, S. Zhang, H.Y. Zhou et al., TransXNet: learning both global and local dynamics with a dual dynamic token mixer for visual recognition. IEEE Trans. Neural Netw. Learn Syst. (2025). https:\/\/doi.org\/10.1109\/TNNLS.2025.3550979","journal-title":"IEEE Trans. Neural Netw. Learn Syst."},{"key":"3379_CR27","unstructured":"W. Luo, Y. Li, R. Urtasun et al., Understanding the effective receptive field in deep convolutional neural networks. In: Proceedings of the 30th International Conference on Neural Information Processing Systems. Curran Associates Inc., Red Hook, NY, USA, NIPS\u201916, pp 4905\u20134913 (2016)"},{"issue":"6","key":"3379_CR28","doi-asserted-by":"publisher","first-page":"162","DOI":"10.3390\/app6060162","volume":"6","author":"A Mesaros","year":"2016","unstructured":"A. Mesaros, T. Heittola, T. Virtanen, Metrics for polyphonic sound event detection. Appl. Sci. -Basel 6(6), 162 (2016)","journal-title":"Appl. Sci. -Basel"},{"key":"3379_CR29","doi-asserted-by":"crossref","unstructured":"A. Mesaros, T. Heittola, T. Virtanen, TUT Database for Acoustic Scene Classification and Sound Event Detection, in 2016 24TH European Signal Processing Conference (EUSIPCO). (IEEE, New York, 2016), pp. 1128\u20131132","DOI":"10.1109\/EUSIPCO.2016.7760424"},{"key":"3379_CR30","unstructured":"D. Mu, Z. Zhang, H. Yue et al., SELD-Mamba: selective state-space model for sound event localization and detection with source distance estimation. arxiv:https:\/\/arxiv.org\/abs\/2408.05057 (2024)"},{"issue":"9","key":"3379_CR31","doi-asserted-by":"publisher","first-page":"1502","DOI":"10.3390\/math13091502","volume":"13","author":"A Mukhamadiyev","year":"2025","unstructured":"A. Mukhamadiyev, I. Khujayarov, D. Nabieva et al., An ensemble of convolutional neural networks for sound event detection. Mathematics 13(9), 1502 (2025)","journal-title":"Mathematics"},{"key":"3379_CR32","unstructured":"H. Nam, S.H. Kim, D. Min et al., Frequency & channel attention for computationally efficient sound event detection. arXiv:2306.11277 (2023)"},{"key":"3379_CR33","doi-asserted-by":"crossref","unstructured":"G. Parascandolo, H. Huttunen, T. Virtanen, Recurrent Neural Networks for Polyphonic Sound Event Detection in Real Life Recordings, in 2016 IEEE International Conference on Acoustics. (speech and signal processing proceedings. (IEEE, New York, 2016), pp.6440\u20136444","DOI":"10.1109\/ICASSP.2016.7472917"},{"issue":"24","key":"3379_CR34","doi-asserted-by":"publisher","first-page":"8375","DOI":"10.3390\/s21248375","volume":"21","author":"C Park","year":"2021","unstructured":"C. Park, D. Kim, H. Ko, Sound event detection by pseudo-labeling in weakly labeled dataset. Sensors 21(24), 8375 (2021)","journal-title":"Sensors"},{"key":"3379_CR35","doi-asserted-by":"crossref","unstructured":"J. Salamon, C. Jacoby, J.P. Bello, A Dataset and Taxonomy for Urban Sound Research. In: Proceedings of the 2014 ACM conference on multimedia (MM\u201914). Assoc Computing Machinery, New York, pp 1041\u20131044 (2014)","DOI":"10.1145\/2647868.2655045"},{"key":"3379_CR36","doi-asserted-by":"publisher","first-page":"123608","DOI":"10.1016\/j.eswa.2024.123608","volume":"249","author":"MY Shams","year":"2024","unstructured":"M.Y. Shams, T. Abd El-Hafeez, E. Hassan, Acoustic data detection in large-scale emergency vehicle sirens and road noise dataset. Expert Syst. Appl. 249, 123608 (2024)","journal-title":"Expert Syst. Appl."},{"key":"3379_CR37","doi-asserted-by":"publisher","first-page":"102471","DOI":"10.1016\/j.ecoinf.2024.102471","volume":"80","author":"B Swaminathan","year":"2024","unstructured":"B. Swaminathan, M. Jagadeesh, S. Vairavasundaram, Multi-label classification for acoustic bird species detection using transfer learning approach. Eco. Inform. 80, 102471 (2024)","journal-title":"Eco. Inform."},{"key":"3379_CR38","doi-asserted-by":"crossref","unstructured":"C. Szegedy, W. Liu, Y. Jia et al., Going Deeper with Convolutions. arXiv:1409.4842 (2014)","DOI":"10.1109\/CVPR.2015.7298594"},{"issue":"5","key":"3379_CR39","doi-asserted-by":"publisher","first-page":"90","DOI":"10.3390\/asi6050090","volume":"6","author":"W Villegas-Ch","year":"2023","unstructured":"W. Villegas-Ch, J. Govea, Application of deep learning in the early detection of emergency situations and security monitoring in public spaces. Appl. Syst. Innov. 6(5), 90 (2023)","journal-title":"Appl. Syst. Innov."},{"key":"3379_CR40","doi-asserted-by":"crossref","unstructured":"X. Wang, X. Zhang, Y. Zi et al., A Frame Loss of Multiple Instance Learning for Weakly Supervised Sound Event Detection, in ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). (IEEE, Singapore, Singapore, 2022), pp. 331\u2013335","DOI":"10.1109\/ICASSP43922.2022.9746435"},{"key":"3379_CR41","doi-asserted-by":"crossref","unstructured":"Y. Wang, J. Li, F. Metze, A comparison of five multiple instance learning pooling functions for sound event detection with weak labeling. arXiv:1810.09050 (2019)","DOI":"10.1109\/ICASSP.2019.8682847"},{"issue":"3","key":"3379_CR42","doi-asserted-by":"publisher","first-page":"569","DOI":"10.1109\/TMM.2019.2933330","volume":"22","author":"X Xia","year":"2020","unstructured":"X. Xia, R. Togneri, F. Sohel et al., Multi-task learning for acoustic event detection using event and frame position information. IEEE Trans. Multim. 22(3), 569\u2013578 (2020)","journal-title":"IEEE Trans. Multim."},{"key":"3379_CR43","doi-asserted-by":"crossref","unstructured":"Y. Xu, Q. Kong, W. Wang et al., Large-Scale Weakly Supervised Audio Classification Using Gated Convolutional Neural Network, in 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). (IEEE, New York, 2018), pp. 121\u2013125","DOI":"10.1109\/ICASSP.2018.8461975"},{"key":"3379_CR44","unstructured":"B. Yang, G. Bender, Q.V. Le et al., CondConv: conditionally parameterized convolutions for efficient inference. arXiv:1904.04971 (2020)"},{"key":"3379_CR45","doi-asserted-by":"publisher","first-page":"186","DOI":"10.1109\/LSP.2024.3509336","volume":"32","author":"T Yoshinaga","year":"2025","unstructured":"T. Yoshinaga, K. Tanaka, Y. Bando et al., Onset-and-offset-aware sound event detection via differentiable frame-to-event mapping. IEEE Signal Process. Lett. 32, 186\u2013190 (2025)","journal-title":"IEEE Signal Process. Lett."},{"key":"3379_CR46","doi-asserted-by":"crossref","unstructured":"X. Zhao, L. Zhang, Y. Pang et al. A single stream network for robust and real-time rgb-d salient object detection, in Computer Vision - ECCV 2020, vol. 12367, ed. by A. Vedaldi, H. Bischof, T. Brox et al. (Springer International Publishing, Cham, 2020), pp. 646\u2013662","DOI":"10.1007\/978-3-030-58542-6_39"},{"key":"3379_CR47","unstructured":"L. Zhu, B. Liao, Q. Zhang et al., Vision Mamba: efficient visual representation learning with bidirectional state space model. arXiv:2401.09417 (2024)"}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03379-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-025-03379-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03379-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T07:53:37Z","timestamp":1777103617000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-025-03379-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,14]]},"references-count":47,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["3379"],"URL":"https:\/\/doi.org\/10.1007\/s00034-025-03379-0","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"value":"0278-081X","type":"print"},{"value":"1531-5878","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,14]]},"assertion":[{"value":"14 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 September 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 September 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 October 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}