{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T21:19:47Z","timestamp":1776979187931,"version":"3.51.4"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T00:00:00Z","timestamp":1774915200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T00:00:00Z","timestamp":1774915200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s11760-026-05301-w","type":"journal-article","created":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T09:16:35Z","timestamp":1774948595000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Sound Event Detection with Manual Feature Fusion"],"prefix":"10.1007","volume":"20","author":[{"given":"Dongsheng","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenlong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,31]]},"reference":[{"issue":"5","key":"5301_CR1","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1109\/MSP.2021.3090678","volume":"38","author":"A Mesaros","year":"2021","unstructured":"Mesaros, A., Heittola, T., Virtanen, T., Plumbley, M.D.: Sound event detection: a tutorial. IEEE Signal Process. Mag. 38(5), 67\u201383 (2021)","journal-title":"IEEE Signal Process. Mag."},{"key":"5301_CR2","doi-asserted-by":"crossref","unstructured":"Wakayama, K., Saito, S.: Cnn-transformer with self-attention network for sound event detection. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 806\u2013810 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747762"},{"issue":"6","key":"5301_CR3","doi-asserted-by":"publisher","first-page":"1291","DOI":"10.1109\/TASLP.2017.2690575","volume":"25","author":"E Cak\u0131r","year":"2017","unstructured":"Cak\u0131r, E., Parascandolo, G., Heittola, T., Huttunen, H., Virtanen, T.: Convolutional recurrent neural networks for polyphonic sound event detection. IEEE Trans. Audio Speech Lang. Process. 25(6), 1291\u20131303 (2017)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"5301_CR4","doi-asserted-by":"crossref","unstructured":"Parascandolo, G., Huttunen, H., Virtanen, T.: Recurrent neural networks for polyphonic sound event detection in real life recordings. In: 2016 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6440\u20136444 (2016)","DOI":"10.1109\/ICASSP.2016.7472917"},{"key":"5301_CR5","unstructured":"Ye, Z., Wang, X., Liu, H., Qian, Y., Tao, R., Yan, L., Ouchi, K.: Sound event detection transformer: an event-based end-to-end model for sound event detection. arXiv preprint arXiv:2110.02011 (2021)"},{"key":"5301_CR6","unstructured":"Cakir, E.: Deep neural networks for sound event detection. PhD thesis, University of Tampere, Finland (2019)"},{"key":"5301_CR7","doi-asserted-by":"crossref","unstructured":"Shao, N., Li, X., Li, X.: Fine-tune the pretrained atst model for sound event detection. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 911\u2013915 (2024)","DOI":"10.1109\/ICASSP48485.2024.10446159"},{"key":"5301_CR8","doi-asserted-by":"publisher","first-page":"186","DOI":"10.1109\/LSP.2024.3509336","volume":"32","author":"T Yoshinaga","year":"2025","unstructured":"Yoshinaga, T., Tanaka, K., Bando, Y., Imoto, K., Morishima, S.: Onset-and-offset-aware sound event detection via differentiable frame-to-event mapping. IEEE Signal Process. Lett. 32, 186\u2013190 (2025)","journal-title":"IEEE Signal Process. Lett."},{"issue":"4","key":"5301_CR9","doi-asserted-by":"publisher","first-page":"910","DOI":"10.1109\/TAI.2022.3173582","volume":"4","author":"TK Chan","year":"2023","unstructured":"Chan, T.K., Chin, C.S.: Lightweight convolutional-iconformer for sound event detection. IEEE Trans. Artif. Intell. 4(4), 910\u2013921 (2023)","journal-title":"IEEE Trans. Artif. Intell."},{"issue":"1","key":"5301_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/1687-4722-2013-1","volume":"2013","author":"T Heittola","year":"2013","unstructured":"Heittola, T., Mesaros, A., Eronen, A., Virtanen, T.: Context-dependent sound event detection. EURASIP J. Audio Speech Music Process 2013(1), 1 (2013)","journal-title":"EURASIP J. Audio Speech Music Process"},{"key":"5301_CR11","doi-asserted-by":"publisher","first-page":"2895","DOI":"10.1109\/TASLP.2020.3029652","volume":"28","author":"Z Shuyang","year":"2020","unstructured":"Shuyang, Z., Heittola, T., Virtanen, T.: Active learning for sound event detection. IEEE Trans. Audio Speech Lang. Process. 28, 2895\u20132905 (2020)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"1","key":"5301_CR12","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1186\/s13636-015-0069-2","volume":"2015","author":"M Espi","year":"2015","unstructured":"Espi, M., Fujimoto, M., Kinoshita, K., Nakatani, T.: Exploiting spectro-temporal locality in deep learning based acoustic event detection. EURASIP J. Audio Speech Music Process 2015(1), 26 (2015)","journal-title":"EURASIP J. Audio Speech Music Process"},{"issue":"2","key":"5301_CR13","doi-asserted-by":"publisher","first-page":"206","DOI":"10.1109\/JSTSP.2019.2908700","volume":"13","author":"H Purwins","year":"2019","unstructured":"Purwins, H., Li, B., Virtanen, T., Schl\u00fcter, J., Chang, S.-Y., Sainath, T.: Deep learning for audio signal processing. IEEE J. Sel. Top. Signal Process 13(2), 206\u2013219 (2019)","journal-title":"IEEE J. Sel. Top. Signal Process"},{"issue":"6","key":"5301_CR14","doi-asserted-by":"publisher","first-page":"162","DOI":"10.3390\/app6060162","volume":"6","author":"A Mesaros","year":"2016","unstructured":"Mesaros, A., Heittola, T., Virtanen, T.: Metrics for polyphonic sound event detection. Appl. Sci. 6(6), 162 (2016)","journal-title":"Appl. Sci."},{"key":"5301_CR15","doi-asserted-by":"crossref","unstructured":"Ebbers, J., Haeb-Umbach, R., Serizel, R.: Threshold independent evaluation of sound event detection scores. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1021\u20131025 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747556"},{"key":"5301_CR16","doi-asserted-by":"crossref","unstructured":"Drossos, K., Mimilakis, S.I., Gharib, S., Li, Y., Virtanen, T.: Sound event detection with depthwise separable and dilated convolutions. In: 2020 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20137 (2020)","DOI":"10.1109\/IJCNN48605.2020.9207532"},{"key":"5301_CR17","doi-asserted-by":"crossref","unstructured":"Komatsu, T., Togami, M., Takahashi, T.: Sound event localization and detection using convolutional recurrent neural networks and gated linear units. In: 2020 28th European Signal Processing Conference (EUSIPCO), pp. 41\u201345 (2021)","DOI":"10.23919\/Eusipco47968.2020.9287372"},{"key":"5301_CR18","doi-asserted-by":"crossref","unstructured":"Nam, H., Kim, S.-H., Ko, B.-Y., Park, Y.-H.: Frequency dynamic convolution: Frequency-adaptive pattern recognition for sound event detection. In: Interspeech 2022, pp. 2763\u20132767 (2022)","DOI":"10.21437\/Interspeech.2022-10127"},{"key":"5301_CR19","doi-asserted-by":"crossref","unstructured":"Kim, S.-H., Nam, H., Park, Y.-H.: Temporal dynamic convolutional neural network for text-independent speaker verification and phonemic analysis. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6742\u20136746 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747421"},{"key":"5301_CR20","doi-asserted-by":"crossref","unstructured":"Xiao, S., Zhang, X., Zhang, P.: Multi-dimensional frequency dynamic convolution with confident mean teacher for sound event detection. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096306"},{"key":"5301_CR21","doi-asserted-by":"crossref","unstructured":"Yue, H., Zhang, Z., Mu, D., Dang, Y., Yin, J., Tang, J.: Full-frequency dynamic convolution: a physical frequency-dependent convolution for sound event detection. arXiv preprint arXiv:2401.04976 (2024)","DOI":"10.1007\/978-3-031-78498-9_18"},{"key":"5301_CR22","doi-asserted-by":"crossref","unstructured":"Cai, X., Chen, J., Liu, Z., Wu, M., Guo, H., Sun, X.: Teffdconv: an improved approach to enhance temporal localization in sound event detection. IEICE Trans. Inf. Syst. E108-D(10), 1250\u20131254 (2025)","DOI":"10.1587\/transinf.2024EDL8085"},{"key":"5301_CR23","unstructured":"Chen, S., Wu, Y., Wang, C., Liu, S., Tompkins, D., Chen, Z., Wei, F.: Beats: audio pre-training with acoustic tokenizers. arXiv preprint arXiv:2212.09058 (2022)"},{"key":"5301_CR24","doi-asserted-by":"crossref","unstructured":"Li, K., Song, Y., Dai, L.-R., McLoughlin, I., Fang, X., Liu, L.: Ast-sed: an effective sound event detection method based on audio spectrogram transformer. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096853"},{"key":"5301_CR25","doi-asserted-by":"crossref","unstructured":"LI, X., Li, X.: Atst: Audio representation learning with teacher-student transformer. In: Interspeech 2022, pp. 4172\u20134176 (2022)","DOI":"10.21437\/Interspeech.2022-10126"},{"issue":"6","key":"5301_CR26","doi-asserted-by":"publisher","first-page":"541","DOI":"10.3390\/e21060541","volume":"21","author":"A Delgado-Bonal","year":"2019","unstructured":"Delgado-Bonal, A., Marshak, A.: Approximate entropy and sample entropy: a comprehensive tutorial. Entropy 21(6), 541 (2019)","journal-title":"Entropy"},{"issue":"5","key":"5301_CR27","doi-asserted-by":"publisher","first-page":"610","DOI":"10.1109\/LSP.2016.2542881","volume":"23","author":"M Rostaghi","year":"2016","unstructured":"Rostaghi, M., Azami, H.: Dispersion entropy: a measure for time-series analysis. IEEE Signal Process. Lett. 23(5), 610\u2013614 (2016)","journal-title":"IEEE Signal Process. Lett."},{"key":"5301_CR28","unstructured":"Li, C., Zhou, A., Yao, A.: Omni-dimensional dynamic convolution. arXiv preprint arXiv:2209.07947 (2022)"},{"issue":"10","key":"5301_CR29","doi-asserted-by":"publisher","first-page":"1733","DOI":"10.1109\/TMM.2015.2428998","volume":"17","author":"D Stowell","year":"2015","unstructured":"Stowell, D., Giannoulis, D., Benetos, E., Lagrange, M., Plumbley, M.D.: Detection and classification of acoustic scenes and events. IEEE Trans. Multimedia 17(10), 1733\u20131746 (2015)","journal-title":"IEEE Trans. Multimedia"},{"key":"5301_CR30","doi-asserted-by":"crossref","unstructured":"Bugalho, M., Port lo, J., Trancoso, I., Pellegrini, T., Abad, A.: Detecting audio events for semantic video search. In: Interspeech 2009, pp. 1151\u20131154 (2009)","DOI":"10.21437\/Interspeech.2009-335"},{"issue":"9\u201310","key":"5301_CR31","doi-asserted-by":"publisher","first-page":"661","DOI":"10.1080\/08839514.2018.1430469","volume":"31","author":"E Babaee","year":"2017","unstructured":"Babaee, E., Anuar, N.B., Wahab, A.W.A., Shamshirband, S., Chronopoulos, A.T.: An overview of audio event detection methods from feature extraction to classification. Appl. Artif. Intell. 31(9\u201310), 661\u2013714 (2017)","journal-title":"Appl. Artif. Intell."},{"issue":"3","key":"5301_CR32","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3322240","volume":"52","author":"S Chandrakala","year":"2019","unstructured":"Chandrakala, S., Jayalakshmi, S.: Environmental audio scene and sound event recognition for autonomous surveillance: a survey and comparative studies. ACM Comput. Surv. 52(3), 1\u201334 (2019)","journal-title":"ACM Comput. Surv."},{"key":"5301_CR33","first-page":"1466","volume":"28","author":"L Lin","year":"2020","unstructured":"Lin, L., Wang, X., Liu, H., Qian, Y.: Specialized decision surface and disentangled feature for weakly-supervised polyphonic sound event detection. IEEE Trans. Audio Speech Lang. Process. 28, 1466\u20131478 (2020)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"5301_CR34","doi-asserted-by":"crossref","unstructured":"Bilen, \u00c7., Ferroni, G., Tuveri, F., Azcarreta, J., Krstulovi\u0107, S.: A framework for the robust evaluation of sound event detection. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 61\u201365 (2020)","DOI":"10.1109\/ICASSP40776.2020.9052995"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05301-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-026-05301-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-026-05301-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T20:32:27Z","timestamp":1776976347000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-026-05301-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,31]]},"references-count":34,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["5301"],"URL":"https:\/\/doi.org\/10.1007\/s11760-026-05301-w","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,31]]},"assertion":[{"value":"17 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 February 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests..","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"231"}}