{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T18:14:40Z","timestamp":1783361680524,"version":"3.54.6"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,6,13]],"date-time":"2025-06-13T00:00:00Z","timestamp":1749772800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,6,13]],"date-time":"2025-06-13T00:00:00Z","timestamp":1749772800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J AUDIO SPEECH MUSIC PROC."],"DOI":"10.1186\/s13636-025-00409-2","type":"journal-article","created":{"date-parts":[[2025,6,13]],"date-time":"2025-06-13T17:01:21Z","timestamp":1749834081000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["ResCapsnet: a capsule network with CRAM and BiGRU for sound event detection"],"prefix":"10.1186","volume":"2025","author":[{"given":"Bing","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenglong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuguo","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenwu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiduo","family":"Mei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,6,13]]},"reference":[{"key":"409_CR1","doi-asserted-by":"crossref","unstructured":"K. Imoto and S. Kyochi, Sound event detection utilizing graph laplacian reg- ularization with event co-occurrence, IEICEe Transactions on Information and Systems E103.D, 1971\u20131977 (2020).","DOI":"10.1587\/transinf.2019EDP7323"},{"key":"409_CR2","doi-asserted-by":"crossref","unstructured":"J.-L. Rouas, J. Louradour, and S. Ambellouis, Audio events detection in public transport vehicle, in IEEE Intelligent Transportation Systems Con- ference. IEEE 733\u2013738 (2006)","DOI":"10.1109\/ITSC.2006.1706829"},{"key":"409_CR3","doi-asserted-by":"crossref","unstructured":"A. Temko, E. Monte, and C. Nadeu, Comparison of sequence discriminant support vector machines for acoustic event classification, in IEEE Inter- national Conference on Acoustics Speech and Signal Processing Proceedings. IEEE, 5, pp. V\u2013V (2006)","DOI":"10.1109\/ICASSP.2006.1661377"},{"key":"409_CR4","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1016\/j.dsp.2022.103434","volume":"123","author":"J Meng","year":"2022","unstructured":"J. Meng, X. Wang, J. Wang, X. Teng, Y. Xu, A capsule network with pixel-based attention BGRU and for sound event detection. Digital Signal Processing 123, 103\u2013434 (2022)","journal-title":"Digital Signal Processing"},{"key":"409_CR5","doi-asserted-by":"crossref","unstructured":"A. Dang, T. H. Vu, and J.-C. Wang, A survey of deep learning for polyphonic sound event detection, in International Conference on Orange Technologies (ICOT). IEEE, 75\u201378 (2017)","DOI":"10.1109\/ICOT.2017.8336092"},{"key":"409_CR6","doi-asserted-by":"publisher","first-page":"1228","DOI":"10.1109\/JSTSP.2011.2146229","volume":"5","author":"N Degara","year":"2011","unstructured":"N. Degara, M.E. Davies, A. Pena, M.D. Plumbley, Onset event decod- ing exploiting the rhythmic structure of polyphonic music. IEEE Journal of Selected Topics in Signal Processing 5, 1228\u20131239 (2011)","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"409_CR7","doi-asserted-by":"publisher","first-page":"1144","DOI":"10.1109\/JSTSP.2011.2159700","volume":"5","author":"JJ Carabias-Orti","year":"2011","unstructured":"J.J. Carabias-Orti, T. Virtanen, P. Vera-Candeas, N. Ruiz-Reyes, F.J. Canadas-Quesada, Musical instrument sound multi-excitation model for non-negative spectrogram factorization. IEEE Journal of Selected Topics in Signal Processing 5, 1144\u20131158 (2011)","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"409_CR8","first-page":"1272","volume":"2010","author":"T Heittola","year":"2010","unstructured":"T. Heittola, A. Mesaros, A. Eronen, T. Virtanen, Audio context recogni- tion using audio event histograms, in, 18th European Signal Processing Conference. IEEE 2010, 1272\u20131276 (2010)","journal-title":"IEEE"},{"key":"409_CR9","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1109\/TNN.2002.806626","volume":"14","author":"G Guo","year":"2003","unstructured":"G. Guo, S.Z. Li, Content-based audio classification and retrieval by support vector machines. IEEE Transactions on Neural Networks 14, 209\u2013215 (2003)","journal-title":"IEEE Transactions on Neural Networks"},{"key":"409_CR10","doi-asserted-by":"crossref","unstructured":"Y. Liu, J. Tang, Y. Song, and L. Dai, A capsule based approach for poly- phonic sound event detection, in 2018 Asia-Pacific Signal and Informa- tion Processing Association Annual Summit and Conference (APSIPA ASC). IEEE, 1853\u20131857 (2018)","DOI":"10.23919\/APSIPA.2018.8659533"},{"key":"409_CR11","doi-asserted-by":"publisher","first-page":"540","DOI":"10.1109\/TASLP.2015.2389618","volume":"23","author":"I McLoughlin","year":"2015","unstructured":"I. McLoughlin, H. Zhang, Z. Xie, Y. Song, W. Xiao, Robust sound event classification using deep neural networks. IEEE\/ACM Transactions on Audio, Speech, and Language Processing 23, 540\u2013552 (2015)","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"409_CR12","doi-asserted-by":"crossref","unstructured":"M. Valenti, S. Squartini, A. Diment, G. Parascandolo, and T. Virtanen, A convolutional neural network approach for acoustic scene classification, in International Joint Conference on Neural Networks (IJCNN). IEEE, 1547\u20131554 (2017)","DOI":"10.1109\/IJCNN.2017.7966035"},{"key":"409_CR13","doi-asserted-by":"crossref","unstructured":"Y. Xu, Q. Kong, Q. Huang, W. Wang, and M. D. Plumbley, Attention and localization based on a deep convolutional recurrent model for weakly supervised audio tagging, in Interspeech 2017, 18th Annual Conference of the International Speech Communication Association, Stockholm, Sweden, August 20-24, 2017. ISCA, 3083\u20133087 (2017)","DOI":"10.21437\/Interspeech.2017-486"},{"key":"409_CR14","doi-asserted-by":"crossref","unstructured":"H. Zhang, I. McLoughlin, and Y. Song, Robust sound event recognition using convolutional neural networks, in IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 559\u2013563 (2015)","DOI":"10.1109\/ICASSP.2015.7178031"},{"key":"409_CR15","doi-asserted-by":"crossref","unstructured":"H. Phan, L. Hertel, M. Maass, and A. Mertins, \u201cRobust audio event recog- nition with 1-max pooling convolutional neural networks,\u201d in Proceedings of Interspeech 2016. San Francisco, USA: ISCA, 3653\u20133657 (2016)","DOI":"10.21437\/Interspeech.2016-123"},{"key":"409_CR16","unstructured":"E. Cak\u0131r, T. Heittola, and T. Virtanen, \u201cDomestic audio tagging with convo- lutional neural networks,\u201d in Detection and Classification of Acoustic Scenes and Events (DCASE) Workshop. Tampere University of Technology. Laboratory of Signal Processing, (2016)"},{"key":"409_CR17","doi-asserted-by":"crossref","unstructured":"H. Nam, S.-H. Kim, B.-Y. Ko, and Y.-H. Park, Frequency dynamic convo- lution: Frequency-adaptive pattern recognition for sound event detection, in Proceedings of Interspeech 2022. ISCA, 2763\u20132767 (2022)","DOI":"10.21437\/Interspeech.2022-10127"},{"key":"409_CR18","unstructured":"J. W. Kim, S. W. Son, Y. Song, H. K. Kim, I. H. Song, and J. E. Lim, Semi-supervsied learning-based sound event detection using freuqency dy- namic convolution with large kernel attention for dcase challenge 2023 task 4, arXiv e-prints, p. arXiv:2306.06461, (2023)"},{"key":"409_CR19","unstructured":"S. Xiao, J. Shen, A. Hu, X. Zhang, P. Zhang, Y. Yan, Sound event detection with weak prediction for dcase 2023 challenge task4a. Tech. Rep., DCASE2023 Challenge (2023)"},{"key":"409_CR20","unstructured":"S. Sabour, N. Frosst, and G. E. Hinton, Dynamic routing between capsules, in Asia-Pacific Signal and Information Processing Association Annual Sum- mit and Conference (APSIPA ASC). Red Hook, NY, USA: Curran Associates Inc.\u00a030, 1853\u20131857 (2018)"},{"key":"409_CR21","doi-asserted-by":"publisher","first-page":"310","DOI":"10.1109\/JSTSP.2019.2902305","volume":"13","author":"F Vesperini","year":"2019","unstructured":"F. Vesperini, L. Gabrielli, E. Principi, S. Squartini, Polyphonic sound event detection by using capsule neural networks. IEEE Journal of Selected Topics in Signal Processing 13, 310\u2013322 (2019)","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"409_CR22","doi-asserted-by":"crossref","unstructured":"T. Iqbal, Y. Xu, Q. Kong, and W. Wang, Capsule routing for sound event detection, in 2018 26th European Signal Processing Conference (EUSIPCO). IEEE, 2255\u20132259 (2018)","DOI":"10.23919\/EUSIPCO.2018.8553198"},{"key":"409_CR23","unstructured":"A. Mesaros, T. Heittola, A. Diment, B. Elizalde, A. Shah, E. Vincent, B. Raj, and T. Virtanen, \u201cDcase 2017 challenge setup: Tasks, datasets and baseline system,\u201d in Proceedings of the Detection and Classification of Acoustic Scenes and Events 2017 Workshop (DCASE2017). Western Finland, Finland: Tam- pere University of Technology. Laboratory of Signal Processing, 85\u201392 (2017)"},{"key":"409_CR24","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, and J. Sun, Deep residual learning for image recognition, in Proceedings of The IEEE Conference on Computer Vision and Pattern Recognition. IEEE, 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"409_CR25","doi-asserted-by":"crossref","unstructured":"A. Guzhov, F. Raue, J. Hees, and A. Dengel, \u201cEsresnet: Environmental sound classification based on visual domain models,\u201d in 2020 25th International Conference on Pattern Recognition (ICPR). IEEE, 4933\u20134940 (2020)","DOI":"10.1109\/ICPR48806.2021.9413035"},{"key":"409_CR26","doi-asserted-by":"publisher","first-page":"1251","DOI":"10.1109\/TASLP.2023.3256088","volume":"31","author":"Q Wang","year":"2023","unstructured":"Q. Wang, J. Du, H.-X. Wu, J. Pan, F. Ma, C.-H. Lee, A four-stage data augmentation approach to resnet-conformer based acoustic modeling for sound event localization and detection. IEEE\/ACM Transactions on Audio, Speech, and Language Processing 31, 1251\u20131264 (2023)","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"409_CR27","unstructured":"Z. C. Lipton, J. Berkowitz, and C. Elkan, A critical review of recurrent neural networks for sequence learning, arXiv e-prints, p. arXiv:1506.00019, (2015)"},{"key":"409_CR28","doi-asserted-by":"crossref","unstructured":"A. Graves, A.-r. Mohamed, and G. Hinton, Speech recognition with deep recurrent neural networks, in IEEE International Conference on Acoustics, Speech and Signal Processing. IEEE, 6645\u20136649 (2013)","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"409_CR29","doi-asserted-by":"crossref","unstructured":"G. Parascandolo, H. Huttunen, and T. Virtanen, Recurrent neural networks for polyphonic sound event detection in real life recordings, in IEEE Inter- national Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 6440\u20136444 (2016)","DOI":"10.1109\/ICASSP.2016.7472917"},{"key":"409_CR30","doi-asserted-by":"crossref","unstructured":"Y. Wang, L. Neves, and F. Metze, Audio-based multimedia event detection using deep recurrent neural networks, in IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE 2742\u2013 2746 (2016)","DOI":"10.1109\/ICASSP.2016.7472176"},{"key":"409_CR31","doi-asserted-by":"publisher","first-page":"2059","DOI":"10.1109\/TASLP.2017.2740002","volume":"25","author":"T Hayashi","year":"2017","unstructured":"T. Hayashi, S. Watanabe, T. Toda, T. Hori, J. Le Roux, K. Takeda, Duration-controlled lstm for polyphonic sound event detection. IEEE\/ACM Transactions on Audio, Speech, and Language Processing 25, 2059\u20132070 (2017)","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"409_CR32","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1109\/JSTSP.2018.2885636","volume":"13","author":"S Adavanne","year":"2019","unstructured":"S. Adavanne, A. Politis, J. Nikunen, T. Virtanen, Sound event localiza- tion and detection of overlapping sources using convolutional recurrent neural networks. IEEE Journal of Selected Topics in Signal Processing 13, 34\u201348 (2019)","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"409_CR33","doi-asserted-by":"crossref","unstructured":"Y. Cao, Q. Kong, T. Iqbal, F. An, W. Wang, and M. D. Plumbley, Poly- phonic sound event detection and localization using a two-stage strategy, arXiv e-prints, p. arXiv:1905.00268, (2019)","DOI":"10.33682\/4jhy-bj81"},{"key":"409_CR34","unstructured":"T. Virtanen, A. Mesaros, T. Heittola, A. Diment, E. Vincent, E. Benetos, and B. M. Elizalde, Proceedings of the Detection and Classification of Acoustic Scenes and Events 2017 Workshop (DCASE2017). Tampere University of Technology. Laboratory of Signal Processing (2017)"},{"key":"409_CR35","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"S. Hochreiter, J. Schmidhuber, Long short-term memory. Neural Com- putation 9, 1735\u20131780 (1997)","journal-title":"Neural Com- putation"},{"key":"409_CR36","unstructured":"J. Chung, C. Gulcehre, K. Cho, and Y. Bengio, Empirical evaluation of gated recurrent neural networks on sequence modeling, CoRR abs\/1412.3555, (2014)"},{"key":"409_CR37","unstructured":"R. Lu and Z. Duan, Bidirectional GRU for sound event detection, in Detec- tion and Classification of Acoustic Scenes and Events (DCASE) Workshop. Tampere University of Technology. Laboratory of Signal Processing (2017)"},{"key":"409_CR38","unstructured":"DCASE 2017 Task4, 2017. [Online]. Available: http:\/\/www.cs.tut.fi\/sgn\/arg\/dcase2017\/challenge\/task-large-scale-sound-event-detection. Accessed: 12 April 2024"},{"key":"409_CR39","doi-asserted-by":"crossref","unstructured":"N. Turpault, R. Serizel, A. Parag Shah, and J. Salamon, Sound event de- tection in domestic environments with weakly labeled data and soundscape synthesis, in Workshop on Detection and Classification of Acoustic Scenes and Events. New York City, United States: Inria (2019)","DOI":"10.33682\/006b-jx26"},{"key":"409_CR40","doi-asserted-by":"crossref","unstructured":"R. Serizel, N. Turpault, A. Shah, and J. Salamon, Sound event detection in synthetic domestic environments, in ICASSP 2020 - 2020 IEEE Inter- national Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE 86\u201390 (2020)","DOI":"10.1109\/ICASSP40776.2020.9054478"},{"key":"409_CR41","doi-asserted-by":"publisher","first-page":"1291","DOI":"10.1109\/TASLP.2017.2690575","volume":"25","author":"E Cak\u0131r","year":"2017","unstructured":"E. Cak\u0131r, G. Parascandolo, T. Heittola, H. Huttunen, T. Virtanen, Con- volutional recurrent neural networks for polyphonic sound event detection. IEEE\/ACM Transactions on Audio, Speech, and Language Processing 25, 1291\u20131303 (2017)","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"409_CR42","doi-asserted-by":"publisher","first-page":"162","DOI":"10.3390\/app6060162","volume":"6","author":"A Mesaros","year":"2016","unstructured":"A. Mesaros, T. Heittola, T. Virtanen, Metrics for polyphonic sound event detection. Applied Sciences 6, 162 (2016)","journal-title":"Applied Sciences"},{"key":"409_CR43","unstructured":"S. Ioffe and C. Szegedy. \u201cBatch normalization: Accelerating deep network training by reducing internal covariate shift,\u201d in Proceedings of the 32nd In- ternational Conference on Machine Learning. Lille, France: PMLR\u00a037, 448\u2013456 (2015)"},{"key":"409_CR44","unstructured":"G. E. Hinton, N. Srivastava, A. Krizhevsky, I. Sutskever, and R. R. Salakhut- dinov, Improving neural networks by preventing co-adaptation of feature detectors, arXiv e-prints, p. arXiv:1207.0580 (2012)"},{"key":"409_CR45","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"N. Srivastava, G. Hinton, A. Krizhevsky, I. Sutskever, R. Salakhutdinov, Dropout: A simple way to prevent neural networks from overfitting. The Journal of Machine Learning Research 15, 1929\u20131958 (2014)","journal-title":"The Journal of Machine Learning Research"},{"key":"409_CR46","unstructured":"D. P. Kingma and J. Ba, \u201cAdam: A method for stochastic optimization,\u201d (2017). [Online]. Available: https:\/\/arxiv.org\/abs\/1412.6980"},{"key":"409_CR47","doi-asserted-by":"crossref","unstructured":"Y. Xu, Q. Kong, W. Wang, and M. D. Plumbley, Large-scale weakly super- vised audio classification using gated convolutional neural network, in 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE 121\u2013125 (2018)","DOI":"10.1109\/ICASSP.2018.8461975"},{"key":"409_CR48","doi-asserted-by":"crossref","unstructured":"L. Xu, L. Wang, S. Bi, H. Liu, and J. Wang, Semi-supervised sound event de- tection with pre-trained model, in ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE 1\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10095687"},{"key":"409_CR49","doi-asserted-by":"publisher","first-page":"2673","DOI":"10.1109\/78.650093","volume":"45","author":"M Schuster","year":"1997","unstructured":"M. Schuster, K.K. Paliwal, Bidirectional recurrent neural networks. IEEE Transactions on Signal Processing 45, 2673\u20132681 (1997)","journal-title":"IEEE Transactions on Signal Processing"},{"key":"409_CR50","doi-asserted-by":"publisher","first-page":"104347","DOI":"10.1016\/j.dsp.2023.104347","volume":"146","author":"K Li","year":"2024","unstructured":"K. Li, S. Yang, L. Zhao, W. Wang, Weakly labeled sound event detection with a capsule-transformer model. Digit. Signal Process. 146, 104347 (2024)","journal-title":"Digit. Signal Process."}],"container-title":["EURASIP Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00409-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13636-025-00409-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00409-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,13]],"date-time":"2025-06-13T17:01:28Z","timestamp":1749834088000},"score":1,"resource":{"primary":{"URL":"https:\/\/asmp-eurasipjournals.springeropen.com\/articles\/10.1186\/s13636-025-00409-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,13]]},"references-count":50,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2025,12]]}},"alternative-id":["409"],"URL":"https:\/\/doi.org\/10.1186\/s13636-025-00409-2","relation":{},"ISSN":["1687-4722"],"issn-type":[{"value":"1687-4722","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,6,13]]},"assertion":[{"value":"29 January 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 June 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 June 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"22"}}