{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:16:00Z","timestamp":1784135760410,"version":"3.55.0"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2023,6,14]],"date-time":"2023-06-14T00:00:00Z","timestamp":1686700800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,6,14]],"date-time":"2023-06-14T00:00:00Z","timestamp":1686700800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s11042-023-15891-z","type":"journal-article","created":{"date-parts":[[2023,6,14]],"date-time":"2023-06-14T06:02:07Z","timestamp":1686722527000},"page":"8129-8143","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":24,"title":["A review of deep learning techniques in audio event recognition (AER) applications"],"prefix":"10.1007","volume":"83","author":[{"given":"Arjun","family":"Prashanth","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"S. L.","family":"Jayalakshmi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"R.","family":"Vedhapriyavadhana","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,6,14]]},"reference":[{"key":"15891_CR1","doi-asserted-by":"publisher","first-page":"38885","DOI":"10.1109\/ACCESS.2022.3166602","volume":"10","author":"A Abbasi","year":"2022","unstructured":"Abbasi A, Javed ARR, Yasin A, Jalil Z, Kryvinska N, Tariq U (2022) A large-scale benchmark dataset for anomaly detection and rare event classification for audio forensics. IEEE Access 10:38885\u201338894","journal-title":"IEEE Access"},{"key":"15891_CR2","doi-asserted-by":"publisher","first-page":"1100","DOI":"10.1109\/TASLP.2023.3244507","volume":"31","author":"Achyut Mani Tripathi and Om Jee Pandey","year":"2023","unstructured":"Achyut Mani Tripathi and Om Jee Pandey (2023) Divide and distill: new outlooks on knowledge distillation for environmental sound classification. IEEEACM Trans Audio, Speech, Language Process 31:1100\u20131113","journal-title":"IEEEACM Trans Audio, Speech, Language Process"},{"key":"15891_CR3","volume-title":"From natural to artificial intelligence, chapter 1","author":"SA Alim","year":"2018","unstructured":"Alim SA, Rashid NKA (2018) Some commonly used speech feature extraction algorithms. In: Lopez-Ruiz R (ed) From natural to artificial intelligence, chapter 1. IntechOpen, Rijeka"},{"key":"15891_CR4","doi-asserted-by":"crossref","unstructured":"Altalbe A (2021) Audio fingerprint analysis for speech processing using deep learning method. Int J Speech Technol:1\u20137","DOI":"10.1007\/s10772-022-09994-5"},{"key":"15891_CR5","doi-asserted-by":"crossref","unstructured":"Alzantot M, Wang Z, Srivastava MB (2019) Deep residual neural networks for audio spoofing detection. arXiv preprint arXiv:1907.00501","DOI":"10.21437\/Interspeech.2019-3174"},{"issue":"4","key":"15891_CR6","doi-asserted-by":"publisher","first-page":"2032","DOI":"10.3390\/s23042032","volume":"23","author":"M Bandara","year":"2023","unstructured":"Bandara M, Jayasundara R, Ariyarathne I, Meedeniya D, Perera C (2023) Forest sound classification dataset: Fsc22. Sensors 23(4):2032","journal-title":"Sensors"},{"issue":"1","key":"15891_CR7","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1166\/jmihi.2021.3313","volume":"11","author":"UA Bhatti","year":"2021","unstructured":"Bhatti UA, Yuan L, Zhaoyuan Y, Nawaz SA, Mehmood A, Bhatti MA, Nizamani MM, Xiao S et al (2021) Predictive data modeling using sp-knn for risk factor evaluation in urban demographical healthcare data. J Med Imaging Health Inform 11(1):7\u201314","journal-title":"J Med Imaging Health Inform"},{"key":"15891_CR8","doi-asserted-by":"crossref","unstructured":"Chandrakala S, Jayalakshmi SL (2019) Environmental audio scene and sound event recognition for autonomous surveillance: a survey and comparative studies. ACM Comput Surv (CSUR) 52(3):1\u201334","DOI":"10.1145\/3322240"},{"key":"15891_CR9","doi-asserted-by":"crossref","unstructured":"Colangelo F, Battisti F, Carli M, Neri A, Calabr\u00f3 F (2017) Enhancing audio surveillance with hierarchical recurrent neural networks. In 2017 14th IEEE international conference on advanced video and signal based surveillance (AVSS), pages 1\u20136. IEEE","DOI":"10.1109\/AVSS.2017.8078496"},{"key":"15891_CR10","doi-asserted-by":"crossref","unstructured":"Drossos K, Adavanne S, Virtanen T (2017) Automated audio captioning with recurrent neural networks. In IEEE workshop on applications of signal processing to audio and acoustics (WASPAA), new Paltz, New York, USA","DOI":"10.1109\/WASPAA.2017.8170058"},{"key":"15891_CR11","doi-asserted-by":"crossref","unstructured":"Fang Y, Liu D, Jiang Z, Wang H et al (2023) Monitoring of sleep breathing states based on audio sensor utilizing mel-scale features in home healthcare. J Healthcare Eng 2023","DOI":"10.1155\/2023\/6197564"},{"issue":"4","key":"15891_CR12","doi-asserted-by":"publisher","first-page":"5089","DOI":"10.1007\/s11042-021-11610-8","volume":"81","author":"L Gao","year":"2022","unstructured":"Gao L, Kele X, Wang H, Peng Y (2022) Multi-representation knowledge distillation for audio classification. Multimed Tools Appl 81(4):5089\u20135112","journal-title":"Multimed Tools Appl"},{"key":"15891_CR13","doi-asserted-by":"publisher","first-page":"3610","DOI":"10.1109\/TIFS.2020.2994740","volume":"15","author":"A Greco","year":"2020","unstructured":"Greco A, Petkov N, Saggese A, Vento M (2020) Aren: a deep learning approach for sound event recognition using a brain inspired representation. IEEE Trans Inform Forensics Sec 15:3610\u20133624","journal-title":"IEEE Trans Inform Forensics Sec"},{"key":"15891_CR14","doi-asserted-by":"crossref","unstructured":"Greco A, Saggese A, Vento M, Vigilante V (2019) Sorenet: a novel deep network for audio surveillance applications. In 2019 IEEE International Conference on Systems, Man and Cybernetics (SMC), pages 546\u2013551","DOI":"10.1109\/SMC.2019.8914435"},{"key":"15891_CR15","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. ProceedIEEE Conf Comput Vision Pattern Recogn:770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"15891_CR16","doi-asserted-by":"publisher","first-page":"109168","DOI":"10.1016\/j.apacoust.2022.109168","volume":"202","author":"O Inik","year":"2023","unstructured":"Inik O (2023) Cnn hyper-parameter optimization for environmental sound classification. Appl Acoust 202:109168","journal-title":"Appl Acoust"},{"key":"15891_CR17","unstructured":"Jiang Z, Soldati A, Schamberg I, Lameira AR, Moran S (2023) Automatic sound event detection and classification of great ape calls using neural networks. arXiv preprint arXiv:2301.02214"},{"key":"15891_CR18","doi-asserted-by":"crossref","unstructured":"K\u00fc\u00e7\u00fckbay SE, Kalkan S et al (2022) Hand-crafted versus learned representations for audio event detection. Multimed Tools Appl:1\u201320","DOI":"10.1007\/s11042-022-12873-5"},{"key":"15891_CR19","unstructured":"Lipton ZC, Berkowitz J, Elkan C (2015) A critical review of recurrent neural networks for sequence learning. arXiv preprint arXiv:1506.00019"},{"key":"15891_CR20","doi-asserted-by":"crossref","unstructured":"Mnasri Z, Rovetta S, Masulli F (2020) Audio surveillance of roads using deep learning and autoencoder-based sample weight initialization. In 2020 IEEE 20th Mediterranean Electrotechnical Conference ( MELECON), pages 99\u2013103","DOI":"10.1109\/MELECON48756.2020.9140594"},{"issue":"4","key":"15891_CR21","doi-asserted-by":"publisher","first-page":"5537","DOI":"10.1007\/s11042-021-11817-9","volume":"81","author":"Z Mnasri","year":"2022","unstructured":"Mnasri Z, Rovetta S, Masulli F (2022) Anomalous sound event detection: a survey of machine learning based methods and applications. Multimed Tools Appl 81(4):5537\u20135586","journal-title":"Multimed Tools Appl"},{"key":"15891_CR22","doi-asserted-by":"publisher","first-page":"109025","DOI":"10.1016\/j.patcog.2022.109025","volume":"133","author":"M Mohaimenuzzaman","year":"2023","unstructured":"Mohaimenuzzaman M, Bergmeir C, West I, Meyer B (2023) Environmental sound classification on the edge: a pipeline for deep acoustic networks on extremely resource constrained devices. Pattern Recogn 133:109025","journal-title":"Pattern Recogn"},{"key":"15891_CR23","doi-asserted-by":"publisher","first-page":"62719","DOI":"10.1109\/ACCESS.2021.3073786","volume":"9","author":"A Mustafa","year":"2021","unstructured":"Mustafa A, Qamhan, Altaheri H, Meftah AH, Muhammad G, Alotaibi YA (2021) Digital audio forensics. Microphone and environment classification using deep learning. IEEE Access 9:62719\u201362733","journal-title":"IEEE Access"},{"key":"15891_CR24","unstructured":"Poorjam AH (2018) Why we take only 12-13 mfcc coefficients in feature extraction?, 05"},{"issue":"2","key":"15891_CR25","doi-asserted-by":"publisher","first-page":"206","DOI":"10.1109\/JSTSP.2019.2908700","volume":"13","author":"H Purwins","year":"2019","unstructured":"Purwins H, Li B, Virtanen T, Schluter J, Chang S-Y, Sainath T (2019) Deep learning for audio signal processing. IEEE J Selected Topics Signal Process 13(2):206\u2013219","journal-title":"IEEE J Selected Topics Signal Process"},{"key":"15891_CR26","doi-asserted-by":"crossref","unstructured":"Ray R, Karthik S, Mathur V, Prashant Kumar G Maragatham ST, Shankarappa RT (2021) Feature genuinization based residual squeeze-and-excitation for audio anti-spoofing in sound ai. In 2021 12th international conference on computing communication and networking technologies (ICCCNT), pages 1\u20135. IEEE","DOI":"10.1109\/ICCCNT51525.2021.9580127"},{"key":"15891_CR27","doi-asserted-by":"crossref","unstructured":"Renaud J, Karam R, Salomon M, Couturier R (2023) Deep learning and gradient boosting for urban environmental noise monitoring in smart cities. Expert Syst Appl:119568","DOI":"10.1016\/j.eswa.2023.119568"},{"key":"15891_CR28","unstructured":"Revay S, Teschke M (2019) Multiclass language identification using deep learning on spectral images of audio signals. CoRR, abs\/1905.04348"},{"key":"15891_CR29","doi-asserted-by":"crossref","unstructured":"Shaer I, Shami A , (2022) Sound event classification in an industrial environment: Pipe leakage detection use case. arXiv preprint arXiv:2205.02706","DOI":"10.1109\/IWCMC55113.2022.9824540"},{"key":"15891_CR30","doi-asserted-by":"crossref","unstructured":"Shim H-J, Jung J-W, Heo H-S, Yoon S-H, Ha-Jin Y (2018) Replay spoofing detection system for automatic speaker verification using multi-task learning of noise classes. In 2018 Conference on Technologies and Applications of Artificial Intelligence (TAAI), pages 172\u2013176","DOI":"10.1109\/TAAI.2018.00046"},{"key":"15891_CR31","doi-asserted-by":"publisher","first-page":"108638","DOI":"10.1016\/j.apacoust.2022.108638","volume":"190","author":"Q Shi","year":"2022","unstructured":"Shi Q, Deng S, Han J (2022) Common subspace learning based semantic feature extraction method for acoustic event recognition. Appl Acoust 190:108638","journal-title":"Appl Acoust"},{"issue":"10","key":"15891_CR32","doi-asserted-by":"publisher","first-page":"1733","DOI":"10.1109\/TMM.2015.2428998","volume":"17","author":"D Stowell","year":"2015","unstructured":"Stowell D, Giannoulis D, Benetos E, Lagrange M, Plumbley MD (2015) Detection and classification of acoustic scenes and events. IEEE Trans Multimedia 17(10):1733\u20131746","journal-title":"IEEE Trans Multimedia"},{"key":"15891_CR33","doi-asserted-by":"publisher","first-page":"e488","DOI":"10.7717\/peerj.488","volume":"2","author":"D Stowell","year":"2014","unstructured":"Stowell D, Plumbley MD (2014) Automatic large-scale classification of bird sounds is strongly improved by unsupervised feature learning. PeerJ 2:e488","journal-title":"PeerJ"},{"issue":"3","key":"15891_CR34","doi-asserted-by":"publisher","first-page":"368","DOI":"10.1111\/2041-210X.13103","volume":"10","author":"D Stowell","year":"2019","unstructured":"Stowell D, Wood MD, Pamu\u0142a H, Stylianou Y, Glotin H (2019) Automatic acoustic detection of birds through deep learning: the first bird audio detection challenge. Methods Ecol Evol 10(3):368\u2013380","journal-title":"Methods Ecol Evol"},{"key":"15891_CR35","unstructured":"Su C, Huang H-Y, Shi S, Guo Y, Wu H (2017) A parallel recurrent neural network for language modeling with pos tags. In Proceedings of the 31st Pacific Asia Conference on Language, Information and Computation, pages 140\u2013147"},{"key":"15891_CR36","doi-asserted-by":"publisher","first-page":"516","DOI":"10.1016\/j.csl.2017.01.001","volume":"45","author":"M Todisco","year":"2017","unstructured":"Todisco M, Delgado H, Evans N (2017) Constant q cepstral coefficients: a spoofing countermeasure for automatic speaker verification. Comput Speech Lang 45:516\u2013535","journal-title":"Comput Speech Lang"},{"key":"15891_CR37","doi-asserted-by":"crossref","unstructured":"Turab M, Kumar T, Bendechache M, Saber T (2022) Investigating multi-feature selection and ensembling for audio classification. arXiv preprint arXiv:2206.07511","DOI":"10.5121\/ijaia.2022.13306"},{"issue":"7","key":"15891_CR38","doi-asserted-by":"publisher","first-page":"3293","DOI":"10.3390\/app12073293","volume":"12","author":"S Venkatesh","year":"2022","unstructured":"Venkatesh S, Moffat D, Miranda ER (2022) You only hear once: a yolo-like algorithm for audio segmentation and sound event detection. Appl Sci 12(7):3293","journal-title":"Appl Sci"},{"key":"15891_CR39","doi-asserted-by":"crossref","unstructured":"Xu Y, Kong Q, Huang Q, Wang W, Plumbley MarkD (2017) Convolutional gated recurrent neural network incorporating spatial features for audio tagging. In 2017 international joint conference on neural networks (IJCNN), pages 3461\u20133466. IEEE","DOI":"10.1109\/IJCNN.2017.7966291"},{"issue":"4","key":"15891_CR40","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1109\/MCAS.2019.2945210","volume":"19","author":"Y Zhao","year":"2019","unstructured":"Zhao Y, Xia X, Togneri R (2019) Applications of deep learning to audio generation. IEEE Circ Syst Magaz 19(4):19\u201338","journal-title":"IEEE Circ Syst Magaz"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15891-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-15891-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15891-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,8]],"date-time":"2024-01-08T07:28:34Z","timestamp":1704698914000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-15891-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,14]]},"references-count":40,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["15891"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-15891-z","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,6,14]]},"assertion":[{"value":"15 January 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 May 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 May 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 June 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflicts of interest to declare that are relevant to the content of this article","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}