{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T16:34:47Z","timestamp":1781109287027,"version":"3.54.1"},"reference-count":87,"publisher":"Springer Science and Business Media LLC","issue":"23","license":[{"start":{"date-parts":[[2023,3,16]],"date-time":"2023-03-16T00:00:00Z","timestamp":1678924800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,3,16]],"date-time":"2023-03-16T00:00:00Z","timestamp":1678924800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2023,9]]},"DOI":"10.1007\/s11042-023-14734-1","type":"journal-article","created":{"date-parts":[[2023,3,26]],"date-time":"2023-03-26T23:46:40Z","timestamp":1679874400000},"page":"36143-36177","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["Time-frequency visual representation and texture features for audio applications: a comprehensive review, recent trends, and challenges"],"prefix":"10.1007","volume":"82","author":[{"given":"Yogita D.","family":"Mistry","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3531-3958","authenticated-orcid":false,"given":"Gajanan K.","family":"Birajdar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Archana M.","family":"Khodke","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,3,16]]},"reference":[{"key":"14734_CR1","doi-asserted-by":"publisher","unstructured":"Abidin S, Togneri R, Sohel F (2017) Enhanced lbp texture features from time frequency representations for acoustic scene classification. In: 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 626\u2013630. https:\/\/doi.org\/10.1109\/ICASSP.2017.7952231","DOI":"10.1109\/ICASSP.2017.7952231"},{"key":"14734_CR2","doi-asserted-by":"publisher","unstructured":"Abidin S, Togneri R, Sohel F (2018) Acoustic scene classification using joint time-frequency image-based feature representations. In: 2018 15th IEEE international conference on advanced video and signal based surveillance (AVSS), pp 1\u20136. https:\/\/doi.org\/10.1109\/AVSS.2018.8639164","DOI":"10.1109\/AVSS.2018.8639164"},{"issue":"11","key":"14734_CR3","doi-asserted-by":"publisher","first-page":"2112","DOI":"10.1109\/TASLP.2018.2854861","volume":"26","author":"S Abidin","year":"2018","unstructured":"Abidin S, Togneri R, Sohel F (2018) Spectrotemporal analysis using local binary pattern variants for acoustic scene classification. IEEE\/ACM Trans Audio Speech Lang Process 26(11):2112\u20132121. https:\/\/doi.org\/10.1109\/TASLP.2018.2854861","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"14734_CR4","doi-asserted-by":"publisher","unstructured":"Abidin S, Xia X, Togneri R, Sohel F (2018) Local binary pattern with random forest for acoustic scene classification. In: 2018 IEEE international conference on multimedia and expo, ICME 2018. IEEE, institute of electrical and electronics engineers, United States, vol 2018-July. https:\/\/doi.org\/10.1109\/ICME.2018.8486578","DOI":"10.1109\/ICME.2018.8486578"},{"key":"14734_CR5","doi-asserted-by":"publisher","unstructured":"Agera N, Chapaneri S, Jayaswal D (2015) Exploring textural features for automatic music genre classification. In: 2015 International conference on computing communication control and automation, pp 822\u2013826. https:\/\/doi.org\/10.1109\/ICCUBEA.2015.164","DOI":"10.1109\/ICCUBEA.2015.164"},{"key":"14734_CR6","doi-asserted-by":"publisher","unstructured":"Ahmed F, Paul PP, Gavrilova M (2016) Music genre classification using a gradient-based local texture descriptor. In: Czarnowski I, Caballero AM, Howlett RJ, Jain LC (eds) Intelligent decision technologies 2016. Springer international publishing, Cham, pp 455\u2013464. https:\/\/doi.org\/10.1007\/978-3-319-39627-9-40","DOI":"10.1007\/978-3-319-39627-9-40"},{"issue":"3","key":"14734_CR7","doi-asserted-by":"publisher","first-page":"260","DOI":"10.1049\/iet-spr.2017.0170","volume":"12","author":"MS Alam","year":"2018","unstructured":"Alam MS, Jassim WA, Zilany MSA (2018) Radon transform of auditory neurograms: a robust feature set for phoneme classification. IET Sig Process 12(3):260\u2013268. https:\/\/doi.org\/10.1049\/iet-spr.2017.0170","journal-title":"IET Sig Process"},{"key":"14734_CR8","doi-asserted-by":"publisher","unstructured":"Ashfaque Mostafa T, Soltaninejad S, McIsaac TL, Cheng I (2021) A comparative study of time frequency representation techniques for freeze of gait detection and prediction. Sensors, vol 21(19). https:\/\/doi.org\/10.3390\/s21196446","DOI":"10.3390\/s21196446"},{"key":"14734_CR9","doi-asserted-by":"publisher","unstructured":"Battaglino D, Lepauloux L, Pilati L, Evans N (2015) Acoustic context recognition using local binary pattern codebooks. In: 2015 IEEE workshop on applications of signal processing to audio and acoustics (WASPAA), pp 1\u20135. https:\/\/doi.org\/10.1109\/WASPAA.2015.7336886","DOI":"10.1109\/WASPAA.2015.7336886"},{"key":"14734_CR10","unstructured":"Bhattacharjee M, Prasanna SRM, Guha P (2018) Time-frequency audio features for speech-music classification"},{"issue":"3","key":"14734_CR11","doi-asserted-by":"publisher","first-page":"329","DOI":"10.1080\/17517575.2018.1557256","volume":"13","author":"UA Bhatti","year":"2019","unstructured":"Bhatti UA, Huang M, Wu D, Zhang Y, Mehmood A, Han H (2019) Recommendation system using feature extraction and pattern recognition in clinical care systems. Enterpr Inf Syst 13(3):329\u2013351. https:\/\/doi.org\/10.1080\/17517575.2018.1557256","journal-title":"Enterpr Inf Syst"},{"issue":"2","key":"14734_CR12","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/JPHOT.2021.3059703","volume":"13","author":"UA Bhatti","year":"2021","unstructured":"Bhatti UA, Ming-Quan Z, Qing-Song H, Ali S, Hussain A, Yuhuan Y, Yu Z, Yuan L, Nawaz SA (2021) Advanced color edge detection using clifford algebra in satellite images. IEEE Photon J 13(2):1\u201320. https:\/\/doi.org\/10.1109\/JPHOT.2021.3059703","journal-title":"IEEE Photon J"},{"key":"14734_CR13","doi-asserted-by":"publisher","first-page":"155783","DOI":"10.1109\/ACCESS.2020.3018544","volume":"8","author":"UA Bhatti","year":"2020","unstructured":"Bhatti UA, Zhaoyuan Y, Linwang Y, Zeeshan Z, Ali NS, Mughair B, Anum M, Ul AQ, Luo W (2020) Geometric algebra applications in geospatial artificial intelligence and remote sensing image processing. IEEE Access 8:155783\u2013155796. https:\/\/doi.org\/10.1109\/ACCESS.2020.3018544","journal-title":"IEEE Access"},{"issue":"11","key":"14734_CR14","doi-asserted-by":"publisher","first-page":"15141","DOI":"10.1007\/s11042-018-6899-z","volume":"78","author":"GK Birajdar","year":"2019","unstructured":"Birajdar GK, Patil MD (2019) Speech and music classification using spectrogram based statistical descriptors and extreme learning machine. Multimed Tools Appl 78(11):15141\u201315168. https:\/\/doi.org\/10.1007\/s11042-018-6899-z","journal-title":"Multimed Tools Appl"},{"key":"14734_CR15","doi-asserted-by":"publisher","first-page":"329","DOI":"10.1007\/s12652-019-01303-4","volume":"11","author":"GK Birajdar","year":"2020","unstructured":"Birajdar GK, Patil MD (2020) Speech\/music classification using visual and spectral chromagram features. J Ambient Intell Humanized Comput 11:329\u2013347. https:\/\/doi.org\/10.1007\/s12652-019-01303-4","journal-title":"J Ambient Intell Humanized Comput"},{"key":"14734_CR16","doi-asserted-by":"publisher","unstructured":"Birajdar GK, Raveendran S (2022) Indian language identification using time-frequency texture features and kernel ELM. J Ambient Intell Humanized Comput:1\u201312. https:\/\/doi.org\/10.1007\/s12652-022-03781-5","DOI":"10.1007\/s12652-022-03781-5"},{"key":"14734_CR17","doi-asserted-by":"publisher","unstructured":"Bisot V, Essid S, Richard G (2015) HOG and subband power distribution image features for acoustic scene classification. In: 2015 23rd European signal processing conference (EUSIPCO), pp 719\u2013723. https:\/\/doi.org\/10.1109\/EUSIPCO.2015.7362477","DOI":"10.1109\/EUSIPCO.2015.7362477"},{"key":"14734_CR18","doi-asserted-by":"publisher","unstructured":"Breve B, Cirillo S, Cuofano M, Desiato D (2020) Perceiving space through sound: mapping human movements into MIDI. In: 26th International conference on distributed multimedia systems, virtual conference center, USA, pp 49\u201356. https:\/\/doi.org\/10.18293\/DMSVIVA20-011","DOI":"10.18293\/DMSVIVA20-011"},{"issue":"1","key":"14734_CR19","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1007\/s11042-021-11077-7","volume":"81","author":"B Breve","year":"2022","unstructured":"Breve B, Cirillo S, Cuofano M, Desiato D (2022) Enhancing spatial perception through sound: mapping human movements into MIDI. Multimed Tools Appl 81(1):73\u201394. https:\/\/doi.org\/10.1007\/s11042-021-11077-7","journal-title":"Multimed Tools Appl"},{"key":"14734_CR20","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1016\/j.precisioneng.2018.12.004","volume":"56","author":"Y Chen","year":"2019","unstructured":"Chen Y, Li H, Hou L, Bu X (2019) Feature extraction using dominant frequency bands and time-frequency image analysis for chatter detection in milling. Precis Eng 56:235\u2013245. https:\/\/doi.org\/10.1016\/j.precisioneng.2018.12.004","journal-title":"Precis Eng"},{"issue":"1","key":"14734_CR21","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1080\/0952813X.2019.1631392","volume":"32","author":"AA Chowdhury","year":"2020","unstructured":"Chowdhury AA, Borkar VS, Birajdar GK (2020) Indian language identification using time-frequency image textural descriptors and gwo-based feature selection. J Exp Theor Artif Intell 32(1):111\u2013132. https:\/\/doi.org\/10.1080\/0952813X.2019.1631392","journal-title":"J Exp Theor Artif Intell"},{"issue":"6","key":"14734_CR22","doi-asserted-by":"publisher","first-page":"611","DOI":"10.1016\/S0020-7373(86)80012-8","volume":"24","author":"J Connolly","year":"1986","unstructured":"Connolly J, Edmonds E, Guzy J, Johnson S, Woodcock A (1986) Automatic speech recognition based on spectrogram reading. Int J Man-Mach Stud 24(6):611\u2013621. https:\/\/doi.org\/10.1016\/S0020-7373(86)80012-8 . http:\/\/www.sciencedirect.com\/science\/article\/pii\/S0020737386800128","journal-title":"Int J Man-Mach Stud"},{"key":"14734_CR23","doi-asserted-by":"crossref","unstructured":"Costa Y, Oliveira L, Koerich A, Gouyon F (2013) Music genre recognition based on visual features with dynamic ensemble of classifiers selection. In: 2013 20th International conference on systems, signals and image processing (IWSSIP), pp 55\u201358","DOI":"10.1109\/IWSSIP.2013.6623448"},{"key":"14734_CR24","doi-asserted-by":"publisher","unstructured":"Costa Y, Oliveira L, Koerich A, Gouyon F (2013) Music genre recognition using gabor filters and LPQ texture descriptors. In: Ruiz-Shulcloper J, Sanniti di Baja G (eds) Progress in pattern recognition, image analysis, computer vision, and applications. Springer Berlin Heidelberg, Berlin, Heidelberg, pp 67\u201374. https:\/\/doi.org\/10.1007\/978-3-642-41827-3-9","DOI":"10.1007\/978-3-642-41827-3-9"},{"issue":"11","key":"14734_CR25","doi-asserted-by":"publisher","first-page":"2723","DOI":"10.1016\/j.sigpro.2012.04.023","volume":"92","author":"Y Costa","year":"2012","unstructured":"Costa Y, Oliveira L, Koerich A, Gouyon F, Martins J (2012) Music genre classification using LBP textural features. Sig Process 92(11):2723\u20132737. https:\/\/doi.org\/10.1016\/j.sigpro.2012.04.023","journal-title":"Sig Process"},{"key":"14734_CR26","unstructured":"Costa YMG, Oliveira LS, Koericb AL, Gouyon F (2011) Music genre recognition using spectrograms. In: 2011 18th International conference on systems, signals and image processing, pp 1\u20134"},{"key":"14734_CR27","doi-asserted-by":"publisher","unstructured":"Costa YMG, Oliveira LS, Koerich AL, Gouyon F (2012) Comparing textural features for music genre classification. In: The 2012 international joint conference on neural networks (IJCNN), pp 1\u20136. https:\/\/doi.org\/10.1109\/IJCNN.2012.6252626","DOI":"10.1109\/IJCNN.2012.6252626"},{"key":"14734_CR28","doi-asserted-by":"publisher","unstructured":"Demir F, Seng\u00fcr A, Cummins N, Amiriparian S, Schuller BW (2018) Low level texture features for snore sound discrimination. In: 2018 40th Annual international conference of the IEEE engineering in medicine and biology society (EMBC), pp 413\u2013416. https:\/\/doi.org\/10.1109\/EMBC.2018.8512459","DOI":"10.1109\/EMBC.2018.8512459"},{"issue":"2","key":"14734_CR29","doi-asserted-by":"publisher","first-page":"367","DOI":"10.1109\/TASL.2012.2226160","volume":"21","author":"J Dennis","year":"2013","unstructured":"Dennis J, Tran HD, Chng ES (2013) Image feature representation of the subband power distribution for robust sound event classification. IEEE Trans Audio Speech Lang Process 21(2):367\u2013377. https:\/\/doi.org\/10.1109\/TASL.2012.2226160","journal-title":"IEEE Trans Audio Speech Lang Process"},{"issue":"2","key":"14734_CR30","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1109\/LSP.2010.2100380","volume":"18","author":"J Dennis","year":"2011","unstructured":"Dennis J, Tran HD, Li H (2011) Spectrogram image feature for sound event classification in mismatched conditions. IEEE Sig Process Lett 18 (2):130\u2013133. https:\/\/doi.org\/10.1109\/LSP.2010.2100380","journal-title":"IEEE Sig Process Lett"},{"issue":"1","key":"14734_CR31","doi-asserted-by":"publisher","first-page":"e191","DOI":"10.1002\/itl2.191","volume":"5","author":"A Dutta","year":"2022","unstructured":"Dutta A, Sil D, Chandra A, Palit S (2022) Cnn based musical instrument identification using time-frequency localized features. Int Technol Lett 5 (1):e191. https:\/\/doi.org\/10.1002\/itl2.191","journal-title":"Int Technol Lett"},{"key":"14734_CR32","doi-asserted-by":"publisher","unstructured":"Felipe GZ, Aguiar RL, Costa YMG, Silla C, Brahnam S, Nanni L, McMurtrey S (2019) Identification of infants\u2019 cry motivation using spectrograms. In: 2019 International conference on systems, signals and image processing (IWSSIP), pp 181\u2013186. https:\/\/doi.org\/10.1109\/IWSSIP.2019.8787318","DOI":"10.1109\/IWSSIP.2019.8787318"},{"key":"14734_CR33","doi-asserted-by":"publisher","unstructured":"Felipe GZ, Maldonado Y, Costa DG, Helal LG (2017) Acoustic scene classification using spectrograms. In: 2017 36th International conference of the chilean computer science society (SCCC), pp 1\u20137. https:\/\/doi.org\/10.1109\/SCCC.2017.8405119","DOI":"10.1109\/SCCC.2017.8405119"},{"key":"14734_CR34","doi-asserted-by":"publisher","unstructured":"Ghosal A, Chakraborty R, Dhara BC, Saha SK (2012) Song\/instrumental classification using spectrogram based contextual features. In: Proceedings of the CUBE international information technology conference, CUBE \u201912. Association for computing machinery, New York, NY, USA, pp 21\u201325. https:\/\/doi.org\/10.1145\/2381716.2381722","DOI":"10.1145\/2381716.2381722"},{"key":"14734_CR35","doi-asserted-by":"publisher","first-page":"01010","DOI":"10.1051\/itmconf\/20203201010","volume":"32","author":"S Godbole","year":"2020","unstructured":"Godbole S, Jadhav V, Birajdar G (2020) Indian language identification using deep learning. ITM Web Conf 32:01010. https:\/\/doi.org\/10.1051\/itmconf\/20203201010","journal-title":"ITM Web Conf"},{"key":"14734_CR36","doi-asserted-by":"publisher","unstructured":"Jassim WA, Harte N (2018) Voice activity detection using neurograms. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 5524\u20135528. https:\/\/doi.org\/10.1109\/ICASSP.2018.8461952","DOI":"10.1109\/ICASSP.2018.8461952"},{"key":"14734_CR37","doi-asserted-by":"publisher","unstructured":"Jog AH, Jugade OA, Kadegaonkar AS, Birajdar GK (2018) Indian language identification using cochleagram based texture descriptors and ann classifier. In: 2018 15th IEEE India council international conference (INDICON), pp 1\u20136. https:\/\/doi.org\/10.1109\/INDICON45594.2018.8987167","DOI":"10.1109\/INDICON45594.2018.8987167"},{"issue":"3","key":"14734_CR38","doi-asserted-by":"publisher","first-page":"210","DOI":"10.1109\/TAU.1973.1162453","volume":"21","author":"D Klatt","year":"1973","unstructured":"Klatt D, Stevens K (1973) On the automatic recognition of continuous speech:implications from a spectrogram-reading experiment. IEEE Trans Audio Electroacoustics 21(3):210\u2013217. https:\/\/doi.org\/10.1109\/TAU.1973.1162453","journal-title":"IEEE Trans Audio Electroacoustics"},{"key":"14734_CR39","doi-asserted-by":"publisher","unstructured":"Kobayashi T, Ye J (2014) Acoustic feature extraction by statistics based local binary pattern for environmental sound classification. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 3052\u20133056. https:\/\/doi.org\/10.1109\/ICASSP.2014.6854161","DOI":"10.1109\/ICASSP.2014.6854161"},{"key":"14734_CR40","doi-asserted-by":"publisher","first-page":"2204","DOI":"10.1016\/j.procs.2017.08.115","volume":"112","author":"EB Lacerda","year":"2017","unstructured":"Lacerda EB, Mello CA (2017) Automatic classification of laryngeal mechanisms in singing based on the audio signal. Procedia Comput Sci 112:2204\u20132212. https:\/\/doi.org\/10.1109\/ICASSP.2014.6854161","journal-title":"Procedia Comput Sci"},{"issue":"4","key":"14734_CR41","doi-asserted-by":"publisher","first-page":"667","DOI":"10.1049\/cje.2019.04.005","volume":"28","author":"Y Li","year":"2019","unstructured":"Li Y, Huang H, Wu Z (2019) Animal sound recognition based on double feature of spectrogram. Chinese J Electron 28(4):667\u2013673. https:\/\/doi.org\/10.1049\/cje.2019.04.005","journal-title":"Chinese J Electron"},{"key":"14734_CR42","doi-asserted-by":"crossref","unstructured":"Lim H, Kim MJ, Kim H (2015) Robust sound event classification using LBP-HOG based bag-of-audio-words feature representation. In: INTERSPEECH, pp 3325\u20133329","DOI":"10.21437\/Interspeech.2015-670"},{"key":"14734_CR43","unstructured":"Matsui T, Goto M, Vert J, Uchiyama Y (2011) Gradient-based musical feature extraction based on scale-invariant feature transform. In: 2011 19th European signal processing conference, pp 724\u2013728"},{"key":"14734_CR44","doi-asserted-by":"publisher","first-page":"1672","DOI":"10.1007\/s00034-019-01203-0","volume":"39","author":"IV McLoughlin","year":"2020","unstructured":"McLoughlin IV, Xie Z, Song Y, Phan H, Palaniappan R (2020) Time-frequency feature fusion for noise-robust audio event classification. Circ Syst Sig Process 39:1672\u20131687. https:\/\/doi.org\/10.1007\/s00034-019-01203-0","journal-title":"Circ Syst Sig Process"},{"key":"14734_CR45","doi-asserted-by":"publisher","unstructured":"Montalvo A, Costa YMG, Calvo JR (2015) Language identification using spectrogram texture. In: Pardo A, Kittler J (eds) Progress in pattern recognition, image analysis, computer vision, and applications. Springer international publishing, Cham, pp 543\u2013550. https:\/\/doi.org\/10.1007\/978-3-319-25751-8-65","DOI":"10.1007\/978-3-319-25751-8-65"},{"key":"14734_CR46","doi-asserted-by":"publisher","first-page":"130","DOI":"10.1016\/j.apacoust.2019.05.020","volume":"155","author":"M Mulimani","year":"2019","unstructured":"Mulimani M, Koolagudi SG (2019) Robust acoustic event classification using fusion fisher vector features. Appl Acoust 155:130\u2013138. https:\/\/doi.org\/10.1016\/j.apacoust.2019.05.020","journal-title":"Appl Acoust"},{"issue":"2","key":"14734_CR47","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1049\/iet-cvi.2017.0075","volume":"12","author":"L Nanni","year":"2018","unstructured":"Nanni L, Aguiar RL, Costa YMG, Brahnam S, Silla CN, Brattin RL, Zhao Z (2018) Bird and whale species identification using sound images. IET Comput Vis 12(2):178\u2013184. https:\/\/doi.org\/10.1049\/iet-cvi.2017.0075","journal-title":"IET Comput Vis"},{"key":"14734_CR48","unstructured":"Nanni L, Costa Y, Brahnam S (2014) Set of texture descriptors for music genre classification"},{"key":"14734_CR49","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1016\/j.patrec.2017.01.013","volume":"88","author":"L Nanni","year":"2017","unstructured":"Nanni L, Costa Y, Lucio D, Silla C, Brahnam S (2017) Combining visual and acoustic features for audio classification tasks. Pattern Recog Lett 88:49\u201356. https:\/\/doi.org\/10.1016\/j.patrec.2017.01.013","journal-title":"Pattern Recog Lett"},{"key":"14734_CR50","doi-asserted-by":"publisher","first-page":"108","DOI":"10.1016\/j.eswa.2015.09.018","volume":"45","author":"L Nanni","year":"2016","unstructured":"Nanni L, Costa YM, Lumini A, Kim MY, Baek SR (2016) Combining visual and acoustic features for music genre classification. Expert Syst Appl 45:108\u2013117. https:\/\/doi.org\/10.1016\/j.eswa.2015.09.018","journal-title":"Expert Syst Appl"},{"key":"14734_CR51","doi-asserted-by":"publisher","unstructured":"Nanni L, Costa YMG, Aguiar RL, Jr CNS, Brahnam S (2018) Ensemble of deep learning, visual and acoustic features for music genre classification. J New Music Res 47(4):383\u2013397. https:\/\/doi.org\/10.1080\/09298215.2018.1438476","DOI":"10.1080\/09298215.2018.1438476"},{"key":"14734_CR52","doi-asserted-by":"publisher","unstructured":"Nanni L, Costa YMG, Lucio DR, Silla C, Brahnam S (2016) Combining visual and acoustic features for bird species classification. In: 2016 IEEE 28th international conference on tools with artificial intelligence (ICTAI), pp 396\u2013401. https:\/\/doi.org\/10.1109\/ICTAI.2016.0067","DOI":"10.1109\/ICTAI.2016.0067"},{"key":"14734_CR53","doi-asserted-by":"publisher","unstructured":"Oo MM, Oo LL (2020) Fusion of Log-Mel spectrogram and GLCM feature in acoustic scene classification. Springer international publishing, Cham, pp 175\u2013187. https:\/\/doi.org\/10.1007\/978-3-030-24344-9-11","DOI":"10.1007\/978-3-030-24344-9-11"},{"key":"14734_CR54","doi-asserted-by":"publisher","first-page":"70","DOI":"10.1016\/j.apacoust.2018.08.003","volume":"142","author":"T \u00d6zseven","year":"2018","unstructured":"\u00d6zseven T (2018) Investigation of the effect of spectrogram images and different texture analysis methods on speech emotion recognition. Appl Acoust 142:70\u201377. https:\/\/doi.org\/10.1016\/j.apacoust.2018.08.003","journal-title":"Appl Acoust"},{"key":"14734_CR55","doi-asserted-by":"crossref","unstructured":"Rahmeni R, Ben Aicha A, Ben Ayed Y (2019) On the contribution of the voice texture for speech spoofing detection. In: 2019 19th International conference on sciences and techniques of automatic control and computer engineering (STA), pp 501\u2013505","DOI":"10.1109\/STA.2019.8717297"},{"issue":"1","key":"14734_CR56","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1109\/TASLP.2014.2375575","volume":"23","author":"A Rakotomamonjy","year":"2015","unstructured":"Rakotomamonjy A, Gasso G (2015) Histogram of gradients of time-frequency representations for audio scene classification. IEEE\/ACM Trans Audio Speech Lang Process 23(1):142\u2013153. https:\/\/doi.org\/10.1109\/TASLP.2014.2375575","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"issue":"3","key":"14734_CR57","doi-asserted-by":"publisher","first-page":"447","DOI":"10.1109\/TMM.2016.2618218","volume":"19","author":"J Ren","year":"2017","unstructured":"Ren J, Jiang X, Yuan J, Magnenat-Thalmann N (2017) Sound-event classification using robust texture features for robot hearing. IEEE Trans Multimed 19(3):447\u2013458. https:\/\/doi.org\/10.1109\/TMM.2016.2618218","journal-title":"IEEE Trans Multimed"},{"key":"14734_CR58","doi-asserted-by":"publisher","unstructured":"Sell G, Clark P (2014) Music tonality features for speech\/music discrimination. 2014. In: 2014 IEEE international conference on acoustics, speech and signal processing (ICASSP) pp 2489\u20132493. https:\/\/doi.org\/10.1109\/ICASSP.2014.6854048","DOI":"10.1109\/ICASSP.2014.6854048"},{"issue":"2","key":"14734_CR59","doi-asserted-by":"publisher","first-page":"485","DOI":"10.1109\/TBME.2018.2849502","volume":"66","author":"RV Sharan","year":"2019","unstructured":"Sharan RV, Abeyratne UR, Swarnkar VR, Porter P (2019) Automatic croup diagnosis using cough sound recognition. IEEE Trans Biomed Eng 66(2):485\u2013495. https:\/\/doi.org\/10.1109\/TBME.2018.2849502","journal-title":"IEEE Trans Biomed Eng"},{"key":"14734_CR60","doi-asserted-by":"publisher","unstructured":"Sharan RV, Moir TJ (2014) Audio surveillance under noisy conditions using time-frequency image feature. In: 2014 19th International conference on digital signal processing, pp 130\u2013135. https:\/\/doi.org\/10.1109\/ICDSP.2014.6900815","DOI":"10.1109\/ICDSP.2014.6900815"},{"key":"14734_CR61","doi-asserted-by":"publisher","unstructured":"Sharan RV, Moir TJ (2015) Cochleagram image feature for improved robustness in sound recognition. In: 2015 IEEE international conference on digital signal processing (DSP), pp 441\u2013444. https:\/\/doi.org\/10.1109\/ICDSP.2015.7251910","DOI":"10.1109\/ICDSP.2015.7251910"},{"key":"14734_CR62","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1016\/j.neucom.2015.02.001","volume":"158","author":"RV Sharan","year":"2015","unstructured":"Sharan RV, Moir TJ (2015) Noise robust audio surveillance using reduced spectrogram image feature and one-against-all SVM. Neurocomputing 158:90\u201399. https:\/\/doi.org\/10.1016\/j.neucom.2015.02.001","journal-title":"Neurocomputing"},{"key":"14734_CR63","doi-asserted-by":"publisher","unstructured":"Sharan RV, Moir TJ (2015) Robust audio surveillance using spectrogram image texture feature. In: 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 1956\u20131960. https:\/\/doi.org\/10.1109\/ICASSP.2015.7178312","DOI":"10.1109\/ICASSP.2015.7178312"},{"key":"14734_CR64","doi-asserted-by":"publisher","unstructured":"Sharan RV, Moir TJ (2015) Subband spectral histogram feature for improved sound recognition in low SNR conditions. In: 2015 IEEE international conference on digital signal processing (DSP), pp 432\u2013435. https:\/\/doi.org\/10.1109\/ICDSP.2015.7251908","DOI":"10.1109\/ICDSP.2015.7251908"},{"key":"14734_CR65","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1016\/j.apacoust.2018.05.030","volume":"140","author":"RV Sharan","year":"2018","unstructured":"Sharan RV, Moir TJ (2018) Pseudo-color cochleagram image feature and sequential feature selection for robust acoustic event recognition. Appl Acoust 140:198\u2013204. https:\/\/doi.org\/10.1016\/j.apacoust.2018.05.030","journal-title":"Appl Acoust"},{"key":"14734_CR66","doi-asserted-by":"publisher","first-page":"107020","DOI":"10.1016\/j.apacoust.2019.107020","volume":"158","author":"G Sharma","year":"2020","unstructured":"Sharma G, Umapathy K, Krishnan S (2020) Trends in audio signal feature extraction methods. Appl Acoust 158:107020. https:\/\/doi.org\/10.1016\/j.apacoust.2019.107020","journal-title":"Appl Acoust"},{"issue":"9","key":"14734_CR67","doi-asserted-by":"publisher","first-page":"1251","DOI":"10.1049\/iet-rsn.2014.0432","volume":"9","author":"X Shi","year":"2015","unstructured":"Shi X, Zhou F, Liu L, Zhao B, Zhang Z (2015) Textural feature extraction based on time-frequency spectrograms of humans and vehicles. IET Radar Sonar Navig 9(9):1251\u20131259. https:\/\/doi.org\/10.1049\/iet-rsn.2014.0432","journal-title":"IET Radar Sonar Navig"},{"key":"14734_CR68","doi-asserted-by":"publisher","unstructured":"Spyrou E, Nikopoulou R, Vernikos I, Mylonas P (2019) Emotion recognition from speech using the bag-of-visual words on audio segment spectrograms. Technologies, vol 7(1). https:\/\/doi.org\/10.3390\/technologies7010020","DOI":"10.3390\/technologies7010020"},{"key":"14734_CR69","unstructured":"Valerio VD, Pereira RM, Costa YMG, Bertolini D, Silla CN (2018) A resampling approach for imbalanceness on music genre classification using spectrograms. In: Thirty-first international florida artificial intelligence research society conference (FLAIRS), pp 500\u2013505"},{"key":"14734_CR70","doi-asserted-by":"publisher","unstructured":"Vyas S, Patil MD, Birajdar GK (2021) Classification of heart sound signals using time-frequency image texture features, Chapter 5, Wiley, pp 81\u2013101. https:\/\/doi.org\/10.1002\/9781119818717.ch5","DOI":"10.1002\/9781119818717.ch5"},{"key":"14734_CR71","doi-asserted-by":"publisher","unstructured":"Wakefield GH (1999) Mathematical representation of joint time-chroma distributions. pp 3807\u20133807-9. https:\/\/doi.org\/10.1117\/12.367679","DOI":"10.1117\/12.367679"},{"key":"14734_CR72","doi-asserted-by":"publisher","unstructured":"Wu H, Zhang M (2012) Gabor-lbp features and combined classifiers for music genre classification. In: Proceedings of the 2012 2nd international conference on computer and information application (ICCIA 2012), pp 419\u2013423. Atlantis Press. https:\/\/doi.org\/10.2991\/iccia.2012.101","DOI":"10.2991\/iccia.2012.101"},{"key":"14734_CR73","doi-asserted-by":"publisher","unstructured":"Wu HQ, Zhang M (2013) Gabor-lbp features and combined classifiers for music genre classification. In: Information technology applications in industry, computer engineering and materials science, advanced materials research, vol 756, pp 4407-4411. Trans Tech Publications Ltd. https:\/\/doi.org\/10.4028\/www.scientific.net\/AMR.756-759.4407","DOI":"10.4028\/www.scientific.net\/AMR.756-759.4407"},{"key":"14734_CR74","doi-asserted-by":"publisher","unstructured":"Wu M, Chen Z, Jang JR, Ren J, Li Y, Lu C (2011) Combining visual and acoustic features for music genre classification. In: 2011 10th International conference on machine learning and applications and workshops, vol 2, pp 124\u2013129. https:\/\/doi.org\/10.1109\/ICMLA.2011.48","DOI":"10.1109\/ICMLA.2011.48"},{"key":"14734_CR75","doi-asserted-by":"publisher","unstructured":"Wu MJ, Jang JSR (2015) Combining acoustic and multilevel visual features for music genre classification. ACM Trans Multimed Comput Commun Appl, vol 12(1). https:\/\/doi.org\/10.1145\/2801127","DOI":"10.1145\/2801127"},{"key":"14734_CR76","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1016\/j.eswa.2019.01.085","volume":"126","author":"J Xie","year":"2019","unstructured":"Xie J, Zhu M (2019) Investigation of acoustic and visual features for acoustic scene classification. Expert Syst Appl 126:20\u201329. https:\/\/doi.org\/10.1016\/j.eswa.2019.01.085","journal-title":"Expert Syst Appl"},{"issue":"6","key":"14734_CR77","doi-asserted-by":"publisher","first-page":"1315","DOI":"10.1109\/TASLP.2017.2690558","volume":"25","author":"W Yang","year":"2017","unstructured":"Yang W, Krishnan S, Yang W, Krishnan S (2017) Combining temporal features by local binary pattern for acoustic scene classification. IEEE\/ACM Trans Audio Speech Lang Proc 25(6):1315\u20131321. https:\/\/doi.org\/10.1109\/TASLP.2017.2690558","journal-title":"IEEE\/ACM Trans Audio Speech Lang Proc"},{"key":"14734_CR78","doi-asserted-by":"publisher","unstructured":"Yang X, Luo J, Wang Y, Zhao X, Li J (2018) Combining auditory perception and visual features for regional recognition of chinese folk songs. In: Proceedings of the 2018 10th international conference on computer and automation engineering, ICCAE 2018. Association for computing machinery, New York, NY, USA, pp 75\u201381. https:\/\/doi.org\/10.1145\/3192975.3193006","DOI":"10.1145\/3192975.3193006"},{"key":"14734_CR79","doi-asserted-by":"publisher","unstructured":"Yasmin G, Das AK (2019) Speech and non-speech audio files discrimination extracting textural and acoustic features. In: Bhattacharyya S, Mukherjee A, Bhaumik H, Das S, Yoshida K (eds) Recent trends in signal and image processing. Springer Singapore, Singapore, pp 197\u2013206. https:\/\/doi.org\/10.1007\/978-981-10-8863-6_20","DOI":"10.1007\/978-981-10-8863-6_20"},{"key":"14734_CR80","doi-asserted-by":"publisher","unstructured":"Ye J, Kobayashi T, Murakawa M, Higuchi T (2015) Acoustic scene classification based on sound textures and events. In: Proceedings of the 23rd ACM international conference on multimedia. Association for computing machinery, New York, NY, USA, pp 1291\u20131294. https:\/\/doi.org\/10.1145\/2733373.2806389","DOI":"10.1145\/2733373.2806389"},{"key":"14734_CR81","doi-asserted-by":"publisher","unstructured":"Yu G, Slotine JJE (2009) Audio classification from time-frequency texture. In: 2009 IEEE international conference on acoustics, speech and signal processing pp 1677\u20131680. https:\/\/doi.org\/10.1109\/ICASSP.2009.4959924","DOI":"10.1109\/ICASSP.2009.4959924"},{"key":"14734_CR82","doi-asserted-by":"publisher","unstructured":"Zhang S, Zhao Z, Xu Z, Bellisario K, Pijanowski BC (2018) Automatic bird vocalization identification based on fusion of spectral pattern and texture features. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 271\u2013275. https:\/\/doi.org\/10.1109\/ICASSP.2018.8462156","DOI":"10.1109\/ICASSP.2018.8462156"},{"issue":"1","key":"14734_CR83","doi-asserted-by":"publisher","first-page":"1","DOI":"10.3390\/electronics12010001","volume":"9","author":"Y Zhang","year":"2020","unstructured":"Zhang Y, Dai S, Song W, Zhang L, Li D (2020) Exposing speech resampling manipulation by local texture analysis on spectrogram images. Electronics 9(1):1\u201323. https:\/\/doi.org\/10.3390\/electronics9010023","journal-title":"Electronics"},{"key":"14734_CR84","doi-asserted-by":"publisher","first-page":"107970","DOI":"10.1016\/j.apacoust.2021.107970","volume":"178","author":"Y Zhang","year":"2021","unstructured":"Zhang Y, Zhang K, Wang J, Su Y (2021) Robust acoustic event recognition using AVMD-PWVD time-frequency image. Appl Acoust 178:107970. https:\/\/doi.org\/10.1016\/j.apacoust.2021.107970","journal-title":"Appl Acoust"},{"key":"14734_CR85","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1016\/j.ecoinf.2018.08.007","volume":"48","author":"RH Zottesso","year":"2018","unstructured":"Zottesso RH, Costa Y, Bertolini D, Oliveira L (2018) Bird species identification using spectrogram and dissimilarity approach. Ecol Inform 48:187\u2013197. https:\/\/doi.org\/10.1109\/ICASSP.1979.1170735","journal-title":"Ecol Inform"},{"key":"14734_CR86","doi-asserted-by":"publisher","unstructured":"Zue V, Cole R (1979) Experiments on spectrogram reading. In: ICASSP \u201979. IEEE international conference on acoustics, speech, and signal processing, vol 4, pp 116\u2013119. https:\/\/doi.org\/10.1109\/ICASSP.1979.1170735","DOI":"10.1109\/ICASSP.1979.1170735"},{"key":"14734_CR87","doi-asserted-by":"publisher","unstructured":"Zue V, Lamel L (1986) An expert spectrogram reader: a knowledge-based approach to speech recognition. In: ICASSP \u201986. IEEE international conference on acoustics, speech, and signal processing, vol 11, pp 1197\u20131200. https:\/\/doi.org\/10.1109\/ICASSP.1986.1168798","DOI":"10.1109\/ICASSP.1986.1168798"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-14734-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-14734-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-14734-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,20]],"date-time":"2023-09-20T10:19:07Z","timestamp":1695205147000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-14734-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,16]]},"references-count":87,"journal-issue":{"issue":"23","published-print":{"date-parts":[[2023,9]]}},"alternative-id":["14734"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-14734-1","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,3,16]]},"assertion":[{"value":"29 December 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 July 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 February 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 March 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}