{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T00:38:28Z","timestamp":1780965508381,"version":"3.54.1"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2023,7,6]],"date-time":"2023-07-06T00:00:00Z","timestamp":1688601600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,7,6]],"date-time":"2023-07-06T00:00:00Z","timestamp":1688601600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-15277-1","type":"journal-article","created":{"date-parts":[[2023,7,6]],"date-time":"2023-07-06T12:03:06Z","timestamp":1688644986000},"page":"13527-13542","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":20,"title":["Music genre classification based on res-gated CNN and attention mechanism"],"prefix":"10.1007","volume":"83","author":[{"given":"Changjiang","family":"Xie","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3174-6203","authenticated-orcid":false,"given":"Huazhu","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kaituo","family":"Mi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhouhan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiawen","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Honglin","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Renjie","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haofeng","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,7,6]]},"reference":[{"key":"15277_CR1","doi-asserted-by":"publisher","first-page":"252","DOI":"10.1016\/j.eswa.2019.06.040","volume":"136","author":"S Abdoli","year":"2019","unstructured":"Abdoli S, Cardinal P, Koerich A L (2019) End-to-end environmental sound classification using a 1d convolutional neural network. Expert Syst Appl 136:252\u2013263","journal-title":"Expert Syst Appl"},{"issue":"16","key":"15277_CR2","doi-asserted-by":"publisher","first-page":"4114","DOI":"10.1109\/TSP.2014.2326991","volume":"62","author":"J And\u00e9n","year":"2014","unstructured":"And\u00e9n J, Mallat S (2014) Deep scattering spectrum. IEEE Trans Signal Process 62(16):4114\u20134128","journal-title":"IEEE Trans Signal Process"},{"key":"15277_CR3","unstructured":"Ba J, Mnih V, Kavukcuoglu K (2015) Multiple object recognition with visual attention. In: ICLR (Poster)"},{"key":"15277_CR4","unstructured":"Cano P, G\u00f3mez E, Gouyon F, Herrera P, Koppenberger M, Ong B, Serra X, Streich S, Wack N (2006) Ismir 2004 audio description contest. Music Technology Group of the Universitat Pompeu Fabra, Tech. Rep"},{"key":"15277_CR5","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers. In: European conference on computer vision. Springer, pp 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"15277_CR6","doi-asserted-by":"crossref","unstructured":"Chen C-F R, Fan Q, Panda R (2021) Crossvit: cross-attention multi-scale vision transformer for image classification. In: Proceedings of the IEEE\/CVF international conference on computer vision , pp 357\u2013366","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"15277_CR7","doi-asserted-by":"crossref","unstructured":"Chen X, Wu Y, Wang Z, Liu S, Li J (2021) Developing real-time streaming transformer transducer for speech recognition on large-scale dataset. In: ICASSP 2021-2021 IEEE International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5904\u20135908","DOI":"10.1109\/ICASSP39728.2021.9413535"},{"key":"15277_CR8","doi-asserted-by":"crossref","unstructured":"Cho K, van Merri\u00ebnboer B, Bahdanau D, Bengio Y (2014) On the properties of neural machine translation: encoder\u2013decoder approaches. In: Proceedings of SSST-8, eighth workshop on syntax, semantics and structure in statistical translation, pp 103\u2013111","DOI":"10.3115\/v1\/W14-4012"},{"key":"15277_CR9","doi-asserted-by":"crossref","unstructured":"Choi K, Fazekas G, Sandler M, Cho K (2017) Convolutional recurrent neural networks for music classification. In: 2017 IEEE International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 2392\u20132396","DOI":"10.1109\/ICASSP.2017.7952585"},{"key":"15277_CR10","unstructured":"Ciresan D C, Meier U, Masci J, Gambardella L M, Schmidhuber J (2011) Flexible, high performance convolutional neural networks for image classification. In: Twenty-second international joint conference on artificial intelligence"},{"issue":"1","key":"15277_CR11","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1109\/TIT.1967.1053964","volume":"13","author":"T Cover","year":"1967","unstructured":"Cover T, Hart P (1967) Nearest neighbor pattern classification. IEEE Trans Inform Theory 13(1):21\u201327","journal-title":"IEEE Trans Inform Theory"},{"key":"15277_CR12","doi-asserted-by":"crossref","unstructured":"Dai J, Liang S, Xue W, Ni C, Liu W (2016) Long short-term memory recurrent neural network based segment features for music genre classification. In: 2016 10th International symposium on chinese spoken language processing (ISCSLP). IEEE, pp 1\u20135","DOI":"10.1109\/ISCSLP.2016.7918369"},{"issue":"11","key":"15277_CR13","doi-asserted-by":"publisher","first-page":"9813","DOI":"10.1109\/TGRS.2020.3044958","volume":"59","author":"Y Dai","year":"2021","unstructured":"Dai Y, Wu Y, Zhou F, Barnard K (2021) Attentional local contrast networks for infrared small target detection. IEEE Trans Geosci Remote Sens 59 (11):9813\u20139824","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"15277_CR14","doi-asserted-by":"crossref","unstructured":"Dieleman S, Schrauwen B (2014) End-to-end learning for music audio. In: 2014 IEEE International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6964\u20136968","DOI":"10.1109\/ICASSP.2014.6854950"},{"issue":"12","key":"15277_CR15","doi-asserted-by":"publisher","first-page":"3150","DOI":"10.1109\/TMM.2019.2918739","volume":"21","author":"Y Dong","year":"2019","unstructured":"Dong Y, Yang X, Zhao X, Li J (2019) Bidirectional convolutional recurrent sparse network (bcrsn): an efficient model for music emotion recognition. IEEE Trans Multimed 21(12):3150\u20133163","journal-title":"IEEE Trans Multimed"},{"key":"15277_CR16","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S et al (2020) An image is worth 16x16 words: transformers for image recognition at scale. arXiv:2010.11929"},{"issue":"1","key":"15277_CR17","doi-asserted-by":"publisher","first-page":"295","DOI":"10.1002\/aris.1440370108","volume":"37","author":"JS Downie","year":"2003","unstructured":"Downie J S (2003) Music information retrieval. Ann Rev Inform Sci Technol 37(1):295\u2013340","journal-title":"Ann Rev Inform Sci Technol"},{"key":"15277_CR18","unstructured":"Dauphin YN, Fan A, Auli M, Grangier D (2017) Language modeling with gated convolutional networks. In: International conference on machine learning. PMLR, pp 933\u2013941"},{"issue":"1","key":"15277_CR19","first-page":"6340","volume":"18","author":"M Freitag","year":"2017","unstructured":"Freitag M, Amiriparian S, Pugachevskiy S, Cummins N, Schuller B (2017) audeep: unsupervised learning of representations from audio with deep recurrent neural networks. J Mach Learn Res 18(1):6340\u20136344","journal-title":"J Mach Learn Res"},{"key":"15277_CR20","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"15277_CR21","unstructured":"Hermann K M, Kocisky T, Grefenstette E, Espeholt L, Kay W, Suleyman M, Blunsom P (2015) Teaching machines to read and comprehend. Advances in Neural Information Processing Systems, 28"},{"issue":"7","key":"15277_CR22","doi-asserted-by":"publisher","first-page":"5966","DOI":"10.1109\/TGRS.2020.3015157","volume":"59","author":"D Hong","year":"2020","unstructured":"Hong D, Gao L, Yao J, Zhang B, Plaza A, Chanussot J (2020) Graph convolutional networks for hyperspectral image classification. IEEE Trans Geosci Remote Sens 59(7):5966\u20135978","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"15277_CR23","doi-asserted-by":"crossref","unstructured":"Kereliuk C, Sturm B L, Larsen J (2015) Deep learning, audio adversaries, and music content analysis. In: 2015 IEEE Workshop on applications of signal processing to audio and acoustics (WASPAA). IEEE, pp 1\u20135","DOI":"10.1109\/WASPAA.2015.7336950"},{"issue":"2","key":"15277_CR24","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1109\/JSTSP.2019.2909479","volume":"13","author":"T Kim","year":"2019","unstructured":"Kim T, Lee J, Nam J (2019) Comparison and analysis of samplecnn architectures for audio classification. IEEE J Selected Topics Signal Process 13(2):285\u2013297","journal-title":"IEEE J Selected Topics Signal Process"},{"key":"15277_CR25","unstructured":"Koerich K M, Esmailpour M, Abdoli S, Britto A S, Koerich A L (2020) Cross-representation transferability of adversarial attacks: from spectrograms to audio waveforms. In: 2020 International joint conference on neural networks (IJCNN). IEEE, pp 1\u20137"},{"key":"15277_CR26","doi-asserted-by":"crossref","unstructured":"Kumar D P, Sowmya BJ, Srinivasa KG et al (2016) A comparative study of classifiers for music genre classification based on feature extractors. In: 2016 IEEE Distributed computing, VLSI, electrical circuits and robotics (DISCOVER). IEEE, pp 190\u2013194","DOI":"10.1109\/DISCOVER.2016.7806258"},{"issue":"2010","key":"15277_CR27","first-page":"1","volume":"10","author":"TL Li","year":"2010","unstructured":"Li TL, Chan A B, Chun AH (2010) Automatic musical pattern feature extraction using convolutional neural network. Genre 10(2010):1\u20131","journal-title":"Genre"},{"key":"15277_CR28","doi-asserted-by":"crossref","unstructured":"Ling W, Dyer C, Black A W, Trancoso I (2015) Two\/too simple adaptations of word2vec for syntax problems. In: Proceedings of the 2015 conference of the North American chapter of the association for computational linguistics: human language technologies, pp 1299\u20131304","DOI":"10.3115\/v1\/N15-1142"},{"key":"15277_CR29","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"15277_CR30","unstructured":"Marchand U, Peeters G (2016) The extended ballroom dataset"},{"key":"15277_CR31","unstructured":"Malhotra P, Vig L, Shroff G, Agarwal P et al (2015) Long short term memory networks for anomaly detection in time series. In: Proceedings, vol 89, pp 89\u201394"},{"key":"15277_CR32","doi-asserted-by":"crossref","unstructured":"Meng F, Lu Z, Wang M, Li H, Jiang W, Liu Q (2015) Encoding source language with convolutional neural network for machine translation. arXiv:1503.01838","DOI":"10.3115\/v1\/P15-1003"},{"key":"15277_CR33","unstructured":"Mnih V, Heess N, Graves A et al (2014) Recurrent models of visual attention. Advances in Neural Information Processing Systems, 27"},{"key":"15277_CR34","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1016\/j.patrec.2017.01.013","volume":"88","author":"L Nanni","year":"2017","unstructured":"Nanni L, Costa YM, Lucio DR, Silla CN Jr, Brahnam S (2017) Combining visual and acoustic features for audio classification tasks. Pattern Recogn Lett 88:49\u201356","journal-title":"Pattern Recogn Lett"},{"key":"15277_CR35","unstructured":"Ngai H, Park Y, Chen J, Parsapoor M (2021) Transformer-based models for question answering on covid19. arXiv:2101.11432"},{"key":"15277_CR36","doi-asserted-by":"crossref","unstructured":"Pons J, Slizovskaia O, Gong R, G\u00f3mez E, Serra X (2017) Timbre analysis of music audio signals with convolutional neural networks. In: 2017 25th European Signal Processing Conference (EUSIPCO). IEEE, pp 2744\u20132748","DOI":"10.23919\/EUSIPCO.2017.8081710"},{"key":"15277_CR37","doi-asserted-by":"crossref","unstructured":"Rush A M, Chopra S, Weston J (2015) A neural attention model for abstractive sentence summarization. In: Proceedings of the 2015 conference on empirical methods in natural language processing , pp 379\u2013389","DOI":"10.18653\/v1\/D15-1044"},{"key":"15277_CR38","doi-asserted-by":"crossref","unstructured":"Shi Y, Wang Y, Wu C, Yeh C-F, Chan J, Zhang F, Le D, Seltzer M (2021) Emformer: efficient memory transformer based acoustic model for low latency streaming speech recognition. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6783\u20136787","DOI":"10.1109\/ICASSP39728.2021.9414560"},{"key":"15277_CR39","doi-asserted-by":"crossref","unstructured":"Sigtia S, Dixon S (2014) Improved music feature learning with deep neural networks. In: 2014 IEEE International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6959\u20136963","DOI":"10.1109\/ICASSP.2014.6854949"},{"key":"15277_CR40","doi-asserted-by":"crossref","unstructured":"Silla Jr C N, Kaestner Celso AA, Koerich A L (2007) Automatic music genre classification using ensemble of classifiers. In: 2007 IEEE International conference on systems, man and cybernetics . IEEE, pp 1687\u20131692","DOI":"10.1109\/ICSMC.2007.4414136"},{"issue":"5","key":"15277_CR41","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1109\/TSA.2002.800560","volume":"10","author":"G Tzanetakis","year":"2002","unstructured":"Tzanetakis G, Cook P (2002) Musical genre classification of audio signals. IEEE Trans Speech Audio Process 10(5):293\u2013302","journal-title":"IEEE Trans Speech Audio Process"},{"key":"15277_CR42","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez A N, Kaiser L, Polosukhin I (2017) Attention is all you need. Advances in Neural Information Processing Systems, 30"},{"key":"15277_CR43","unstructured":"Wang F, Tax DMJ (2016) Survey on the attention based rnn model and its applications in computer vision. arXiv:1601.06823"},{"key":"15277_CR44","doi-asserted-by":"crossref","unstructured":"Wang Z, Muknahallipatna S, Fan M, Okray A, Lan C (2019) Music classification using an improved crnn with multi-directional spatial dependencies in both time and frequency dimensions. In: 2019 International joint conference on neural networks (IJCNN). IEEE, pp 1\u20138","DOI":"10.1109\/IJCNN.2019.8852128"},{"key":"15277_CR45","doi-asserted-by":"publisher","first-page":"311","DOI":"10.1162\/tacl_a_00368","volume":"9","author":"W Xu","year":"2021","unstructured":"Xu W, Carpuat M (2021) Editor: an edit-based transformer with repositioning for neural machine translation with soft lexical constraints. Trans Assoc Comput Linguist 9:311\u2013328","journal-title":"Trans Assoc Comput Linguist"},{"key":"15277_CR46","unstructured":"Xu C, Maddage N C, Shao X, Cao F, Tian Q (2003) Musical genre classification using support vector machines. In: 2003 IEEE International conference on acoustics, speech, and signal processing, 2003. Proceedings.(ICASSP\u201903)., vol 5. IEEE, pp V\u2013429"},{"key":"15277_CR47","unstructured":"Xu K, Ba J, Kiros R, Cho K, Courville A, Salakhudinov R, Zemel R, Bengio Y (2015) Show, attend and tell: neural image caption generation with visual attention. In: International conference on machine learning. PMLR, pp 2048\u20132057"},{"key":"15277_CR48","doi-asserted-by":"crossref","unstructured":"Yang H, Zhang W-Q (2019) Music genre classification using duplicated convolutional layers in neural networks.. In: INTERSPEECH, pp 3382\u20133386","DOI":"10.21437\/Interspeech.2019-1298"},{"key":"15277_CR49","doi-asserted-by":"publisher","first-page":"19629","DOI":"10.1109\/ACCESS.2020.2968170","volume":"8","author":"R Yang","year":"2020","unstructured":"Yang R, Feng L, Wang H, Yao J, Luo S (2020) Parallel recurrent convolutional neural networks-based music genre classification method for mobile devices. IEEE Access 8:19629\u201319637","journal-title":"IEEE Access"},{"key":"15277_CR50","doi-asserted-by":"crossref","unstructured":"Yang C-H H, Qi J, Chen S Y-C, Chen P-Y, Siniscalchi S M, Ma X, Lee C-H (2021) Decentralizing feature extraction with quantum convolutional neural network for automatic speech recognition. In: ICASSP 2021-2021 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 6523\u20136527","DOI":"10.1109\/ICASSP39728.2021.9413453"},{"key":"15277_CR51","doi-asserted-by":"crossref","unstructured":"Zhang P, Zheng X, Zhang W, Li S, Qian S, He W, Zhang S, Wang Z (2015) A deep neural network for modeling music. In: Proceedings of the 5th ACM on international conference on multimedia retrieval, pp 379\u2013386","DOI":"10.1145\/2671188.2749367"},{"key":"15277_CR52","doi-asserted-by":"crossref","unstructured":"Zhang W, Lei W, Xu X, Xing X (2016) Improved music genre classification with convolutional neural networks. In: Interspeech, pp 3304\u20133308","DOI":"10.21437\/Interspeech.2016-1236"},{"key":"15277_CR53","doi-asserted-by":"crossref","unstructured":"Zhang T, Gong X, Chen CLP (2021) Bmt-net: broad multitask transformer network for sentiment analysis. IEEE Transactions on Cybernetics","DOI":"10.1109\/TCYB.2021.3050508"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15277-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-15277-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15277-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,26]],"date-time":"2024-01-26T10:30:32Z","timestamp":1706265032000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-15277-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,6]]},"references-count":53,"journal-issue":{"issue":"5","published-online":{"date-parts":[[2024,2]]}},"alternative-id":["15277"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-15277-1","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,7,6]]},"assertion":[{"value":"21 May 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 February 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 April 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 July 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"No potential conflict of interest was reported by the authors.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}}]}}