{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T00:46:11Z","timestamp":1777941971696,"version":"3.51.4"},"reference-count":87,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T00:00:00Z","timestamp":1758585600000},"content-version":"vor","delay-in-days":265,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"DOI":"10.13039\/100020595","name":"National Science and Technology Council","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100020595","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006403","name":"Feng Chia University","doi-asserted-by":"publisher","award":["NSTC 113-2221-E-035-072"],"award-info":[{"award-number":["NSTC 113-2221-E-035-072"]}],"id":[{"id":"10.13039\/501100006403","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Procedia Computer Science"],"published-print":{"date-parts":[[2025]]},"DOI":"10.1016\/j.procs.2025.09.288","type":"journal-article","created":{"date-parts":[[2025,11,6]],"date-time":"2025-11-06T22:13:48Z","timestamp":1762467228000},"page":"1679-1688","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Discerning Music Genres: Exploring Neural Network Architectures for Automated Classification"],"prefix":"10.1016","volume":"270","author":[{"given":"Eric","family":"Odle","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pei-Chun","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Amin","family":"Farjudian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.procs.2025.09.288_bib1","unstructured":"Stevens, C., Byron, T.: Universals in music processing. The Oxford handbook of music psychology, 14\u201323 (2009)."},{"key":"10.1016\/j.procs.2025.09.288_bib2","doi-asserted-by":"crossref","unstructured":"Fritz, T., Jentschke, S., Gosselin, N., Sammler, D., Peretz, I., Turner, R., Friederici, A.D., Koelsch, S.: Universal recognition of three basic emotions in music. Current biology 19(7), 573\u2013576 (2009).","DOI":"10.1016\/j.cub.2009.02.058"},{"key":"10.1016\/j.procs.2025.09.288_bib3","doi-asserted-by":"crossref","unstructured":"Aucouturier, J.-J., Pachet, F.: Representing musical genre: A state of the art. Journal of new music research 32(1), 83\u201393 (2003).","DOI":"10.1076\/jnmr.32.1.83.16801"},{"key":"10.1016\/j.procs.2025.09.288_bib4","unstructured":"Holt, F.: Genre formation in popular music. Musik & Forskning 28, 77\u201396 (2003)."},{"key":"10.1016\/j.procs.2025.09.288_bib5","doi-asserted-by":"crossref","unstructured":"Lena, J.C., Peterson, R.A.: Classification as culture: Types and trajectories of music genres. American sociological review 73(5), 697\u2013718 (2008).","DOI":"10.1177\/000312240807300501"},{"key":"10.1016\/j.procs.2025.09.288_bib6","doi-asserted-by":"crossref","unstructured":"Lee, C.-H., Shih, J.-L., Yu, K.-M., Lin, H.-S.: Automatic music genre classification based on modulation spectral analysis of spectral and cepstral features. IEEE Transactions on Multimedia 11(4), 670\u2013682 (2009).","DOI":"10.1109\/TMM.2009.2017635"},{"key":"10.1016\/j.procs.2025.09.288_bib7","unstructured":"Mandel, M.I. and Ellis, D.: Song-Level Features and Support Vector Machines for Music Classification. In ISMIR, pp. 594-599 (2005)."},{"key":"10.1016\/j.procs.2025.09.288_bib8","unstructured":"Scaringella, N. and Zoia, G.: On the Modeling of Time Information for Automatic Genre Recognition Systems in Audio Signals. In ISMIR, pp. 666-671 (2005)."},{"key":"10.1016\/j.procs.2025.09.288_bib9","unstructured":"Burred, J.J., Lerch, A.: A hierarchical approach to automatic musical genre classification. In: Proceedings of the 6th International Conference on Digital Audio Effects, pp. 8\u201311 (2003)."},{"key":"10.1016\/j.procs.2025.09.288_bib10","doi-asserted-by":"crossref","unstructured":"Silla, C.N., Koerich, A.L., Kaestner, C.A.: A machine learning approach to automatic music genre classification. Journal of the Brazilian Computer Society 14, 7\u201318 (2008).","DOI":"10.1007\/BF03192561"},{"key":"10.1016\/j.procs.2025.09.288_bib11","doi-asserted-by":"crossref","unstructured":"Rosner, A., Kostek, B.: Automatic music genre classification based on musical instrument track separation. Journal of Intelligent Information Systems 50, 363\u2013384 (2018).","DOI":"10.1007\/s10844-017-0464-5"},{"key":"10.1016\/j.procs.2025.09.288_bib12","doi-asserted-by":"crossref","unstructured":"Vishnupriya, S., Meenakshi, K.: Automatic music genre classification using convolution neural network. In: 2018 International Conference on Computer Communication and Informatics (ICCCI), pp. 1\u20134 (2018).","DOI":"10.1109\/ICCCI.2018.8441340"},{"key":"10.1016\/j.procs.2025.09.288_bib13","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"10.1016\/j.procs.2025.09.288_bib14","unstructured":"Sun, P., Cao, J., Jiang, Y., Zhang, R., Xie, E., Yuan, Z., Wang, C., Luo, P.: Transtrack: Multiple object tracking with transformer. arXiv preprint arXiv:2012.15460 (2020)."},{"key":"10.1016\/j.procs.2025.09.288_bib15","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., Ellis, D.P., Freedman, D., Jansen, A., Lawrence, W., Moore, R.C., Plakal, M. and Ritter, M.: Audio set: An ontology and human-labeled dataset for audio events. In 2017 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 776-780 (2017).","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"10.1016\/j.procs.2025.09.288_bib16","unstructured":"Van Den Oord, A., Dieleman, S., Zen, H., Simonyan, K., Vinyals, O., Graves, A., Kalchbrenner, N., Senior, A. and Kavukcuoglu, K.: Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499, 12 (2016)."},{"key":"10.1016\/j.procs.2025.09.288_bib17","unstructured":"Wyse, L., Kamath, P. and Gupta, C.: An Integrated System Architecture for Generative Audio Modeling."},{"key":"10.1016\/j.procs.2025.09.288_bib18","unstructured":"Bridle, J.S. and Brown, M.D.: An experimental automatic word recognition system. JSRU report, 1003(5), 33 (1974)."},{"key":"10.1016\/j.procs.2025.09.288_bib19","unstructured":"Mermelstein, P.: Distance measures for speech recognition, psychological and instrumental. Pattern recognition and artificial intelligence, 116, 374-388 (1976)."},{"key":"10.1016\/j.procs.2025.09.288_bib20","unstructured":"Kelly, A.C. and Gobl, C.: A comparison of mel-frequency cepstral coefficient (MFCC) calculation techniques. Journal of Computing, 3(10), 62-66 (2011)."},{"key":"10.1016\/j.procs.2025.09.288_bib21","doi-asserted-by":"crossref","unstructured":"Sahidullah, M. and Saha, G.: Design, analysis and experimental evaluation of block based transformation in MFCC computation for speaker recognition. Speech communication, 54(4), pp.543-565 (2012).","DOI":"10.1016\/j.specom.2011.11.004"},{"key":"10.1016\/j.procs.2025.09.288_bib22","doi-asserted-by":"crossref","unstructured":"Ghosh, P., Mahapatra, S., Jana, S. and Jha, R.K.: A study on music genre classification using machine learning. International Journal of Engineering Business and Social Science, 1(04), 308-320 (2023).","DOI":"10.58451\/ijebss.v1i04.55"},{"key":"10.1016\/j.procs.2025.09.288_bib23","doi-asserted-by":"crossref","unstructured":"Chathuranga, D. and Jayaratne, L.: Automatic music genre classification of audio signals with machine learning approaches. GSTF Journal on Computing (JoC), 3, 1-12 (2013).","DOI":"10.7603\/s40601-013-0014-0"},{"key":"10.1016\/j.procs.2025.09.288_bib24","doi-asserted-by":"crossref","unstructured":"Pelchat, N. and Gelowitz, C.M.: Neural network music genre classification. Canadian Journal of Electrical and Computer Engineering, 43(3), 170-173 (2020).","DOI":"10.1109\/CJECE.2020.2970144"},{"key":"10.1016\/j.procs.2025.09.288_bib25","doi-asserted-by":"crossref","unstructured":"Cai, H., Pu, T., Luo, Y. and Zhou, X., 2021, May. Music genre prediction based on machine learning. In 2021 IEEE International Conference on Artificial Intelligence and Industrial Design (AIID), pp. 198-201 (2021).","DOI":"10.1109\/AIID51893.2021.9456579"},{"key":"10.1016\/j.procs.2025.09.288_bib26","doi-asserted-by":"crossref","unstructured":"Prabhakar, S.K. and Lee, S.W.: Holistic approaches to music genre classification using efficient transfer and deep learning techniques. Expert Systems with Applications, 211, pp.118636 (2023).","DOI":"10.1016\/j.eswa.2022.118636"},{"key":"10.1016\/j.procs.2025.09.288_bib27","doi-asserted-by":"crossref","unstructured":"Walczak, S., Cerpa, N.: Artificial neural networks. In: Advanced Methodologies and Technologies in Artificial Intelligence, Computer Simulation, and Human- computer Interaction, pp. 40\u201353 (2019).","DOI":"10.4018\/978-1-5225-7368-5.ch004"},{"key":"10.1016\/j.procs.2025.09.288_bib28","doi-asserted-by":"crossref","unstructured":"Rosenblatt, F.: The perceptron: a probabilistic model for information storage and organization in the brain [j]. Psychological review 65(6), 386\u2013408 (1958).","DOI":"10.1037\/h0042519"},{"key":"10.1016\/j.procs.2025.09.288_bib29","doi-asserted-by":"crossref","unstructured":"Rumelhart, D.E., McClelland, J.L., Group, P.R., et al.: Parallel distributed processing, volume 1: Explorations in the microstructure of cognition: Foundations (1986).","DOI":"10.7551\/mitpress\/5236.001.0001"},{"key":"10.1016\/j.procs.2025.09.288_bib30","doi-asserted-by":"crossref","unstructured":"Peretto, P.: Collective properties of neural networks: a statistical physics approach. Biological cybernetics 50(1), 51\u201362 (1984).","DOI":"10.1007\/BF00317939"},{"key":"10.1016\/j.procs.2025.09.288_bib31","doi-asserted-by":"crossref","unstructured":"Frean, M.: The upstart algorithm: A method for constructing and training feedforward neural networks. Neural computation 2(2), 198\u2013209 (1990).","DOI":"10.1162\/neco.1990.2.2.198"},{"key":"10.1016\/j.procs.2025.09.288_bib32","doi-asserted-by":"crossref","unstructured":"Hubel, D.H., Wiesel, T.N.: Brain mechanisms of vision. Scientific American 241(3), 150\u2013163 (1979).","DOI":"10.1038\/scientificamerican0979-150"},{"key":"10.1016\/j.procs.2025.09.288_bib33","unstructured":"LeCun, Y., Bengio, Y., et al.: Convolutional networks for images, speech, and time series. The handbook of brain theory and neural networks, 3361(10) (1995)."},{"key":"10.1016\/j.procs.2025.09.288_bib34","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., Fei-Fei, L.: Imagenet: A large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255 (2009).","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"10.1016\/j.procs.2025.09.288_bib35","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems 25 (2012)."},{"key":"10.1016\/j.procs.2025.09.288_bib36","series-title":"A comparison of LSTM and GRU networks for learning symbolic sequences. In: Science and Information Conference, pp. 771\u2013785","author":"Cahuantzi","year":"2023"},{"key":"10.1016\/j.procs.2025.09.288_bib37","doi-asserted-by":"crossref","unstructured":"Cho, K., Van Merri\u02d9enboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., Bengio, Y.: Learning phrase representations using RNN encoder- decoder for statistical machine translation. arXiv preprint arXiv:1406.1078 (2014).","DOI":"10.3115\/v1\/D14-1179"},{"key":"10.1016\/j.procs.2025.09.288_bib38","doi-asserted-by":"crossref","unstructured":"Hassanzadeh, H.R., Wang, M.D.: Deeperbind: Enhancing prediction of sequence specificities of DNA binding proteins. In: 2016 IEEE International Conference on Bioinformatics and Biomedicine (BIBM), pp. 178\u2013183 (2016).","DOI":"10.1109\/BIBM.2016.7822515"},{"key":"10.1016\/j.procs.2025.09.288_bib39","doi-asserted-by":"crossref","unstructured":"Pollastri, G., Przybylski, D., Rost, B., Baldi, P.: Improving the prediction of protein secondary structure in three and eight classes using recurrent neural networks and profiles. Proteins: Structure, Function, and Bioinformatics, 47(2), 228\u2013235 (2002).","DOI":"10.1002\/prot.10082"},{"key":"10.1016\/j.procs.2025.09.288_bib40","doi-asserted-by":"crossref","unstructured":"Hill, S.T., Kuintzle, R., Teegarden, A., Merrill III, E., Danaee, P., Hendrix, D.A.: A deep recurrent neural network discovers complex biological rules to decipher RNA protein-coding potential. Nucleic acids research 46(16), 8105\u20138113 (2018).","DOI":"10.1093\/nar\/gky567"},{"key":"10.1016\/j.procs.2025.09.288_bib41","doi-asserted-by":"crossref","unstructured":"Giles, C.L., Lawrence, S., Tsoi, A.C.: Rule inference for financial prediction using recurrent neural networks. In: Proceedings of the IEEE\/IAFE 1997 Computational Intelligence for Financial Engineering (CIFEr), pp. 253\u2013259 (1997).","DOI":"10.1109\/CIFER.1997.618945"},{"key":"10.1016\/j.procs.2025.09.288_bib42","unstructured":"Di Persio, L., Honchar, O., et al.: Recurrent neural networks approach to the financial forecast of google assets. International journal of Mathematics and Computers in simulation 11, 7\u201313 (2017)."},{"key":"10.1016\/j.procs.2025.09.288_bib43","doi-asserted-by":"crossref","unstructured":"Moghar, A., Hamiche, M.: Stock market prediction using LSTM recurrent neural network. Procedia computer science 170, 1168\u20131173 (2020).","DOI":"10.1016\/j.procs.2020.03.049"},{"key":"10.1016\/j.procs.2025.09.288_bib44","series-title":"Application of lstm neural networks in language mod- elling. In: Text, Speech, and Dialogue: 16th International Conference, TSD 2013, Pilsen, Czech Republic, September 1-5, 2013. Proceedings 16, pp. 105\u2013112","author":"Soutner","year":"2013"},{"key":"10.1016\/j.procs.2025.09.288_bib45","unstructured":"Trofmovich, J.: Comparison of neural network architectures for sentiment analysis of Russian tweets. In: Computational Linguistics and Intellectual Technologies: Proceedings of the International Conference Dialogue, pp. 50\u201359 (2016)."},{"key":"10.1016\/j.procs.2025.09.288_bib46","unstructured":"Merity, S., Keskar, N.S., Socher, R.: Regularizing and optimizing LSTM language models. arXiv preprint arXiv:1708.02182 (2017)."},{"key":"10.1016\/j.procs.2025.09.288_bib47","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"10.1016\/j.procs.2025.09.288_bib48","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"10.1016\/j.procs.2025.09.288_bib49","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., Stoyanov, V.: Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"10.1016\/j.procs.2025.09.288_bib50","doi-asserted-by":"crossref","unstructured":"Floridi, L., Chiriatti, M.: Gpt-3: Its nature, scope, limits, and consequences. Minds and Machines 30, 681\u2013694 (2020).","DOI":"10.1007\/s11023-020-09548-1"},{"key":"10.1016\/j.procs.2025.09.288_bib51","unstructured":"Radford, A., Narasimhan, K., Salimans, T., Sutskever, I., et al.: Improving language understanding by generative pre-training (2018)."},{"key":"10.1016\/j.procs.2025.09.288_bib52","doi-asserted-by":"crossref","unstructured":"Lauer, J., Zhou, M., Ye, S., Menegas, W., Schneider, S., Nath, T., Rahman, M.M., Di Santo, V., Soberanes, D., Feng, G., et al.: Multianimal pose estimation, identification and tracking with deeplabcut. Nature Methods 19(4), 496\u2013504 (2022).","DOI":"10.1038\/s41592-022-01443-0"},{"key":"10.1016\/j.procs.2025.09.288_bib53","doi-asserted-by":"crossref","unstructured":"Mock, F., Kretschmer, F., Kriese, A., B\u02d9ocker, S., Marz, M.: Taxonomic classification of DNA sequences beyond sequence similarity using deep neural networks. Proceedings of the National Academy of Sciences, 119(35), 2122636119 (2022).","DOI":"10.1073\/pnas.2122636119"},{"key":"10.1016\/j.procs.2025.09.288_bib54","doi-asserted-by":"crossref","unstructured":"Jahshan, Z., Yavits, L.: Vital: Vision transformer based low coverage sars-cov-2 lineage assignment. Bioinformatics, 093 (2024).","DOI":"10.1093\/bioinformatics\/btae093"},{"key":"10.1016\/j.procs.2025.09.288_bib55","doi-asserted-by":"crossref","unstructured":"Li, Y., Rao, S., Solares, J.R.A., Hassaine, A., Ramakrishnan, R., Canoy, D., Zhu, Y., Rahimi, K., Salimi-Khorshidi, G.: Behrt: transformer for electronic health records. Scientific reports 10(1), 7155 (2020).","DOI":"10.1038\/s41598-020-62922-y"},{"key":"10.1016\/j.procs.2025.09.288_bib56","doi-asserted-by":"crossref","unstructured":"Garg, M., Gajjar, P., Shah, P., Shukla, M., Acharya, B., Gerogiannis, V.C., Kanavos, A.: Comparative analysis of deep learning architectures and vision transformers for musical key estimation. Information 14(10), 527 (2023).","DOI":"10.3390\/info14100527"},{"key":"10.1016\/j.procs.2025.09.288_bib57","unstructured":"Van Rossum, G., Drake, F.L.: Python 3 Reference Manual. CreateSpace, Scotts Valley, CA (2009)."},{"key":"10.1016\/j.procs.2025.09.288_bib58","doi-asserted-by":"crossref","unstructured":"Tzanetakis, G., Cook, P.: Musical genre classification of audio signals. IEEE Transactions on speech and audio processing 10(5), 293\u2013302 (2002).","DOI":"10.1109\/TSA.2002.800560"},{"key":"10.1016\/j.procs.2025.09.288_bib59","unstructured":"Sturm, B.L.: The gtzan dataset: Its contents, its faults, their effects on evaluation, and its future use. arXiv preprint arXiv:1306.1461 (2013)."},{"key":"10.1016\/j.procs.2025.09.288_bib60","doi-asserted-by":"crossref","unstructured":"Seo, W., Cho, S.-H., Teisseyre, P., Lee, J.: A short survey and comparison of CNN- based music genre classification using multiple spectral features. IEEE Access (2023).","DOI":"10.1109\/ACCESS.2023.3346883"},{"key":"10.1016\/j.procs.2025.09.288_bib61","doi-asserted-by":"crossref","unstructured":"Tang, H., Chen, N.: Combining CNN and broad learning for music classification. IEICE TRANSACTIONS on Information and Systems, 103(3), 695\u2013701 (2020).","DOI":"10.1587\/transinf.2019EDP7175"},{"key":"10.1016\/j.procs.2025.09.288_bib62","unstructured":"Palanisamy, K., Singhania, D., Yao, A.: Rethinking CNN models for audio classification. arXiv preprint arXiv:2007.11154 (2020)."},{"key":"10.1016\/j.procs.2025.09.288_bib63","doi-asserted-by":"crossref","unstructured":"Greer, T., Shi, X., Ma, B., Narayanan, S.: Multi-modal, multi-task, music bert: A context-aware music encoder based on transformers (2022).","DOI":"10.21203\/rs.3.rs-2090671\/v1"},{"key":"10.1016\/j.procs.2025.09.288_bib64","doi-asserted-by":"crossref","unstructured":"McFee, B., Rafel, C., Liang, D., Ellis, D.P., McVicar, M., Battenberg, E., Nieto, O.: librosa: Audio and music signal analysis in python. In: SciPy, pp. 18\u201324 (2015).","DOI":"10.25080\/Majora-7b98e3ed-003"},{"key":"10.1016\/j.procs.2025.09.288_bib65","doi-asserted-by":"crossref","unstructured":"Kumar, N., Kaushal, R., Agarwal, S., Singh, Y.B.: CNN based approach for speech emotion recognition using mfcc, croma and stft handcrafted features. In: 2021 3rd International Conference on Advances in Computing, Communication Control and Networking (ICAC3N), pp. 981\u2013985, (2021).","DOI":"10.1109\/ICAC3N53548.2021.9725750"},{"key":"10.1016\/j.procs.2025.09.288_bib66","unstructured":"Paszke, A., Gross, S., Chintala, S., Chanan, G., Yang, E., DeVito, Z., Lin, Z., Desmaison, A., Antiga, L., Lerer, A.: Automatic differentiation in pytorch. In: NIPS-W (2017)."},{"key":"10.1016\/j.procs.2025.09.288_bib67","doi-asserted-by":"crossref","unstructured":"Lafrance, M., Hawley, L., et al.: Embodied subjectivities in the lyrical and musical expression of PJ Harvey and Bj\u02d9ork. Music Theory Online 14(4) (2008).","DOI":"10.30535\/mto.14.4.1"},{"key":"10.1016\/j.procs.2025.09.288_bib68","unstructured":"Radiohead: Pablo Honey. Capitol Records (1993)."},{"key":"10.1016\/j.procs.2025.09.288_bib69","unstructured":"Radiohead: Hail to the Thief. Capitol Records (2003)."},{"key":"10.1016\/j.procs.2025.09.288_bib70","doi-asserted-by":"crossref","unstructured":"McLeod, K.: Genres, subgenres, sub-subgenres and more: Musical and social differentiation within electronic\/dance music communities. Journal of popular music studies 13(1), 59\u201375 (2001).","DOI":"10.1111\/j.1533-1598.2001.tb00013.x"},{"key":"10.1016\/j.procs.2025.09.288_bib71","doi-asserted-by":"crossref","unstructured":"Lee, D., Hosanagar, K.: How do recommender systems affect sales diversity? a cross-category investigation via randomized field experiment. Information Systems Research 30(1), 239\u2013259 (2019).","DOI":"10.1287\/isre.2018.0800"},{"key":"10.1016\/j.procs.2025.09.288_bib72","doi-asserted-by":"crossref","unstructured":"Muchitsch, V., Werner, A.: The mediation of genre, identity, and difference in contemporary (popular) music streaming. Twentieth-Century Music, 1\u201327 (2024).","DOI":"10.1017\/S1478572223000270"},{"key":"10.1016\/j.procs.2025.09.288_bib73","unstructured":"Bergstra, J., Bardenet, R., Bengio, Y., K\u00b4egl, B.: Algorithms for hyperparameter optimization. Advances in neural information processing systems 24 (2011)."},{"key":"10.1016\/j.procs.2025.09.288_bib74","doi-asserted-by":"crossref","unstructured":"Wolpert, D.H.: The lack of a priori distinctions between learning algorithms. Neural computation 8(7), 1341\u20131390 (1996).","DOI":"10.1162\/neco.1996.8.7.1341"},{"key":"10.1016\/j.procs.2025.09.288_bib75","series-title":"Musicoder: A universal music-acoustic encoder based on trans- former. In: MultiMedia Modeling: 27th International Conference, MMM 2021, Prague, Czech Republic, June 22\u201324, 2021, Proceedings, Part I 27, pp. 417\u2013429","author":"Zhao","year":"2021"},{"key":"10.1016\/j.procs.2025.09.288_bib76","doi-asserted-by":"crossref","unstructured":"Zhao, H., Zhang, C., Zhu, B., Ma, Z., Zhang, K.: S3t: Self-supervised pre-training with swin transformer for music classification. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 606\u2013610 (2022).","DOI":"10.1109\/ICASSP43922.2022.9746056"},{"key":"10.1016\/j.procs.2025.09.288_bib77","doi-asserted-by":"crossref","unstructured":"Harryanto, A.A.A., Gunawan, K., Nagano, R., Sutoyo, R.: Music classification model development based on audio recognition using transformer model. In: 2022 3rd International Conference on Artificial Intelligence and Data Sciences (AiDAS), pp. 258\u2013263 (2022).","DOI":"10.1109\/AiDAS56890.2022.9918787"},{"key":"10.1016\/j.procs.2025.09.288_bib78","unstructured":"Bertin-Mahieux, T., Ellis, D.P., Whitman, B., Lamere, P.: The million song dataset (2011)."},{"key":"10.1016\/j.procs.2025.09.288_bib79","series-title":"Evaluation of algorithms using games: The case of music tagging. In: ISMIR, pp. 387\u2013392","author":"Law","year":"2009"},{"key":"10.1016\/j.procs.2025.09.288_bib80","doi-asserted-by":"crossref","unstructured":"Khasgiwala, Y., Tailor, J.: Vision transformer for music genre classification using mel-frequency cepstrum coefficient. In: 2021 IEEE 4th International Conference on Computing, Power and Communication Technologies (GUCON), pp. 1\u20135 (2021).","DOI":"10.1109\/GUCON50781.2021.9573568"},{"key":"10.1016\/j.procs.2025.09.288_bib81","doi-asserted-by":"crossref","unstructured":"Mahmood, K., Mahmood, R., Van Dijk, M.: On the robustness of vision transformers to adversarial examples. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7838\u20137847 (2021).","DOI":"10.1109\/ICCV48922.2021.00774"},{"key":"10.1016\/j.procs.2025.09.288_bib82","doi-asserted-by":"crossref","unstructured":"Huang, X., Kroening, D., Ruan, W., Sharp, J., Sun, Y., Thamo, E., Wu, M., Yi, X.: A survey of safety and trustworthiness of deep neural networks: Verification, testing, adversarial attack and defence, and interpretability. Computer Science Review 37, 100270 (2020).","DOI":"10.1016\/j.cosrev.2020.100270"},{"key":"10.1016\/j.procs.2025.09.288_bib83","doi-asserted-by":"crossref","unstructured":"Ebrahimi, J., Rao, A., Lowd, D., Dou, D.: Hotflip: White-box adversarial examples for text classification. arXiv preprint arXiv:1712.06751 (2017).","DOI":"10.18653\/v1\/P18-2006"},{"key":"10.1016\/j.procs.2025.09.288_bib84","doi-asserted-by":"crossref","unstructured":"Gil, Y., Chai, Y., Gorodissky, O., Berant, J.: White-to-black: Efficient distillation of black-box adversarial attacks. arXiv preprint arXiv:1904.02405 (2019).","DOI":"10.18653\/v1\/N19-1139"},{"key":"10.1016\/j.procs.2025.09.288_bib85","doi-asserted-by":"crossref","unstructured":"Papernot, N., McDaniel, P., Goodfellow, I., Jha, S., Celik, Z.B., Swami, A.: Practical black-box attacks against machine learning. In: Proceedings of the 2017 ACM on Asia Conference on Computer and Communications Security, pp. 506\u2013519 (2017).","DOI":"10.1145\/3052973.3053009"},{"key":"10.1016\/j.procs.2025.09.288_bib86","series-title":"Black-box adversarial attacks with limited queries and information. In: International Conference on Machine Learning, pp. 2137\u20132146","author":"Ilyas","year":"2018"},{"key":"10.1016\/j.procs.2025.09.288_bib87","unstructured":"Madry, A., Makelov, A., Schmidt, L., Tsipras, D., Vladu, A.: Towards deep learning models resistant to adversarial attacks. stat 1050, 9 (2017)."}],"container-title":["Procedia Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1877050925029618?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1877050925029618?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,12,21]],"date-time":"2025-12-21T08:31:58Z","timestamp":1766305918000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1877050925029618"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":87,"alternative-id":["S1877050925029618"],"URL":"https:\/\/doi.org\/10.1016\/j.procs.2025.09.288","relation":{},"ISSN":["1877-0509"],"issn-type":[{"value":"1877-0509","type":"print"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Discerning Music Genres: Exploring Neural Network Architectures for Automated Classification","name":"articletitle","label":"Article Title"},{"value":"Procedia Computer Science","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.procs.2025.09.288","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2025 The Author(s). Published by Elsevier B.V.","name":"copyright","label":"Copyright"}]}}