{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T06:12:17Z","timestamp":1780380737849,"version":"3.54.1"},"reference-count":88,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2019,3,4]],"date-time":"2019-03-04T00:00:00Z","timestamp":1551657600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2020,2]]},"DOI":"10.1007\/s00521-019-04076-1","type":"journal-article","created":{"date-parts":[[2019,3,4]],"date-time":"2019-03-04T13:33:26Z","timestamp":1551706406000},"page":"1067-1093","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":33,"title":["One deep music representation to rule them all? A comparative analysis of different representation learning strategies"],"prefix":"10.1007","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5744-9034","authenticated-orcid":false,"given":"Jaehun","family":"Kim","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Juli\u00e1n","family":"Urbano","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cynthia C. S.","family":"Liem","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alan","family":"Hanjalic","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,3,4]]},"reference":[{"issue":"4","key":"4076_CR1","doi-asserted-by":"publisher","first-page":"668","DOI":"10.1109\/JPROC.2008.916370","volume":"96","author":"MA Casey","year":"2008","unstructured":"Casey MA, Veltkamp RC, Goto M, Leman M, Rhodes C, Slaney M (2008) Content-based music information retrieval: current directions and future challenges. Proc IEEE 96(4):668\u2013696. https:\/\/doi.org\/10.1109\/JPROC.2008.916370","journal-title":"Proc IEEE"},{"issue":"1","key":"4076_CR2","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1023\/A:1007379606734","volume":"28","author":"R Caruana","year":"1997","unstructured":"Caruana R (1997) Multitask learning. Mach Learn 28(1):41\u201375. https:\/\/doi.org\/10.1023\/A:1007379606734 . ISSN: 1573-0565","journal-title":"Mach Learn"},{"issue":"8","key":"4076_CR3","doi-asserted-by":"publisher","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","volume":"35","author":"Y Bengio","year":"2013","unstructured":"Bengio Y, Courville AC, Vincent P (2013) Representation learning: a review and new perspectives. IEEE Trans Pattern Anal Mach Intell 35(8):1798\u20131828. https:\/\/doi.org\/10.1109\/TPAMI.2013.50 . ISSN: 0162-8828","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"4076_CR4","doi-asserted-by":"publisher","unstructured":"Liu W, Mei T, Zhang Y, Che C, Luo J (2015) Multi-task deep visual-semantic embedding for video thumbnail selection. In: IEEE conference on computer vision and pattern recognition CVPR, Boston, MA, USA, pp 3707\u20133715. https:\/\/doi.org\/10.1109\/CVPR.2015.7298994","DOI":"10.1109\/CVPR.2015.7298994"},{"key":"4076_CR5","doi-asserted-by":"crossref","unstructured":"Bingel J, S\u00f8gaard A (2017) Identifying beneficial task relations for multi-task learning in deep neural networks. In: Proceedings of the 15th conference of the European chapter of the association for computational linguistics, vol 2. Association for Computational Linguistics, Valencia, Spain, pp 164\u2013169","DOI":"10.18653\/v1\/E17-2026"},{"issue":"1","key":"4076_CR6","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1007\/s11263-014-0767-8","volume":"113","author":"S Li","year":"2015","unstructured":"Li S, Liu Z-Q, Chan AB (2015) Heterogeneous multi-task learning for human pose estimation with deep convolutional neural network. Int J Comput Vis 113(1):19\u201336. https:\/\/doi.org\/10.1007\/s11263-014-0767-8 . ISSN: 1573-1405","journal-title":"Int J Comput Vis"},{"key":"4076_CR7","doi-asserted-by":"publisher","unstructured":"Zhang W, Li R, Zeng T, Sun Q, Kumar S, Ye J, Ji S (2015) Deep model based transfer and multi-task learning for biological image analysis. In: Proceedings of the 21th ACM SIGKDD international conference on knowledge discovery and data mining KDD, Sydney. ACM, NSW, Australia, pp 1475\u20131484. https:\/\/doi.org\/10.1145\/2783258.2783304 . ISBN: 978-1-4503-3664-2","DOI":"10.1145\/2783258.2783304"},{"key":"4076_CR8","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1007\/978-3-319-10599-4_7","volume-title":"Computer Vision \u2013 ECCV 2014","author":"Zhanpeng Zhang","year":"2014","unstructured":"Zhang Z, Luo Z, Loy CC, Tang X (2014) Facial landmark detection by deep multi-task learning. In: Computer vision\u2014ECCV 13th European conference, proceedings, part VI. Springer, Zurich, Switzerland, pp 94\u2013108. https:\/\/doi.org\/10.1007\/978-3-319-10599-4_7"},{"key":"4076_CR9","unstructured":"Kaiser L, Gomez AN, Shazeer N, Vaswani A, Parmar N, Jones L, Uszkoreit J (2017) One model to learn them all. arXiv:abs\/1706.05137"},{"key":"4076_CR10","doi-asserted-by":"publisher","unstructured":"Rick Chang J-H, Li C-L, P\u00f3czos B, Vijaya Kumar BVK (2017) One network to solve them all\u2014solving linear inverse problems using deep projection models. In: IEEE international conference on computer vision, ICCV. IEEE Computer Society, Venice, Italy, pp 5889\u20135898. https:\/\/doi.org\/10.1109\/ICCV.2017.627","DOI":"10.1109\/ICCV.2017.627"},{"issue":"4","key":"4076_CR11","doi-asserted-by":"publisher","first-page":"337","DOI":"10.1080\/09298215.2011.603834","volume":"40","author":"J Weston","year":"2011","unstructured":"Weston J, Bengio S, Hamel P (2011) Multi-tasking with joint semantic spaces for large-scale music annotation and retrieval. J New Music Res 40(4):337\u2013348. https:\/\/doi.org\/10.1080\/09298215.2011.603834","journal-title":"J New Music Res"},{"key":"4076_CR12","unstructured":"Aytar Y, Vondrick C, Torralba A (2016) Soundnet: Learning sound representations from unlabeled video. In: Advances in neural information processing systems 29: annual conference on neural information processing systems. Barcelona, Spain, pp 892\u2013900"},{"key":"4076_CR13","unstructured":"Hamel P, Eck D (2010) Learning features from music audio with deep belief networks. In: Proceedings of the 11th international society for music information retrieval conference, ISMIR. Utrecht, Netherlands, pp 339\u2013344"},{"key":"4076_CR14","doi-asserted-by":"crossref","unstructured":"Boulanger-Lewandowski N, Bengio Y, Vincent P (2012) Modeling temporal dependencies in high-dimensional sequences: application to polyphonic music generation and transcription. In: Proceedings of the 29th international conference on machine learning, ICML. Omnipress, Edinburgh, Scotland, UK","DOI":"10.1109\/ICASSP.2013.6638244"},{"key":"4076_CR15","doi-asserted-by":"publisher","unstructured":"Schl\u00fcter J, B\u00f6ck S (2014) Improved musical onset detection with convolutional neural networks. In: IEEE international conference on acoustics, speech and signal processing, ICASSP. IEEE, Florence, Italy, pp 6979\u20136983. https:\/\/doi.org\/10.1109\/ICASSP.2014.6854953","DOI":"10.1109\/ICASSP.2014.6854953"},{"key":"4076_CR16","unstructured":"Choi K, Fazekas G, Sandler MB (2016) Automatic tagging using deep convolutional neural networks. In: Proceedings of the 17th international society for music information retrieval conference, ISMIR. New York City, USA, pp 805\u2013811"},{"key":"4076_CR17","unstructured":"van den Oord A, Dieleman S, Schrauwen B (2013) Deep content-based music recommendation. In: Advances in neural information processing systems 26 NIPS. Lake Tahoe, NV, USA, pp 2643\u20132651"},{"key":"4076_CR18","doi-asserted-by":"publisher","first-page":"258","DOI":"10.1007\/978-3-319-53547-0_25","volume-title":"Latent Variable Analysis and Signal Separation","author":"Pritish Chandna","year":"2017","unstructured":"Chandna P, Miron M, Janer J, G\u00f3mez E (2017) Monoaural audio source separation using deep convolutional neural networks. In: Latent variable analysis and signal separation\u201413th international conference, LVA\/ICA, Proceedings. Grenoble, France, pp 258\u2013266. https:\/\/doi.org\/10.1007\/978-3-319-53547-0_25 . ISBN: 978-3-319-53547-0"},{"key":"4076_CR19","unstructured":"Jeong I-Y, Lee K (2016) Learning temporal features using a deep neural network and its application to music genre classification. In: Proceedings of the 17th international society for music information retrieval conference, ISMIR. New York City, USA, pp 434\u2013440"},{"issue":"1","key":"4076_CR20","doi-asserted-by":"publisher","first-page":"208","DOI":"10.1109\/TASLP.2016.2632307","volume":"25","author":"Y Han","year":"2017","unstructured":"Han Y, Kim J-H, Lee K (2017) Deep convolutional neural networks for predominant instrument recognition in polyphonic music. IEEE\/ACM Trans Audio Speech Lang Process 25(1):208\u2013221. https:\/\/doi.org\/10.1109\/TASLP.2016.2632307 . ISSN: 2329-9290","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"4076_CR21","unstructured":"Simonyan K, Zisserman A (2015) Very deep convolutional networks for large-scale image recognition. In: 3th international conference on learning representations, ICLR, San Diego, CA, USA"},{"key":"4076_CR22","doi-asserted-by":"publisher","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: IEEE conference on computer vision and pattern recognition, CVPR. IEEE Computer Society, Las Vegas, NV, USA, pp 770\u2013778. https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"4076_CR23","doi-asserted-by":"publisher","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed SE, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: IEEE conference on computer vision and pattern recognition, CVPR. IEEE Computer Society, Boston, MA, USA, pp 1\u20139. https:\/\/doi.org\/10.1109\/CVPR.2015.7298594","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"4076_CR24","unstructured":"Mikolov T, Sutskever I, Chen K, Corrado GS, Dean J (2013) Distributed representations of words and phrases and their compositionality. In: Advances in neural information processing systems 26 NIPS. Lake Tahoe, NV, USA, pp 3111\u20133119"},{"key":"4076_CR25","unstructured":"Dieleman S, Brakel P, Schrauwen B (2011) Audio-based music classification with a pretrained convolutional network. In: Proceedings of the 12th international society for music information retrieval conference, ISMIR. University of Miami, Miami, FL, USA. pp 669\u2013674. ISBN: 9780615548654"},{"key":"4076_CR26","unstructured":"Choi K, Fazekas G, Sandler MB, Cho K (2017) Transfer learning for music classification and regression tasks. In: Proceedings of the 18th international society for music information retrieval conference, ISMIR. Suzhou, China, pp 141\u2013149"},{"key":"4076_CR27","unstructured":"van\u00a0den Oord A, Dieleman S, Schrauwen B (2014) Transfer learning by supervised pre-training for audio-based music classification. In: Proceedings of the 15th international society for music information retrieval conference, ISMIR. Taipei, Taiwan, pp 29\u201334"},{"key":"4076_CR28","unstructured":"Liang D, Zhan M, Ellis DPW (2015) Content-aware collaborative music recommendation using pre-trained neural networks. In: Proceedings of the 16th international society for music information retrieval conference, ISMIR. M\u00e1laga, Spain, pp 295\u2013301"},{"key":"4076_CR29","doi-asserted-by":"crossref","unstructured":"Misra I, Shrivastava A, Gupta A, Hebert M (2016) Cross-stitch networks for multi-task learning. In: IEEE conference on computer vision and pattern recognition. CVPR. IEEE Computer Society, Las Vegas, NV, USA, pp 3994\u20134003","DOI":"10.1109\/CVPR.2016.433"},{"key":"4076_CR30","unstructured":"Bertin-Mahieux T, Ellis DPW, Whitman B, Lamere P (2011) The million song dataset. In: Proceedings of the 12th international society for music information retrieval conference, ISMIR. University of Miami, Miami, FL, USA. pp 591\u2013596"},{"key":"4076_CR31","unstructured":"Bengio Y, Lamblin P, Popovici D, Larochelle H (2006) Greedy layer-wise training of deep networks. In: Advances in neural information processing systems 19. NIPS. MIT Press, Vancouver, BC, Canada, pp 153\u2013160"},{"key":"4076_CR32","doi-asserted-by":"publisher","unstructured":"Vincent P, Larochelle H, Bengio Y, Manzagol P-A (2008) Extracting and composing robust features with denoising autoencoders. In: Proceedings of the 25th international conference on machine learning ICML. ACM, Helsinki, Finland, pp 1096\u20131103. https:\/\/doi.org\/10.1145\/1390156.1390294","DOI":"10.1145\/1390156.1390294"},{"key":"4076_CR33","unstructured":"Smolensky P (1986) Information processing in dynamical systems: Foundations of harmony theory. Technical report, University of Colorado, Boulder, Department of Computer Science"},{"issue":"7","key":"4076_CR34","doi-asserted-by":"publisher","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"GE Hinton","year":"2006","unstructured":"Hinton GE, Osindero S, Teh Y-W (2006) A fast learning algorithm for deep belief nets. Neural Comput 18(7):1527\u20131554. https:\/\/doi.org\/10.1162\/neco.2006.18.7.1527","journal-title":"Neural Comput"},{"key":"4076_CR35","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Bing X, Warde-Farley D, Ozair S, Courville A, Bengio Y (2014) Generative adversarial nets. In: Advances in neural information processing systems 27. NIPS. Curran Associates Inc., Montreal, QC, Canada, pp 2672\u20132680"},{"key":"4076_CR36","doi-asserted-by":"publisher","unstructured":"Han X, Leung T, Jia Y, Sukthankar R, Berg AC (2015) Matchnet: unifying feature and metric learning for patch-based matching. In: IEEE conference on computer vision and pattern recognition, CVPR. IEEE Computer Society, Boston, MA, USA, pp 3279\u20133286. https:\/\/doi.org\/10.1109\/CVPR.2015.7298948","DOI":"10.1109\/CVPR.2015.7298948"},{"key":"4076_CR37","doi-asserted-by":"publisher","unstructured":"Arandjelovic R, Zisserman A (2017) Look, listen and learn. In: IEEE international conference on computer vision, ICCV. IEEE Computer Society, Venice, Italy, pp 609\u2013617. https:\/\/doi.org\/10.1109\/ICCV.2017.73","DOI":"10.1109\/ICCV.2017.73"},{"key":"4076_CR38","unstructured":"Huang Y-S, Chou S-Y, Yang Y-H (2018) Generating music medleys via playing music puzzle games. In: Proceedings of the thirty-second conference on artificial intelligence, AAAI. AAAI Press, New Orleans, LA, USA, pp 2281\u20132288"},{"key":"4076_CR39","volume-title":"Introduction to modern information retrieval","author":"G Salton","year":"1984","unstructured":"Salton G, McGill M (1984) Introduction to modern information retrieval. McGraw-Hill Book Company, New York City. ISBN: 0-07-054484-0"},{"issue":"2","key":"4076_CR40","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1080\/09298210802479284","volume":"37","author":"P Lamere","year":"2008","unstructured":"Lamere P (2008) Social tagging and music information retrieval. J New Music Res 37(2):101\u2013114. https:\/\/doi.org\/10.1080\/09298210802479284 . ISSN: 0929-8215","journal-title":"J New Music Res"},{"key":"4076_CR41","unstructured":"Hamel P, Davies MEP, Yoshii K, Goto M (2013) Transfer learning in MIR: sharing learned latent representations for music audio classification and similarity. In: Proceedings of the 14th international society for music information retrieval conference, ISMIR. Curitiba, Brazil, pp 9\u201314"},{"key":"4076_CR42","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/978-3-642-15883-4_14","volume-title":"Machine Learning and Knowledge Discovery in Databases","author":"Edith Law","year":"2010","unstructured":"Law E, Settles B, Mitchell TM (2010) Learning to tag from open vocabulary labels. In: Machine learning and knowledge discovery in databases, European conference, ECML PKDD, Proceedings. Part II. Springer, Barcelona, Spain, pp 211\u2013226"},{"key":"4076_CR43","unstructured":"Hofmann T (1999) Probabilistic latent semantic analysis. In: UAI: proceedings of the fifteenth conference on uncertainty in artificial intelligence. Morgan Kaufmann, Stockholm, Sweden, pp 289\u2013296"},{"key":"4076_CR44","unstructured":"Schl\u00fcter J (2016) Learning to pinpoint singing voice from weakly labeled examples. In: Proceedings of the 17th international society for music information retrieval conference, ISMIR. New York City, USA, pp 44\u201350"},{"key":"4076_CR45","doi-asserted-by":"publisher","unstructured":"Hershey S, Chaudhuri S, Ellis DPW, Gemmeke JF, Jansen A, Moore RC, Plakal M, Platt D, Saurous RA, Seybold B, Slaney M, Weiss RJ, Wilson KW (2017) CNN architectures for large-scale audio classification. In: IEEE international conference on acoustics, speech and signal processing, ICASSP. IEEE, New Orleans, LA, USA, pp 131\u2013135. https:\/\/doi.org\/10.1109\/ICASSP.2017.7952132","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"4076_CR46","unstructured":"Lee H, Pham PT, Largman Y, Ng AY (2009) Unsupervised feature learning for audio classification using convolutional deep belief networks. In: Advances in neural information processing systems 22. NIPS. Curran Associates Inc, Vancouver, BC, Canada, pp 1096\u20131104"},{"key":"4076_CR47","doi-asserted-by":"publisher","unstructured":"Humphrey EJ, Bello JP (2012) Rethinking automatic chord recognition with convolutional neural networks. In: 11th international conference on machine learning and applications, ICMLA. IEEE, Boca Raton, FL, USA, pp 357\u2013362. https:\/\/doi.org\/10.1109\/ICMLA.2012.220","DOI":"10.1109\/ICMLA.2012.220"},{"key":"4076_CR48","doi-asserted-by":"crossref","unstructured":"Nakashika T, Garcia C, Takiguchi T (2012) Local-feature-map integration using convolutional neural networks for music genre classification. In: INTERSPEECH, 13th annual conference of the international speech communication association. ISCA, Portland, OR, USA, pp 1752\u20131755","DOI":"10.21437\/Interspeech.2012-478"},{"key":"4076_CR49","unstructured":"Ullrich K, Schl\u00fcter J, Grill T (2015) Boundary detection in music structure analysis using convolutional neural networks. In: Proceedings of the 16th international society for music information retrieval conference, ISMIR. M\u00e1laga, Spain, pp 417\u2013422"},{"key":"4076_CR50","doi-asserted-by":"publisher","unstructured":"Piczak KJ (2015) Environmental sound classification with convolutional neural networks. In: 25th IEEE international workshop on machine learning for signal processing, MLSP. IEEE, Boston, MA, USA, pp 1\u20136. https:\/\/doi.org\/10.1109\/MLSP.2015.7324337","DOI":"10.1109\/MLSP.2015.7324337"},{"key":"4076_CR51","doi-asserted-by":"publisher","first-page":"429","DOI":"10.1007\/978-3-319-22482-4_50","volume-title":"Latent Variable Analysis and Signal Separation","author":"Andrew J. R. Simpson","year":"2015","unstructured":"Simpson AJR, Roma G, Plumbley MD (2015) Deep karaoke: extracting vocals from musical mixtures using a convolutional deep neural network. In: Latent variable analysis and signal separation\u201412th international conference, LVA\/ICA, Proceedings. Springer, Liberec, Czech Republic, pp 429\u2013436. https:\/\/doi.org\/10.1007\/978-3-319-22482-4_50 . ISBN: 978-3-319-22482-4"},{"key":"4076_CR52","doi-asserted-by":"publisher","unstructured":"Phan H, Hertel L, Maa\u00df M, Mertins A (2016) Robust audio event recognition with 1-max pooling convolutional neural networks. In: INTERSPEECH 17th annual conference of the international speech communication association. ISCA, San Francisco, CA, USA, pp 3653\u20133657. https:\/\/doi.org\/10.21437\/Interspeech.2016-123","DOI":"10.21437\/Interspeech.2016-123"},{"key":"4076_CR53","doi-asserted-by":"publisher","unstructured":"Pons J, Lidy T, Serra X (2016) Experimenting with musically motivated convolutional neural networks. In: 14th international workshop on content-based multimedia indexing, CBMI. IEEE, Bucharest, Romania, pp 1\u20136. https:\/\/doi.org\/10.1109\/CBMI.2016.7500246","DOI":"10.1109\/CBMI.2016.7500246"},{"key":"4076_CR54","doi-asserted-by":"publisher","unstructured":"Stasiak B, Monko J (2016) Analysis of time-frequency representations for musical onset detection with convolutional neural network. In: Proceedings of the federated conference on computer science and information systems, FedCSIS. Gda\u0144sk, Poland, pp 147\u2013152. https:\/\/doi.org\/10.15439\/2016F558","DOI":"10.15439\/2016F558"},{"key":"4076_CR55","doi-asserted-by":"publisher","unstructured":"Su H, Zhang H, Zhang X, Gao G (2016) Convolutional neural network for robust pitch determination. In: IEEE international conference on acoustics, speech and signal processing, ICASSP. IEEE, Shanghai, China. pp 579\u2013583. https:\/\/doi.org\/10.1109\/ICASSP.2016.7471741","DOI":"10.1109\/ICASSP.2016.7471741"},{"issue":"6","key":"4076_CR56","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1145\/3065386","volume":"60","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2017) Imagenet classification with deep convolutional neural networks. Commun ACM 60(6):84\u201390. https:\/\/doi.org\/10.1145\/3065386","journal-title":"Commun ACM"},{"key":"4076_CR57","doi-asserted-by":"publisher","unstructured":"Dieleman S, Schrauwen B (2014) End-to-end learning for music audio. In: IEEE international conference on acoustics, speech and signal processing, ICASSP. IEEE, Florence, Italy, pp 6964\u20136968. https:\/\/doi.org\/10.1109\/ICASSP.2014.6854950","DOI":"10.1109\/ICASSP.2014.6854950"},{"key":"4076_CR58","unstructured":"van\u00a0den Oord A, Dieleman S, Zen H, Simonyan K, Vinyals O, Graves A, Kalchbrenner N, Senior AW, Kavukcuoglu K (2016) Wavenet: a generative model for raw audio. In: The 9th ISCA speech synthesis workshop, SSW. ISCA, Sunnyvale, CA, USA, p 125"},{"key":"4076_CR59","doi-asserted-by":"publisher","unstructured":"Jaitly N, Hinton GE (2011) Learning a better representation of speech soundwaves using restricted boltzmann machines. In: IEEE international conference on acoustics, speech, and signal processing, ICASSP. IEEE, Prague, Czech Republic, pp 5884\u20135887. https:\/\/doi.org\/10.1109\/ICASSP.2011.5947700","DOI":"10.1109\/ICASSP.2011.5947700"},{"key":"4076_CR60","unstructured":"Lee J, Park J, Kim KL, Nam J (2017) Sample-level deep convolutional neural networks for music auto-tagging using raw waveforms. In: 14th sound and music computing conference, SMC, Espoo, Finland"},{"key":"4076_CR61","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: accelerating deep network training by reducing internal covariate shift. In: Proceedings of the 32nd international conference on machine learning, ICML. JMLR, Inc, Lille, France, pp 448\u2013456"},{"key":"4076_CR62","unstructured":"Nair V, Hinton GE (2010) Rectified linear units improve restricted boltzmann machines. In: Proceedings of the 27th international conference on machine learning ICML. Omnipress, Haifa, Israel, pp 807\u2013814"},{"issue":"1","key":"4076_CR63","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava N, Hinton GE, Krizhevsky A, Sutskever I, Salakhutdinov R (2014) Dropout: a simple way to prevent neural networks from overfitting. J Mach Learn Res 15(1):1929\u20131958","journal-title":"J Mach Learn Res"},{"key":"4076_CR64","unstructured":"Nam J, Herrera J, Slaney M, Smith JO (2012) Learning sparse feature representations for music annotation and retrieval. In: Proceedings of the 13th international society for music information retrieval conference, ISMIR. FEUP Edi\u00e7\u00f5es, Porto, Portugal, pp 565\u2013570"},{"key":"4076_CR65","doi-asserted-by":"crossref","unstructured":"Choi K, Fazekas G, Sandler MB, Cho K (2018) A comparison of audio signal preprocessing methods for deep neural networks on music tagging. In: 26th European signal processing conference. EUSIPCO. IEEE, Roma, Italy, pp 1870\u20131874","DOI":"10.23919\/EUSIPCO.2018.8553106"},{"key":"4076_CR66","doi-asserted-by":"publisher","unstructured":"D\u00f6rfler M, Grill T, Bammer R, Flexer A (2018) Basic filters for convolutional neural networks applied to music: training or design? Neural Comput Appl https:\/\/doi.org\/10.1007\/s00521-018-3704-x . ISSN: 1433-3058","DOI":"10.1007\/s00521-018-3704-x"},{"key":"4076_CR67","unstructured":"Kingma DP, Ba J (2015) Adam: a method for stochastic optimization. In: 3th International conference on learning representations, ICLR, San Diego, CA, USA"},{"key":"4076_CR68","unstructured":"Paszke A, Gross S, Chintala S, Chanan G, Yang E, DeVito Z, Lin Z, Desmaison A, Antiga L, Lerer A (2017) Automatic differentiation in PyTorch. In: NIPS-W"},{"key":"4076_CR69","doi-asserted-by":"publisher","first-page":"2825","DOI":"10.1007\/s13398-014-0173-7.2","volume":"12","author":"F Pedregosa","year":"2012","unstructured":"Pedregosa F, Varoquaux G, Gramfort A, Michel V, Thirion B, Grisel O, Blondel M, Prettenhofer P, Weiss R, Dubourg V, Vanderplas J, Passos A, Cournapeau D, Brucher M, Perrot M, Duchesnay D (2012) Scikit-learn: machine learning in python. J Mach Learn Res 12:2825\u20132830. https:\/\/doi.org\/10.1007\/s13398-014-0173-7.2 . ISSN: 15324435","journal-title":"J Mach Learn Res"},{"key":"4076_CR70","doi-asserted-by":"publisher","unstructured":"McFee B, Raffel C, Liang D, Ellis DPW, McVicar M, Battenberg M, Nieto O (2015) librosa: audio and music signal analysis in python. In: Kathryn H, James B (eds) Proceedings of the 14th python in science conference SciPy. Austin, TX, USA, pp 18 \u2013 24. https:\/\/doi.org\/10.25080\/Majora-7b98e3ed-003","DOI":"10.25080\/Majora-7b98e3ed-003"},{"key":"4076_CR71","unstructured":"Defferrard M, Benzi K, Vandergheynst P, Bresson X (2017) FMA: a dataset for music analysis. In: Proceedings of the 18th international society for music information retrieval conference, ISMIR. Suzhou, China, pp 316\u2013323"},{"issue":"5","key":"4076_CR72","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1109\/TSA.2002.800560","volume":"10","author":"G Tzanetakis","year":"2002","unstructured":"Tzanetakis G, Cook PR (2002) Musical genre classification of audio signals. IEEE Trans Speech Audio Process 10(5):293\u2013302. https:\/\/doi.org\/10.1109\/TSA.2002.800560 . ISSN: 1063-6676","journal-title":"IEEE Trans Speech Audio Process"},{"issue":"11","key":"4076_CR73","doi-asserted-by":"publisher","first-page":"2059","DOI":"10.1109\/TMM.2015.2478068","volume":"17","author":"C Kereliuk","year":"2015","unstructured":"Kereliuk C, Sturm BL, Larsen J (2015) Deep learning and music adversaries. IEEE Trans Multimed 17(11):2059\u20132071. https:\/\/doi.org\/10.1109\/TMM.2015.2478068 . ISSN: 1520-9210","journal-title":"IEEE Trans Multimed"},{"issue":"5","key":"4076_CR74","doi-asserted-by":"publisher","first-page":"1832","DOI":"10.1109\/TSA.2005.858509","volume":"14","author":"G Fabien","year":"2006","unstructured":"Fabien G, Anssi K, Simon D, Alonso M, George T, Uhle C, Pedro C (2006) An experimental comparison of audio tempo induction algorithms. IEEE Trans Audio Speech Lang Process 14(5):1832\u20131844. https:\/\/doi.org\/10.1109\/TSA.2005.858509 . ISSN: 1558-7916","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"4076_CR75","doi-asserted-by":"publisher","unstructured":"Marchand U, Peeters G (2016) Scale and shift invariant time\/frequency representation using auditory statistics: application to rhythm description. In: 26th IEEE international workshop on machine learning for signal processing, MLSP. IEEE, Salerno, Italy, pp 1\u20136. https:\/\/doi.org\/10.1109\/MLSP.2016.7738904","DOI":"10.1109\/MLSP.2016.7738904"},{"key":"4076_CR76","unstructured":"Bosch JJ, Janer J, Fuhrmann F, Herrera P (2012) A comparison of sound segregation techniques for predominant instrument recognition in musical audio signals. In: Proceedings of the 13th international society for music information retrieval conference, ISMIR. FEUP Edi\u00e7\u00f5es, Porto, Portugal, pp 559\u2013564"},{"key":"4076_CR77","doi-asserted-by":"publisher","unstructured":"Soleymani M, Caro MN, Schmidt EM, Sha C-Y, Yang Y-H (2013) 1000 songs for emotional analysis of music. In: Proceedings of the 2nd ACM international workshop on crowdsourcing for multimedia CrowdMM@ACM multimedia. ACM, Barcelona, Spain, pp 1\u20136. https:\/\/doi.org\/10.1145\/2506364.2506365 . ISBN: 978-1-4503-2396-3","DOI":"10.1145\/2506364.2506365"},{"key":"4076_CR78","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-13287-2","volume-title":"Music recommendation and discovery\u2013the long tail, long fail, and long play in the digital music space","author":"C \u00d2scar","year":"2010","unstructured":"\u00d2scar C (2010) Music recommendation and discovery\u2013the long tail, long fail, and long play in the digital music space. Springer, Berlin. https:\/\/doi.org\/10.1007\/978-3-642-13287-2 . ISBN: 978-3-642-13286-5"},{"issue":"2","key":"4076_CR79","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1080\/09298215.2014.894533","volume":"43","author":"BL Sturm","year":"2014","unstructured":"Sturm BL (2014) The state of the art ten years after a state of the art: future research in music information retrieval. J New Music Res 43(2):147\u2013172. https:\/\/doi.org\/10.1080\/09298215.2014.894533","journal-title":"J New Music Res"},{"issue":"2","key":"4076_CR80","doi-asserted-by":"publisher","first-page":"3:1","DOI":"10.1145\/2967507","volume":"14","author":"BL Sturm","year":"2016","unstructured":"Sturm BL (2016) The \u201cHorse\u201d inside: seeking causes behind the behaviors of music content analysis systems. Comput Entertain 14(2):3:1\u20133:32. https:\/\/doi.org\/10.1145\/2967507","journal-title":"Comput Entertain"},{"issue":"3","key":"4076_CR81","doi-asserted-by":"publisher","first-page":"715734","DOI":"10.1017\/S0954579405050340","volume":"17","author":"P Jonathan","year":"2005","unstructured":"Jonathan P, Russell James A, Peterson Bradley S (2005) The circumplex model of affect: an integrative approach to affective neuroscience, cognitive development, and psychopathology. Dev Psychopathol 17(3):715734. https:\/\/doi.org\/10.1017\/S0954579405050340 . ISSN: 1469-2198","journal-title":"Dev Psychopathol"},{"key":"4076_CR82","volume-title":"Design and analysis of experiments","author":"DC Montgomery","year":"2012","unstructured":"Montgomery DC (2012) Design and analysis of experiments, 8th edn. Wiley, Hoboken","edition":"8"},{"key":"4076_CR83","doi-asserted-by":"publisher","DOI":"10.1002\/9781119974017","volume-title":"Optimal design of experiments: a case study approach","author":"P Goos","year":"2011","unstructured":"Goos P, Jones B (2011) Optimal design of experiments: a case study approach, 1st edn. Wiley, Hoboken","edition":"1"},{"issue":"1","key":"4076_CR84","doi-asserted-by":"publisher","first-page":"185","DOI":"10.1016\/0004-3702(89)90049-0","volume":"40","author":"GE Hinton","year":"1989","unstructured":"Hinton GE (1989) Connectionist learning procedures. Artif Intell 40(1):185\u2013234. https:\/\/doi.org\/10.1016\/0004-3702(89)90049-0 . ISSN: 0004-3702","journal-title":"Artif Intell"},{"key":"4076_CR85","doi-asserted-by":"publisher","unstructured":"Hu Y, Koren Y, Volinsky C (2008) Collaborative filtering for implicit feedback datasets. In: Proceedings of the 8th IEEE international conference on data mining (ICDM). IEEE Computer Society, Pisa, Italy, pp 263\u2013272. https:\/\/doi.org\/10.1109\/ICDM.2008.22","DOI":"10.1109\/ICDM.2008.22"},{"key":"4076_CR86","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511790942","volume-title":"Data analysis using regression and multilevel\/hierarchical models","author":"A Gelman","year":"2006","unstructured":"Gelman A, Hill J (2006) Data analysis using regression and multilevel\/hierarchical models. Cambridge University Press, Cambridge"},{"key":"4076_CR87","volume-title":"Variance components","author":"SR Searle","year":"2006","unstructured":"Searle SR, Casella G, McCulloch CE (2006) Variance components. Wiley, Hoboken"},{"issue":"November","key":"4076_CR88","first-page":"2579","volume":"9","author":"L Maaten van der","year":"2008","unstructured":"van der Maaten L, Hinton G (2008) Visualizing data using t-SNE. J Mach Learn Res 9(November):2579\u20132605","journal-title":"J Mach Learn Res"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-019-04076-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00521-019-04076-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-019-04076-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,12]],"date-time":"2022-09-12T23:42:55Z","timestamp":1663026175000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00521-019-04076-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,3,4]]},"references-count":88,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2020,2]]}},"alternative-id":["4076"],"URL":"https:\/\/doi.org\/10.1007\/s00521-019-04076-1","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,3,4]]},"assertion":[{"value":"7 December 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 February 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 March 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with ethical standards"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}