{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T04:30:23Z","timestamp":1784521823897,"version":"3.55.0"},"reference-count":47,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key R&amp;D Program of China","award":["2022YFF0901800"],"award-info":[{"award-number":["2022YFF0901800"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372365"],"award-info":[{"award-number":["62372365"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62302383"],"award-info":[{"award-number":["62302383"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62176205"],"award-info":[{"award-number":["62176205"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072367"],"award-info":[{"award-number":["62072367"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"China\/Shaanxi Postdoctoral Science Foundation","award":["2023M742792\/2023BSHYDZZ21"],"award-info":[{"award-number":["2023M742792\/2023BSHYDZZ21"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/taslp.2024.3402115","type":"journal-article","created":{"date-parts":[[2024,5,16]],"date-time":"2024-05-16T17:33:01Z","timestamp":1715880781000},"page":"2764-2776","source":"Crossref","is-referenced-by-count":10,"title":["Genre Classification Empowered by Knowledge-Embedded Music Representation"],"prefix":"10.1109","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5274-7988","authenticated-orcid":false,"given":"Han","family":"Ding","sequence":"first","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5996-057X","authenticated-orcid":false,"given":"Linwei","family":"Zhai","sequence":"additional","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4603-4914","authenticated-orcid":false,"given":"Cui","family":"Zhao","sequence":"additional","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0750-6990","authenticated-orcid":false,"given":"Fei","family":"Wang","sequence":"additional","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3845-1646","authenticated-orcid":false,"given":"Ge","family":"Wang","sequence":"additional","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9348-2982","authenticated-orcid":false,"given":"Wei","family":"Xi","sequence":"additional","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1389-0068","authenticated-orcid":false,"given":"Zhi","family":"Wang","sequence":"additional","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1037-9619","authenticated-orcid":false,"given":"Jizhong","family":"Zhao","sequence":"additional","affiliation":[{"name":"Faculty of Electronic and Information Engineering, Xi&#x0027;an Jiaotong University, Xi&#x0027;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2006.1598089"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3184558.3191822"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952585"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CBMI.2016.7500246"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746996"},{"key":"ref6","article-title":"A novel multimodal music genre classifier using hierarchical attention and convolutional neural network","author":"Agrawal","year":"2020"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10097131"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/93.556537"},{"key":"ref9","first-page":"1","article-title":"Mel frequency cepstral coefficients for music modeling","volume-title":"Proc. ISMIR","author":"Logan","year":"2000"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2002.800560"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/1180639.1180665"},{"key":"ref12","first-page":"509","article-title":"High-level music descriptor extraction algorithm based on combination of multi-channel CNNs and LSTM","volume-title":"Proc. ISMIR","author":"Chen","year":"2017"},{"key":"ref13","first-page":"546","article-title":"Automatic musical pattern feature extraction using convolutional neural network","volume-title":"Proc. IMECS","author":"Li","year":"2010"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414405"},{"key":"ref15","first-page":"714","article-title":"Representation learning of music using artist labels","volume-title":"Proc. ISMIR","author":"Park","year":"2017"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2017.2713830"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2984665"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/2926718"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3184558.3191823"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.2478\/eletel-2014-0042"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2009.2012913"},{"key":"ref22","article-title":"Multi-label zero-shot learning via concept embedding","author":"Sandouk","year":"2016"},{"key":"ref23","first-page":"67","article-title":"Zero-shot learning for audio-based music classification and tagging","volume-title":"Proc. ISMIR","author":"Choi","year":"2019"},{"key":"ref24","first-page":"438","article-title":"OpenMIC-2018: An open data-set for multiple instrument recognition","volume-title":"Proc. ISMIR","author":"Humphrey","year":"2018"},{"key":"ref25","article-title":"librosa: Audio and music signal analysis in python","author":"McFee","year":"2020"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"ref27","first-page":"1","article-title":"Gated graph sequence neural networks","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li","year":"2016"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1017\/S1351324916000334"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1145\/2623330.2623732"},{"key":"ref32","article-title":"Word2Vec tutorial - the skip-gram model"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939754"},{"key":"ref34","article-title":"Embedding projector","year":"2023"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref36","article-title":"ZLPR: A. novel loss for multi-label classification","author":"Su","year":"2022"},{"key":"ref38","first-page":"316","article-title":"FMA: A dataset for music analysis","volume-title":"Proc. ISMIR","author":"Defferrard","year":"2017"},{"key":"ref39","first-page":"1","article-title":"Adam: A method for stochastic pptimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kingma","year":"2017"},{"key":"ref40","article-title":"Max and coincidence neurons in neural networks","author":"Lee","year":"2021"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414337"},{"key":"ref42","first-page":"673","article-title":"Contrastive learning of musical representations","volume-title":"Proc. ISMIR","author":"Spijkervet","year":"2021"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref44","article-title":"GTZAN dataset - music genre classification","author":"Tzanetakis","year":"2002"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1080\/09298210802479268"},{"key":"ref46","first-page":"1","article-title":"The MTG-Jamendo dataset for automatic music tagging","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bogdanov","year":"2019"},{"key":"ref47","article-title":"ZSL_music_tagging","year":"2019"},{"key":"ref48","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2021"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/10304349\/10531231.pdf?arnumber=10531231","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,30]],"date-time":"2024-05-30T17:55:57Z","timestamp":1717091757000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10531231\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/taslp.2024.3402115","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}