{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T17:56:25Z","timestamp":1781286985869,"version":"3.54.1"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T00:00:00Z","timestamp":1781222400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T00:00:00Z","timestamp":1781222400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100018521","name":"International Science and Technology Cooperation Program of Jiangsu Province","doi-asserted-by":"publisher","award":["BY2021086, BY20230682"],"award-info":[{"award-number":["BY2021086, BY20230682"]}],"id":[{"id":"10.13039\/501100018521","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012221","name":"Jinling Institute of Technology","doi-asserted-by":"publisher","award":["jit-b-2021-09"],"award-info":[{"award-number":["jit-b-2021-09"]}],"id":[{"id":"10.13039\/501100012221","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Integration of Science and Education","award":["2024KJRH16"],"award-info":[{"award-number":["2024KJRH16"]}]},{"name":"Research Project on Educational Reform at Jinling University of Science and Technology in 2023","award":["JYJG202322"],"award-info":[{"award-number":["JYJG202322"]}]},{"DOI":"10.13039\/100020761","name":"High Level Innovation and Entrepreneurial Research Team Program in Jiangsu","doi-asserted-by":"publisher","award":["SJCX25_1300"],"award-info":[{"award-number":["SJCX25_1300"]}],"id":[{"id":"10.13039\/100020761","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Academic Degree and Postgraduate Education Reform Project of Jinling Institute of Technology","award":["YJSJG25_07"],"award-info":[{"award-number":["YJSJG25_07"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-026-21719-3","type":"journal-article","created":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T16:58:15Z","timestamp":1781283495000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Music emotion recognition using tri-modal fusion of lyrics, vocals, and accompaniment with cross-attention"],"prefix":"10.1007","volume":"85","author":[{"given":"Wenfeng","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6023-7005","authenticated-orcid":false,"given":"Yuehao","family":"Deng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuli","family":"Liao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chunyang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiqi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,12]]},"reference":[{"issue":"5","key":"21719_CR1","doi-asserted-by":"publisher","first-page":"559","DOI":"10.1017\/S0140525X08005293","volume":"31","author":"PN Juslin","year":"2008","unstructured":"Juslin PN, V\u00e4stfj\u00e4ll D (2008) Emotional responses to music: the need to consider underlying mechanisms. Behav Brain Sci 31(5):559\u2013575. https:\/\/doi.org\/10.1017\/S0140525X08005293","journal-title":"Behav Brain Sci"},{"key":"21719_CR2","doi-asserted-by":"publisher","first-page":"9663","DOI":"10.1038\/s41598-021-88431-0","volume":"11","author":"N Holz","year":"2021","unstructured":"Holz N, Larrouy-Maestri P, Poeppel D (2021) The paradoxical role of emotional intensity in the perception of vocal affect. Sci Rep 11:9663. https:\/\/doi.org\/10.1038\/s41598-021-88431-0","journal-title":"Sci Rep"},{"issue":"6","key":"21719_CR3","doi-asserted-by":"publisher","first-page":"776","DOI":"10.12068\/j.issn.1005-3026.2024.06.003","volume":"45","author":"D Han","year":"2024","unstructured":"Han D, Kong Y, Zhan Y, Liu Y (2024) Research on emotion recognition method of music multimodal data. J Northeast Univ Nat Sci 45(6):776\u2013787. https:\/\/doi.org\/10.12068\/j.issn.1005-3026.2024.06.003","journal-title":"J Northeast Univ Nat Sci"},{"key":"21719_CR4","doi-asserted-by":"publisher","unstructured":"Chiang W-C, Wang J-S, Hsu Y-L (2014) A music emotion recognition algorithm with hierarchical SVM based classifiers. In: Proceedings of the 2014 international symposium on computer, consumer and control (IS3C), pp 1249\u20131252. https:\/\/doi.org\/10.1109\/IS3C.2014.323","DOI":"10.1109\/IS3C.2014.323"},{"key":"21719_CR5","unstructured":"Hsu Y-J, Chen C-P (2017) Going deep: Improving music emotion recognition with layers of support vector machines. In: Proceedings of the 20th international conference on digital audio effects (DAFx-17). Edinburgh, UK, pp 1\u20138. https:\/\/dafx2017.napier.ac.uk\/papers\/DAFx17_paper_17.pdf"},{"key":"21719_CR6","doi-asserted-by":"publisher","unstructured":"Wang Z, Zhang S, Li X, Li H (2023) Frequency embedded regularization network for continuous music emotion recognition. In: Proceedings of the 2023 ACM international conference on multimedia retrieval (ICMR), pp 217\u2013225. https:\/\/doi.org\/10.1145\/3555048.3571102","DOI":"10.1145\/3555048.3571102"},{"key":"21719_CR7","doi-asserted-by":"publisher","unstructured":"Delbouys R, Henaff P, Piccoli A, Richard G (2018) Music emotion recognition with spectro-temporal representations and multi-task learning. In: Proceedings of the 19th international society for music information retrieval conference (ISMIR), pp 423\u2013430. https:\/\/doi.org\/10.5281\/zenodo.1416027","DOI":"10.5281\/zenodo.1416027"},{"key":"21719_CR8","doi-asserted-by":"publisher","unstructured":"Grekow J (2019) Deep learning approach to music emotion recognition. In: Kaczmarek KA et al (eds) Computer vision and graphics. ICCVG 2018. Lecture Notes in Computer Science. Springer, Cham, vol 11114, pp 367\u2013376. https:\/\/doi.org\/10.1007\/978-3-030-00837-3_31","DOI":"10.1007\/978-3-030-00837-3_31"},{"key":"21719_CR9","doi-asserted-by":"publisher","unstructured":"Gong Y, Chung Y-A, Glass J (2021) AST: Audio spectrogram transformer. In: Proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp 5704\u20135713. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00565","DOI":"10.1109\/ICCV48922.2021.00565"},{"key":"21719_CR10","doi-asserted-by":"publisher","first-page":"157716","DOI":"10.1109\/ACCESS.2024.3484470","volume":"12","author":"X Jiang","year":"2024","unstructured":"Jiang X, Zhang Y, Lin G, Yu L (2024) Music emotion recognition based on deep learning: a review. IEEE Access 12:157716\u2013157738. https:\/\/doi.org\/10.1109\/ACCESS.2024.3484470","journal-title":"IEEE Access"},{"key":"21719_CR11","doi-asserted-by":"publisher","unstructured":"Panda R (2023) Audio features for music emotion recognition: a survey. In: Proceedings of the 2023 international conference on computational science and computational intelligence (CSCI), pp 1\u20136. https:\/\/doi.org\/10.1109\/CSCI60327.2023.10540250","DOI":"10.1109\/CSCI60327.2023.10540250"},{"key":"21719_CR12","doi-asserted-by":"publisher","first-page":"103976","DOI":"10.1109\/ACCESS.2024.3430850","volume":"12","author":"S Kalateh","year":"2024","unstructured":"Kalateh S, Estrada-Jimenez LA, Nikghadam-Hojjati S, Barata J (2024) A systematic review on multimodal emotion recognition: building blocks, current state, applications, and challenges. IEEE Access 12:103976\u2013104008. https:\/\/doi.org\/10.1109\/ACCESS.2024.3430850","journal-title":"IEEE Access"},{"issue":"2","key":"21719_CR13","doi-asserted-by":"publisher","first-page":"448","DOI":"10.1109\/TASL.2007.911513","volume":"16","author":"Y-H Yang","year":"2008","unstructured":"Yang Y-H, Lin Y-C, Su Y-F, Chen HH (2008) A regression approach to music emotion recognition. IEEE Trans Audio Speech Lang Process 16(2):448\u2013460. https:\/\/doi.org\/10.1109\/TASL.2007.911513","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"21719_CR14","doi-asserted-by":"publisher","unstructured":"Li X, Zhang Q, Guo Y (2016) A deep bidirectional long short-term memory based multi-scale approach for music dynamic emotion prediction. In: Proceedings of the 2016 international conference on audio, language and image processing (ICALIP), pp 1\u20135. https:\/\/doi.org\/10.1109\/ICALIP.2016.7841685","DOI":"10.1109\/ICALIP.2016.7841685"},{"key":"21719_CR15","doi-asserted-by":"publisher","unstructured":"Xie S, Zhang Z, Cao Y, Lin Y, Bao J, Yao Z, Dai Q, Hu H (2019) Attention-based LSTM for speech emotion classification. In: Proceedings of the 2019 international conference on intelligent transportation, big data & smart city (ICITBS), pp 337\u2013340. https:\/\/doi.org\/10.1109\/ICITBS.2019.00083","DOI":"10.1109\/ICITBS.2019.00083"},{"issue":"5","key":"21719_CR16","doi-asserted-by":"publisher","first-page":"1114","DOI":"10.3969\/j.issn.1002-0802.2019.05.014","volume":"52","author":"C-F Chen","year":"2019","unstructured":"Chen C-F (2019) Music audio sentiment classification based on CNN-LSTM. Commun Technol 52(5):1114\u20131118. https:\/\/doi.org\/10.3969\/j.issn.1002-0802.2019.05.014","journal-title":"Commun Technol"},{"key":"21719_CR17","doi-asserted-by":"publisher","unstructured":"Baevski A, Hsu W-N, Xu Q, Babu A, Gu J, Auli M (2022) Data2Vec: a general framework for self-supervised learning in speech, vision and language. In: Proceedings of the international conference on learning representations (ICLR). https:\/\/doi.org\/10.48550\/arXiv.2202.03555","DOI":"10.48550\/arXiv.2202.03555"},{"key":"21719_CR18","doi-asserted-by":"publisher","unstructured":"Li Y, Yuan R, Zhang G, Ma Y, Chen X, Yin H, Xiao C, Lin C, Ragni A, Benetos E et al (2024) MERT: acoustic music understanding model with large-scale self-supervised training. In: Proceedings of the international conference on learning representations (ICLR). https:\/\/doi.org\/10.48550\/arXiv.2306.00107","DOI":"10.48550\/arXiv.2306.00107"},{"key":"21719_CR19","doi-asserted-by":"publisher","unstructured":"Liu Y, Zhao Y, Ju L, Shi S (2020) Multi-modal music emotion classification based on optimized residual network. In: Proceedings of the 2020 international conference on audio, language and image processing (ICALIP), pp 520\u2013524. https:\/\/doi.org\/10.1109\/ICALIP48876.2020.9133088","DOI":"10.1109\/ICALIP48876.2020.9133088"},{"key":"21719_CR20","doi-asserted-by":"publisher","unstructured":"Li H, Wang X (2020) Research on multi-modal music emotion classification based on audio and lyric. In: Proceedings of the 2020 3rd international conference on computer information science and application (CISAT), pp 1\u20135. https:\/\/doi.org\/10.1109\/CISAT50164.2020.00007","DOI":"10.1109\/CISAT50164.2020.00007"},{"key":"21719_CR21","doi-asserted-by":"publisher","unstructured":"Zou Y, Wang X (2022) Speech emotion recognition with co-attention based multi-level acoustic information. In: Proceedings of the 2022 5th international conference on computer information science and application (CISAT), pp 1\u20135. https:\/\/doi.org\/10.1109\/CISAT55588.2022.10024321","DOI":"10.1109\/CISAT55588.2022.10024321"},{"key":"21719_CR22","doi-asserted-by":"publisher","unstructured":"Zhao R, Mao K (2022) Music-CRN: an efficient content-based music classification and recommendation network. In: Proceedings of the 2022 international joint conference on neural networks (IJCNN), pp 1\u20138. https:\/\/doi.org\/10.1109\/IJCNN55064.2022.9892585","DOI":"10.1109\/IJCNN55064.2022.9892585"},{"key":"21719_CR23","unstructured":"Wang X (2024) Multimodal emotion analysis method based on cross-modal cross-attention network. Master\u2019s Thesis, Beijing University of Posts and Telecommunications"},{"key":"21719_CR24","doi-asserted-by":"publisher","unstructured":"Yang S, Li Q, Zhang W, Li M, Wu Y (2024) COSMIC: music emotion recognition combining structure analysis and modal interaction. In: Proceedings of the ACM international conference on multimedia retrieval (ICMR), pp 1\u20139. https:\/\/doi.org\/10.1145\/3641409.3659321","DOI":"10.1145\/3641409.3659321"},{"key":"21719_CR25","doi-asserted-by":"publisher","unstructured":"Mao K, Zhao R (2022) Multi-modal music emotion recognition with hierarchical cross-modal attention network. In: Proceedings of the 2022 international conference on acoustics, speech and signal processing (ICASSP), pp 3518\u20133522. https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9747321","DOI":"10.1109\/ICASSP43922.2022.9747321"},{"key":"21719_CR26","doi-asserted-by":"publisher","unstructured":"Wu Z, Gong Z, Koo J, Hirschberg J (2024) Multimodal multi-loss fusion network for sentiment analysis. In: Proceedings of the 2024 conference of the North American chapter of the association for computational linguistics (NAACL), pp 1\u201315. https:\/\/doi.org\/10.18653\/v1\/2024.naacl-long.123","DOI":"10.18653\/v1\/2024.naacl-long.123"},{"key":"21719_CR27","unstructured":"de Berardinis J, Cangelosi A, Coutinho E (2020) The multiple voices of musical emotions: source separation for improving music emotion recognition models and their interpretability. In: Proceedings of the 21st international society for music information retrieval conference (ISMIR), pp 21\u201326"},{"issue":"50","key":"21719_CR28","doi-asserted-by":"publisher","first-page":"2154","DOI":"10.21105\/joss.02154","volume":"5","author":"R Hennequin","year":"2020","unstructured":"Hennequin R, Khlif A, Voituret F, Moussallam M (2020) Spleeter: a fast and efficient music source separation tool with pre-trained models. J Open Source Softw 5(50):2154. https:\/\/doi.org\/10.21105\/joss.02154","journal-title":"J Open Source Softw"},{"key":"21719_CR29","doi-asserted-by":"publisher","unstructured":"D\u00e9fossez A, Usunier N, Bottou L, Bach F (2020) Demucs: deep extractor for music sources with extra neural network tools. In: Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 73\u201377. https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9054728","DOI":"10.1109\/ICASSP40776.2020.9054728"},{"key":"21719_CR30","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference on empirical methods in natural language processing and the 9th international joint conference on natural language processing (EMNLP-IJCNLP), pp 4171\u20134186. https:\/\/doi.org\/10.18653\/v1\/D19-1423","DOI":"10.18653\/v1\/D19-1423"},{"issue":"3","key":"21719_CR31","doi-asserted-by":"publisher","first-page":"94","DOI":"10.3778\/j.issn.1002-8331.2206-0113","volume":"59","author":"Z Zhong","year":"2023","unstructured":"Zhong Z, Wang H, Su G, Liu L, Pei D (2023) Fusion CNN-BiLSTM and self-attention model for music emotion recognition. Comput Eng Appl 59(3):94\u2013103. https:\/\/doi.org\/10.3778\/j.issn.1002-8331.2206-0113","journal-title":"Comput Eng Appl"},{"key":"21719_CR32","doi-asserted-by":"publisher","unstructured":"Zhang Q, Li Y, Wang X (2019) The PMEmo dataset for music emotion recognition. In: Proceedings of the 2019 ACM international conference on multimedia retrieval (ICMR), pp 37\u201342. https:\/\/doi.org\/10.1145\/3323873.3325023","DOI":"10.1145\/3323873.3325023"},{"key":"21719_CR33","unstructured":"Liu Y, Ott M, Goyal N, Du J, Joshi M, Chen D, Levy O, Lewis M, Zettlemoyer L, Stoyanov V (2019) RoBERTa: a robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692"},{"key":"21719_CR34","unstructured":"Yan S, Yang D, Ding X, He B, Zhang M (2020) A speaker system based on CLDNN music emotion recognition algorithm. In: Proceedings of the 2020 IEEE 9th international conference on information technology and software engineering (ICETIS). Harbin, China, pp 1\u20135"},{"key":"21719_CR35","doi-asserted-by":"publisher","first-page":"177509","DOI":"10.1109\/ACCESS.2025.3614020","volume":"13","author":"MA Talaghat","year":"2025","unstructured":"Talaghat MA, Parvinnia E, Mehrabi M, Boostani R (2025) A multimodal deep network for music emotion recognition using audio chorus and lyrics. IEEE Access 13:177509\u2013177519. https:\/\/doi.org\/10.1109\/ACCESS.2025.3614020","journal-title":"IEEE Access"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21719-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-026-21719-3","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21719-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T16:58:26Z","timestamp":1781283506000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-026-21719-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,12]]},"references-count":35,"journal-issue":{"issue":"6","published-online":{"date-parts":[[2026,6]]}},"alternative-id":["21719"],"URL":"https:\/\/doi.org\/10.1007\/s11042-026-21719-3","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,12]]},"assertion":[{"value":"10 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 May 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 June 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 June 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The PMEmo dataset used in this study is publicly available at\n                      \n                      . Due to copyright restrictions on music content, the self-constructed MCEmo dataset cannot be fully shared publicly. However, de-identified feature data (e.g., lyrical semantic vectors, vocals\/accompaniment feature matrices) and model training code are available from the corresponding authors (Wenfeng Li: wenfenglee88@jit.edu.cn) upon reasonable request.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Availability of Data and Material"}},{"value":"The authors declare no competing financial or non-financial interests relevant to this study.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"559"}}