{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:39:49Z","timestamp":1776886789901,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":16,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733471","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:29:43Z","timestamp":1750876183000},"page":"2018-2022","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Bridging the Gap Between Semantic and User Preference Spaces for Multi-modal Music Representation Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-0719-3780","authenticated-orcid":false,"given":"Xiaofeng","family":"Pan","sequence":"first","affiliation":[{"name":"NetEase Inc., Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9583-0637","authenticated-orcid":false,"given":"Jing","family":"Chen","sequence":"additional","affiliation":[{"name":"NetEase Inc., Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1283-3788","authenticated-orcid":false,"given":"Haitong","family":"Zhang","sequence":"additional","affiliation":[{"name":"NetEase Inc., Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3613-094X","authenticated-orcid":false,"given":"Menglin","family":"Xing","sequence":"additional","affiliation":[{"name":"NetEase Inc., Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6693-6293","authenticated-orcid":false,"given":"Jiayi","family":"Wei","sequence":"additional","affiliation":[{"name":"NetEase Inc., Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3919-298X","authenticated-orcid":false,"given":"Xuefeng","family":"Mu","sequence":"additional","affiliation":[{"name":"NetEase Inc., Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0528-8982","authenticated-orcid":false,"given":"Zhongqian","family":"Xie","sequence":"additional","affiliation":[{"name":"NetEase Inc., Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2021.3071082"},{"key":"e_1_3_2_1_3_1","volume-title":"Ast: Audio spectrogram transformer. arXiv preprint arXiv:2104.01778","author":"Gong Yuan","year":"2021","unstructured":"Yuan Gong, Yu-An Chung, and James Glass. 2021. Ast: Audio spectrogram transformer. arXiv preprint arXiv:2104.01778 (2021)."},{"key":"e_1_3_2_1_4_1","first-page":"28708","article-title":"Masked autoencoders that listen","volume":"35","author":"Huang Po-Yao","year":"2022","unstructured":"Po-Yao Huang, Hu Xu, Juncheng Li, Alexei Baevski, Michael Auli, Wojciech Galuba, Florian Metze, and Christoph Feichtenhofer. 2022. Masked autoencoders that listen. Advances in Neural Information Processing Systems, Vol. 35 (2022), 28708--28720.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_5_1","volume-title":"A survey of convolutional neural networks: analysis, applications, and prospects","author":"Li Zewen","year":"2021","unstructured":"Zewen Li, Fan Liu, Wenjie Yang, Shouheng Peng, and Jun Zhou. 2021. A survey of convolutional neural networks: analysis, applications, and prospects. IEEE transactions on neural networks and learning systems, Vol. 33, 12 (2021), 6999--7019."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413528"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3591106.3592237"},{"key":"e_1_3_2_1_8_1","volume-title":"Contrastive learning of musical representations. arXiv preprint arXiv:2103.09410","author":"Spijkervet Janne","year":"2021","unstructured":"Janne Spijkervet and John Ashley Burgoyne. 2021. Contrastive learning of musical representations. arXiv preprint arXiv:2103.09410 (2021)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3591106.3592274"},{"key":"e_1_3_2_1_10_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the 1st Workshop on NLP for Music and Audio (NLP4MusA). 6--12","author":"Watanabe Kento","year":"2020","unstructured":"Kento Watanabe and Masataka Goto. 2020. Lyrics information processing: Analysis, generation, and applications. In Proceedings of the 1st Workshop on NLP for Music and Audio (NLP4MusA). 6--12."},{"key":"e_1_3_2_1_12_1","volume-title":"COURIER: Contrastive User Intention Reconstruction for Large-Scale Pre-Train of Image Features. arXiv preprint arXiv:2306.05001","author":"Yang Jia-Qi","year":"2023","unstructured":"Jia-Qi Yang, Chenglei Dai, OU Dan, Ju Huang, De-Chuan Zhan, Qingwen Liu, Xiaoyi Zeng, and Yang Yang. 2023. COURIER: Contrastive User Intention Reconstruction for Large-Scale Pre-Train of Image Features. arXiv preprint arXiv:2306.05001 (2023)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3652583.3658045"},{"key":"e_1_3_2_1_14_1","volume-title":"MERT: Acoustic Music Understanding Model with Large-Scale Self-supervised Training. In The Twelfth International Conference on Learning Representations.","author":"Yizhi LI","year":"2023","unstructured":"LI Yizhi, Ruibin Yuan, Ge Zhang, Yinghao Ma, Xingran Chen, Hanzhi Yin, Chenghao Xiao, Chenghua Lin, Anton Ragni, Emmanouil Benetos, et al. 2023. MERT: Acoustic Music Understanding Model with Large-Scale Self-supervised Training. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_15_1","volume-title":"Bootstrapping Contrastive Learning Enhanced Music Cold-Start Matching. In Companion Proceedings of the ACM Web Conference","author":"Zhao Xinping","year":"2023","unstructured":"Xinping Zhao, Ying Zhang, Qiang Xiao, Yuming Ren, and Yingchun Yang. 2023. Bootstrapping Contrastive Learning Enhanced Music Cold-Start Matching. In Companion Proceedings of the ACM Web Conference 2023. 351--355."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219823"}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","location":"Chicago IL USA","acronym":"ICMR '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733471","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:10:42Z","timestamp":1755749442000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733471"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":16,"alternative-id":["10.1145\/3731715.3733471","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733471","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}