{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T03:46:41Z","timestamp":1781063201193,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100006433","name":"Barcelona Supercomputing Center","doi-asserted-by":"publisher","award":["IM-2024-2-0034"],"award-info":[{"award-number":["IM-2024-2-0034"]}],"id":[{"id":"10.13039\/501100006433","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Secretar\u00eda de Estado de Digitalizaci\u00f3n e Inteligencia Artificial and the European Union-Next Generation EU","award":["SI-100929-2023-1"],"award-info":[{"award-number":["SI-100929-2023-1"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3756871","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:44:48Z","timestamp":1761371088000},"page":"13640-13643","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["OMAR-RQ: Open Music Audio Representation Model Trained with Multi-Feature Masked Token Prediction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4121-5089","authenticated-orcid":false,"given":"Pablo","family":"Alonso-Jim\u00e9nez","sequence":"first","affiliation":[{"name":"Music Technology Group, Universitat Pompeu Fabra, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4017-537X","authenticated-orcid":false,"given":"Pedro","family":"Ramoneda","sequence":"additional","affiliation":[{"name":"Music Technology Group, Universitat Pompeu Fabra, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8456-0990","authenticated-orcid":false,"given":"R. Oguz","family":"Araz","sequence":"additional","affiliation":[{"name":"Music Technology Group, Universitat Pompeu Fabra, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3848-7574","authenticated-orcid":false,"given":"Andrea","family":"Poltronieri","sequence":"additional","affiliation":[{"name":"Music Technology Group, Universitat Pompeu Fabra, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9469-0633","authenticated-orcid":false,"given":"Dmitry","family":"Bogdanov","sequence":"additional","affiliation":[{"name":"Music Technology Group, Universitat Pompeu Fabra, Barcelona, Spain"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Alonso-Jim\u00e9nez Pablo","year":"2022","unstructured":"Pablo Alonso-Jim\u00e9nez, Dmitry Bogdanov, and Xavier Serra. 2022. Music Representation Learning Based on Editorial Metadata from Discogs. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_2_1","volume-title":"Efficient Supervised Training of Audio Transformers for Music Representation Learning. In Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Alonso-Jim\u00e9nez Pablo","year":"2023","unstructured":"Pablo Alonso-Jim\u00e9nez, Xavier Serra, and Dmitry Bogdanov. 2023. Efficient Supervised Training of Audio Transformers for Music Representation Learning. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_3_1","volume-title":"Daniel PW Ellis, and Brian Whitman","author":"Berenzweig Adam","year":"2004","unstructured":"Adam Berenzweig, Beth Logan, Daniel PW Ellis, and Brian Whitman. 2004. A Large-scale Evaluation of Acoustic and Subjective Music-similarity Measures. Computer Music Journal (2004), 63-76."},{"key":"e_1_3_2_1_4_1","volume-title":"An Expert Ground Truth Set for Audio Chord Recognition and Music Analysis. In Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Burgoyne John Ashley","year":"2011","unstructured":"John Ashley Burgoyne, Jonathan Wild, and Ichiro Fujinaga. 2011. An Expert Ground Truth Set for Audio Chord Recognition and Music Analysis. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_5_1","volume-title":"Self-Supervised Learning with Random-Projection Quantizer for Speech Recognition. In Int. Conf. on Machine Learning (ICML).","author":"Chiu Chung-Cheng","year":"2022","unstructured":"Chung-Cheng Chiu, James Qin, Yu Zhang, Jiahui Yu, and Yonghui Wu. 2022. Self-Supervised Learning with Random-Projection Quantizer for Speech Recognition. In Int. Conf. on Machine Learning (ICML)."},{"key":"e_1_3_2_1_6_1","unstructured":"Tri Dao Daniel Y. Fu Stefano Ermon Atri Rudra and Christopher R\u00e9. 2022. FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_7_1","volume-title":"High Fidelity Neural Audio Compression. arXiv preprint arXiv:2210.13438","author":"D\u00e9fossez Alexandre","year":"2022","unstructured":"Alexandre D\u00e9fossez, Jade Copet, Gabriel Synnaeve, and Yossi Adi. 2022. High Fidelity Neural Audio Compression. arXiv preprint arXiv:2210.13438 (2022)."},{"key":"e_1_3_2_1_8_1","volume-title":"Neural Audio Synthesis of Musical Notes with Wavenet Autoencoders. In Int. Conf. on Machine Learning (ICML).","author":"Engel Jesse","year":"2017","unstructured":"Jesse Engel, Cinjon Resnick, Adam Roberts, Sander Dieleman, Mohammad Norouzi, Douglas Eck, and Karen Simonyan. 2017. Neural Audio Synthesis of Musical Notes with Wavenet Autoencoders. In Int. Conf. on Machine Learning (ICML)."},{"key":"e_1_3_2_1_9_1","volume-title":"Classical and Jazz Music Databases. In Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Goto Masataka","year":"2002","unstructured":"Masataka Goto, Hiroki Hashiguchi, Takuichi Nishimura, and Ryuichi Oka. 2002. RWC Music Database: Popular, Classical and Jazz Music Databases. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_10_1","volume-title":"Conformer: Convolution-Augmented Transformer for Speech Recognition. arXiv preprint arXiv:2005.08100","author":"Gulati Anmol","year":"2020","unstructured":"Anmol Gulati, James Qin, Chung-Cheng Chiu, Niki Parmar, Yu Zhang, Jiahui Yu, Wei Han, Shibo Wang, Zhengdong Zhang, Yonghui Wu, et al., 2020. Conformer: Convolution-Augmented Transformer for Speech Recognition. arXiv preprint arXiv:2005.08100 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"Advances in Neural Information Processing Systems","volume":"35","author":"Huang Po-Yao","year":"2022","unstructured":"Po-Yao Huang, Hu Xu, Juncheng Li, Alexei Baevski, Michael Auli, Wojciech Galuba, Florian Metze, and Christoph Feichtenhofer. 2022. Masked Autoencoders that Listen. Advances in Neural Information Processing Systems, Vol. 35 (2022)."},{"key":"e_1_3_2_1_12_1","volume-title":"Robust Training of Vector Quantized Bottleneck Models. In 2020 Int. Joint Conf. on Neural Networks (IJCNN).","author":"\u0141a'ncucki Adrian","year":"2020","unstructured":"Adrian \u0141a'ncucki, Jan Chorowski, Guillaume Sanchez, Ricard Marxer, Nanxin Chen, Hans JGA Dolfing, Sameer Khurana, Tanel Alum\u00e4e, and Antoine Laurent. 2020. Robust Training of Vector Quantized Bottleneck Models. In 2020 Int. Joint Conf. on Neural Networks (IJCNN)."},{"key":"e_1_3_2_1_13_1","volume-title":"Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Law Edith","year":"2009","unstructured":"Edith Law, Kris West, Michael I Mandel, Mert Bay, and J Stephen Downie. 2009. Evaluation of Algorithms Using Games: The Case of Music Tagging.. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_14_1","unstructured":"Ugo Marchand Quentin Fresnel and Geoffroy Peeters. 2015. GTZAN-rhythm: Extending the GTZAN Test-Set With Beat Downbeat and Swing Annotations. (2015)."},{"key":"e_1_3_2_1_15_1","volume-title":"Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Mauch Matthias","year":"2009","unstructured":"Matthias Mauch, Chris Cannam, Matthew Davies, Simon Dixon, Christopher Harte, Sefki Kolozali, Dan Tidhar, and Mark Sandler. 2009. OMRAS2 Metadata Project 2009. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_16_1","volume-title":"Finite Scalar Quantization: VQ-VAE Made Simple. In Int. Conf. on Learning Representations (ICLR).","author":"Mentzer Fabian","year":"2024","unstructured":"Fabian Mentzer, David Minnen, Eirikur Agustsson, and Michael Tschannen. 2024. Finite Scalar Quantization: VQ-VAE Made Simple. In Int. Conf. on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_17_1","volume-title":"Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Nieto Oriol","year":"2019","unstructured":"Oriol Nieto, Matthew C McCallum, Matthew EP Davies, Andrew Robertson, Adam M Stark, and Eran Egozy. 2019. The Harmonix Set: Beats, Downbeats, and Functional Segment Annotations of Western Popular Music.. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3539018"},{"key":"e_1_3_2_1_19_1","volume-title":"Contrastive Learning of Musical Representations. In Int. Society for Music Information Retrieval Conf. (ISMIR).","author":"Spijkervet Janne","year":"2021","unstructured":"Janne Spijkervet and John Ashley Burgoyne. 2021. Contrastive Learning of Musical Representations. In Int. Society for Music Information Retrieval Conf. (ISMIR)."},{"key":"e_1_3_2_1_20_1","first-page":"127063","article-title":"Roformer","volume":"568","author":"Su Jianlin","year":"2024","unstructured":"Jianlin Su, Murtadha Ahmed, Yu Lu, Shengfeng Pan, Wen Bo, and Yunfeng Liu. 2024. Roformer: Enhanced Transformer with Rotary Position Embedding. Neurocomputing, Vol. 568 (2024), 127063.","journal-title":"Enhanced Transformer with Rotary Position Embedding. Neurocomputing"},{"key":"e_1_3_2_1_21_1","volume-title":"Deepnet: Scaling Transformers to 1,000 Layers","author":"Wang Hongyu","year":"2024","unstructured":"Hongyu Wang, Shuming Ma, Li Dong, Shaohan Huang, Dongdong Zhang, and Furu Wei. 2024. Deepnet: Scaling Transformers to 1,000 Layers. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448314"},{"key":"e_1_3_2_1_23_1","volume-title":"Int. Conf. on Learning Representations (ICLR).","author":"Yizhi Li","year":"2023","unstructured":"Li Yizhi, Ruibin Yuan, Ge Zhang, Yinghao Ma, Xingran Chen, Hanzhi Yin, Chenghao Xiao, Chenghua Lin, Anton Ragni, Emmanouil Benetos, et al., 2023. MERT: Acoustic music understanding model with large-scale self-supervised training. In Int. Conf. on Learning Representations (ICLR)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3756871","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:11:17Z","timestamp":1765307477000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3756871"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":23,"alternative-id":["10.1145\/3746027.3756871","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3756871","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}