{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T14:57:05Z","timestamp":1773932225862,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,4,25]],"date-time":"2022-04-25T00:00:00Z","timestamp":1650844800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Zhejiang Natural Science Foundation award","award":["LR19F020006"],"award-info":[{"award-number":["LR19F020006"]}]},{"name":"National Natural Science Foundation of China under Grant award","award":["61836002, 62072397, 62037001"],"award-info":[{"award-number":["61836002, 62072397, 62037001"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,4,25]]},"DOI":"10.1145\/3485447.3512011","type":"proceedings-article","created":{"date-parts":[[2022,4,25]],"date-time":"2022-04-25T05:13:07Z","timestamp":1650863587000},"page":"2906-2915","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":13,"title":["Contrastive Learning with Positive-Negative Frame Mask for Music Representation"],"prefix":"10.1145","author":[{"given":"Dong","family":"Yao","sequence":"first","affiliation":[{"name":"Zhejiang University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhou","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shengyu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Zhejiang University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jieming","family":"Zhu","sequence":"additional","affiliation":[{"name":"Huawei Noah\u2019s Ark Lab, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yudong","family":"Zhu","sequence":"additional","affiliation":[{"name":"Huawei Noah\u2019s Ark Lab, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rui","family":"Zhang","sequence":"additional","affiliation":[{"name":"www.ruizhang.info, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiuqiang","family":"He","sequence":"additional","affiliation":[{"name":"Huawei Noah\u2019s Ark Lab, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,4,25]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Learning representations by maximizing mutual information across views. Advances in Neural Information Processing Systems 32","author":"Bachman Philip","year":"2019","unstructured":"Philip Bachman, R. Devon Hjelm, and William Buchwalter. 2019. Learning representations by maximizing mutual information across views. Advances in Neural Information Processing Systems 32 (2019). arxiv:1906.00910"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00156"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Mathilde Caron Piotr Bojanowski Armand Joulin and Matthijs Douze. 2018. Deep clustering for unsupervised learning of visual features. Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics) 11218 LNCS (2018) 139\u2013156.","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"e_1_3_2_1_4_1","volume-title":"Unsupervised Learning of Visual Features by Contrasting Cluster Assignments. NeurIPS","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. 2020. Unsupervised Learning of Visual Features by Contrasting Cluster Assignments. NeurIPS (2020), 1\u201313. arXiv:2006.09882"},{"key":"e_1_3_2_1_5_1","volume-title":"SingGAN: Generative Adversarial Network For High-Fidelity Singing Voice Generation. 0","author":"Chen Feiyang","year":"2021","unstructured":"Feiyang Chen, Rongjie Huang, Chenye Cui, Yi Ren, Jinglin Liu, and Zhou Zhao. 2021. SingGAN: Generative Adversarial Network For High-Fidelity Singing Voice Generation. 0 (2021). arxiv:2110.07468http:\/\/arxiv.org\/abs\/2110.07468"},{"key":"e_1_3_2_1_6_1","unstructured":"Ting Chen Simon Kornblith Mohammad Norouzi and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. arXivFigure 1(2020). arXiv:2002.05709"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the 17th International Society for Music Information Retrieval Conference, ISMIR 2016(2016)","author":"Choi Keunwoo","year":"2016","unstructured":"Keunwoo Choi, Gy\u00f6rgy Fazekas, and Mark Sandler. 2016. Automatic tagging using deep convolutional neural networks. Proceedings of the 17th International Society for Music Information Retrieval Conference, ISMIR 2016(2016), 805\u2013811. arxiv:1606.00298"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952585"},{"key":"e_1_3_2_1_9_1","volume-title":"Conference and Signal Processing. 2014","year":"2014","unstructured":"Ieee\u00a0International Conference and Signal Processing. 2014. END-TO-END LEARNING FOR MUSIC AUDIO Sander Dieleman, Benjamin Schrauwen Electronics and information systems department. Icassp (2014), 7014\u20137018."},{"key":"e_1_3_2_1_10_1","volume-title":"NAACL HLT 2019 - 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies - Proceedings of the Conference 1(2019)","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming\u00a0Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of deep bidirectional transformers for language understanding. NAACL HLT 2019 - 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies - Proceedings of the Conference 1(2019), 4171\u20134186. arXiv:1810.04805"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.4310\/HHA.2007.v9.n1.a16"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"e_1_3_2_1_14_1","volume-title":"Bootstrap your own latent: A new approach to self-supervised Learning. 200","author":"Grill Jean-Bastien","year":"2020","unstructured":"Jean-Bastien Grill, Florian Strub, Florent Altch\u00e9, Corentin Tallec, Pierre\u00a0H. Richemond, Elena Buchatskaya, Carl Doersch, Bernardo\u00a0Avila Pires, Zhaohan\u00a0Daniel Guo, Mohammad\u00a0Gheshlaghi Azar, Bilal Piot, Koray Kavukcuoglu, R\u00e9mi Munos, and Michal Valko. 2020. Bootstrap your own latent: A new approach to self-supervised Learning. 200 (2020). arXiv:2006.07733http:\/\/arxiv.org\/abs\/2006.07733"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.100"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_1_17_1","unstructured":"Olivier\u00a0J. H\u00e9naff Ali Razavi Carl Doersch S.\u00a0M. Ali Eslami and Aaron Van Den Oord. 2019. Data-efficient image recognition with contrastive predictive coding. arXiv2018(2019). arXiv:1905.09272"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475437"},{"key":"e_1_3_2_1_19_1","volume-title":"IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings 2020-May","author":"Jiang Chaoya","year":"2020","unstructured":"Chaoya Jiang, Deshun Yang, and Xiaoou Chen. 2020. Similarity Learning for Cover Song Identification Using Cross-Similarity Matrices of Multi-Level Deep Sequences. ICASSP, IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings 2020-May (2020), 26\u201330."},{"key":"e_1_3_2_1_20_1","unstructured":"Ziyu Jiang Tianlong Chen Bobak Mortazavi and Zhangyang Wang. 2021. Self-Damaging Contrastive Learning. (2021). arXiv:2106.02990http:\/\/arxiv.org\/abs\/2106.02990"},{"key":"e_1_3_2_1_21_1","volume-title":"Kipf and Max Welling","author":"N.","year":"2016","unstructured":"Thomas\u00a0N. Kipf and Max Welling. 2016. Variational Graph Auto-Encoders. 2 (2016), 1\u20133. arXiv:1611.07308http:\/\/arxiv.org\/abs\/1611.07308"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00202"},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the 10th International Society for Music Information Retrieval Conference, ISMIR 2009Ismir","author":"Law Edith","year":"2009","unstructured":"Edith Law, Kris West, Michael Mandel, Mert Bay, and J. Stephen Downie. 2009. Evaluation of algorithms using games: The case of music tagging. Proceedings of the 10th International Society for Music Information Retrieval Conference, ISMIR 2009Ismir (2009), 387\u2013392."},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 14th Sound and Music Computing Conference 2017, SMC 2017","author":"Lee Jongpil","year":"2019","unstructured":"Jongpil Lee, Jiyoung Park, Keunhyoung\u00a0Luke Kim, and Juhan Nam. 2019. Sample-level deep convolutional neural networks for music auto-tagging using raw waveforms. Proceedings of the 14th Sound and Music Computing Conference 2017, SMC 2017 (2019), 220\u2013226. arXiv:1703.01789"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054458"},{"key":"e_1_3_2_1_26_1","unstructured":"Jinglin Liu Chengxi Li Yi Ren Feiyang Chen and Zhou Zhao. 2021. DiffSinger: Singing Voice Synthesis via Shallow Diffusion Mechanism. (2021). arxiv:2105.02446http:\/\/arxiv.org\/abs\/2105.02446"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Daisuke Niizumi Daiki Takeuchi Yasunori Ohishi Noboru Harada and Kunio Kashino. 2021. BYOL for Audio: Self-Supervised Learning for General-Purpose Audio Representation. (2021). arXiv:2103.06695http:\/\/arxiv.org\/abs\/2103.06695","DOI":"10.1109\/IJCNN52387.2021.9534474"},{"key":"e_1_3_2_1_28_1","unstructured":"Jordi Pons and Xavier Serra. 2019. musicnn: Pre-trained convolutional neural networks for music audio tagging. (2019) 4\u20135. arXiv:1909.06654http:\/\/arxiv.org\/abs\/1909.06654"},{"key":"e_1_3_2_1_29_1","first-page":"1","article-title":"Generating diverse high-fidelity images with VQ-VAE-2","volume":"2019","author":"Razavi Ali","year":"2019","unstructured":"Ali Razavi, A\u00e4ron van\u00a0den Oord, and Oriol Vinyals. 2019. Generating diverse high-fidelity images with VQ-VAE-2. Advances in Neural Information Processing Systems 32, NeurIPS 2019(2019), 1\u201311. arXiv:1906.00446","journal-title":"Advances in Neural Information Processing Systems 32, NeurIPS"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413721"},{"key":"e_1_3_2_1_31_1","volume-title":"PortaSpeech: Portable and High-Quality Generative Text-to-Speech. NeurIPS","author":"Ren Yi","year":"2021","unstructured":"Yi Ren, Jinglin Liu, and Zhou Zhao. 2021. PortaSpeech: Portable and High-Quality Generative Text-to-Speech. NeurIPS (2021). arxiv:2109.15166http:\/\/arxiv.org\/abs\/2109.15166"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","unstructured":"Aaqib Saeed David Grangier and Neil Zeghidour. 2021. Contrastive Learning of General-Purpose Audio Representations. (2021) 3875\u20133879. https:\/\/doi.org\/10.1109\/icassp39728.2021.9413528 arXiv:2010.10915","DOI":"10.1109\/icassp39728.2021.9413528"},{"key":"e_1_3_2_1_33_1","unstructured":"Janne Spijkervet and John\u00a0Ashley Burgoyne. 2021. Contrastive Learning of Musical Representations. (2021). arxiv:2103.09410http:\/\/arxiv.org\/abs\/2103.09410"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 15th International Society for Music Information Retrieval Conference, ISMIR 2014Ismir","author":"van\u00a0den Oord A\u00e4ron","year":"2014","unstructured":"A\u00e4ron van\u00a0den Oord, Sander Dieleman, and Benjamin Schrauwen. 2014. Transfer learning by supervised pre-training for audio-based music classification. Proceedings of the 15th International Society for Music Information Retrieval Conference, ISMIR 2014Ismir (2014), 29\u201334."},{"key":"e_1_3_2_1_36_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N. Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in Neural Information Processing Systems 2017-Decem Nips(2017) 5999\u20136009. arXiv:1706.03762"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","unstructured":"Ho-Hsiang Wu Chieh-Chi Kao Qingming Tang Ming Sun Brian McFee Juan\u00a0Pablo Bello and Chao Wang. 2021. Multi-Task Self-Supervised Pre-Training for Music Classification. (2021) 556\u2013560. https:\/\/doi.org\/10.1109\/icassp39728.2021.9414405 arXiv:2102.03229","DOI":"10.1109\/icassp39728.2021.9414405"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2018.8486531"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2018.8486531"},{"key":"e_1_3_2_1_40_1","volume-title":"IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings 2020-May","author":"Yesiler Furkan","year":"2020","unstructured":"Furkan Yesiler, Joan Serra, and Emilia Gomez. 2020. Accurate and Scalable Version Identification Using Musically-Motivated Embeddings. ICASSP, IEEE International Conference on Acoustics, Speech and Signal Processing - Proceedings 2020-May, Icassp (2020), 21\u201325."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/673"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053839"},{"key":"e_1_3_2_1_43_1","volume-title":"Barlow Twins: Self-Supervised Learning via Redundancy Reduction.","author":"Zbontar Jure","year":"2021","unstructured":"Jure Zbontar, Li Jing, Ishan Misra, Yann LeCun, and St\u00e9phane Deny. 2021. Barlow Twins: Self-Supervised Learning via Redundancy Reduction. (2021). arXiv:2103.03230http:\/\/arxiv.org\/abs\/2103.03230"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","unstructured":"Mingliang Zeng Xu Tan Rui Wang Zeqian Ju Tao Qin and Tie-Yan Liu. 2021. MusicBERT: Symbolic Music Understanding with Large-Scale Pre-Training. (2021) 791\u2013800. https:\/\/doi.org\/10.18653\/v1\/2021.findings-acl.70 arXiv:2106.05630","DOI":"10.18653\/v1"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462908"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","unstructured":"Yilun Zhao and Jia Guo. 2021. MusiCoder: A Universal Music-Acoustic Encoder Based on Transformer. Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics) 12572 LNCS (2021) 417\u2013429. https:\/\/doi.org\/10.1007\/978-3-030-67832-6_34 arXiv:2008.00781","DOI":"10.1007\/978-3-030-67832-6_34"}],"event":{"name":"WWW '22: The ACM Web Conference 2022","location":"Virtual Event, Lyon France","acronym":"WWW '22","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2022"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3485447.3512011","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3485447.3512011","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:30:06Z","timestamp":1750188606000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3485447.3512011"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,4,25]]},"references-count":46,"alternative-id":["10.1145\/3485447.3512011","10.1145\/3485447"],"URL":"https:\/\/doi.org\/10.1145\/3485447.3512011","relation":{},"subject":[],"published":{"date-parts":[[2022,4,25]]},"assertion":[{"value":"2022-04-25","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}