{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:30:24Z","timestamp":1765308624498,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","funder":[{"DOI":"10.13039\/100017052","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62201524"],"award-info":[{"award-number":["62201524"]}],"id":[{"id":"10.13039\/100017052","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beijing Natural Science Foundation","award":["4252011"],"award-info":[{"award-number":["4252011"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["CUC25CGJ01"],"award-info":[{"award-number":["CUC25CGJ01"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754590","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:50:47Z","timestamp":1761371447000},"page":"826-835","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["FG-Midiformer: A Symbolic Music Understanding Model towards Fine-Grained Learning of Multi-Attributes"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3407-4318","authenticated-orcid":false,"given":"Haonan","family":"Cheng","sequence":"first","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0266-2460","authenticated-orcid":false,"given":"Junwei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2089-5050","authenticated-orcid":false,"given":"Hengyan","family":"Huang","sequence":"additional","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3562-5612","authenticated-orcid":false,"given":"Long","family":"Ye","sequence":"additional","affiliation":[{"name":"Communication University of China, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Peter V Desouza, Jennifer C Lai, and Robert L Mercer.","author":"Brown Peter F","year":"1992","unstructured":"Peter F Brown, Vincent J Della Pietra, Peter V Desouza, Jennifer C Lai, and Robert L Mercer. 1992. Class-based n-gram models of natural language. Computational linguistics, Vol. 18, 4 (1992), 467-480."},{"key":"e_1_3_2_1_2_1","volume-title":"Visual attention: The past 25 years. Vision research","author":"Carrasco Marisa","year":"2011","unstructured":"Marisa Carrasco. 2011. Visual attention: The past 25 years. Vision research, Vol. 51, 13 (2011), 1484-1525."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2008.916370"},{"issue":"4673","key":"e_1_3_2_1_4_1","first-page":"226","article-title":"Melody retrieval on the web","volume":"2002","author":"Chai W","year":"2001","unstructured":"W Chai and B Vercoe. 2001. Melody retrieval on the web. Multimedia Computing and Networking 2002, 4673, 226-241.","journal-title":"Multimedia Computing and Networking"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696409.3700221"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475405"},{"key":"e_1_3_2_1_7_1","unstructured":"Yi-Hui Chou I Chen Chin-Jui Chang Joann Ching Yi-Hsuan Yang et al. 2021. MidiBERT-piano: large-scale pre-training for symbolic music understanding. arXiv preprint arXiv:2107.05223 (2021)."},{"key":"e_1_3_2_1_8_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_9_1","first-page":"4171","volume-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers). 4171-4186."},{"key":"e_1_3_2_1_10_1","volume-title":"ZEN: Pre-training Chinese text encoder enhanced by n-gram representations. arXiv preprint arXiv:1911.00720","author":"Diao Shizhe","year":"2019","unstructured":"Shizhe Diao, Jiaxin Bai, Yan Song, Tong Zhang, and Yonggang Wang. 2019. ZEN: Pre-training Chinese text encoder enhanced by n-gram representations. arXiv preprint arXiv:1911.00720 (2019)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1609\/aiide.v16i1.7408"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11023-020-09548-1"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 21st International Society for Music Information Retrieval Conference. 534-541","author":"Foscarin Francesco","year":"2020","unstructured":"Francesco Foscarin, Andrew Mcleod, Philippe Rigaux, Florent Jacquemard, and Masahiko Sakai. 2020. ASAP: a dataset of aligned scores and performances for piano transcription. In Proceedings of the 21st International Society for Music Information Retrieval Conference. 534-541."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-04125-9_29"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.2197\/ipsjjip.27.278"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i1.16091"},{"key":"e_1_3_2_1_17_1","volume-title":"Music transformer: Generating music with long-term structure","author":"Anna Huang Cheng-Zhi","year":"2018","unstructured":"Cheng-Zhi Anna Huang, Ashish Vaswani, Jakob Uszkoreit, Noam Shazeer, Curtis Hawthorne, AM Dai, MD Hoffman, and D Eck. 2018. Music transformer: Generating music with long-term structure (2018). arXiv preprint arXiv:1809.04281 (2018)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413671"},{"key":"e_1_3_2_1_19_1","volume-title":"EMOPIA: A multi-modal pop piano dataset for emotion recognition and emotion-based music generation. arXiv preprint arXiv:2108.01374","author":"Hung Hsiao-Tzu","year":"2021","unstructured":"Hsiao-Tzu Hung, Joann Ching, Seungheon Doh, Nabin Kim, Juhan Nam, and Yi-Hsuan Yang. 2021. EMOPIA: A multi-modal pop piano dataset for emotion recognition and emotion-based music generation. arXiv preprint arXiv:2108.01374 (2021)."},{"key":"e_1_3_2_1_20_1","volume-title":"Parallels and nonparallels between language and music. Music perception","author":"Jackendoff Ray","year":"2009","unstructured":"Ray Jackendoff. 2009. Parallels and nonparallels between language and music. Music perception, Vol. 26, 3 (2009), 195-204."},{"key":"e_1_3_2_1_21_1","first-page":"908","article-title":"VirtuosoNet: A Hierarchical RNN-based System for Modeling Expressive Piano Performance","author":"Jeong Dasaem","year":"2019","unstructured":"Dasaem Jeong, Taegyun Kwon, Yoojin Kim, Kyogu Lee, and Juhan Nam. 2019b. VirtuosoNet: A Hierarchical RNN-based System for Modeling Expressive Piano Performance.. In ISMIR. 908-915.","journal-title":"ISMIR."},{"key":"e_1_3_2_1_22_1","volume-title":"International conference on machine learning. PMLR, 3060-3070","author":"Jeong Dasaem","year":"2019","unstructured":"Dasaem Jeong, Taegyun Kwon, Yoojin Kim, and Juhan Nam. 2019a. Graph neural network for music score data and modeling expressive piano performance. In International conference on machine learning. PMLR, 3060-3070."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612140"},{"key":"e_1_3_2_1_24_1","volume-title":"Deep composer classification using symbolic representation. arXiv preprint arXiv:2010.00823","author":"Kim Sunghyeon","year":"2020","unstructured":"Sunghyeon Kim, Hyeyoon Lee, Sunjong Park, Jinho Lee, and Keunwoo Choi. 2020. Deep composer classification using symbolic representation. arXiv preprint arXiv:2010.00823 (2020)."},{"key":"e_1_3_2_1_25_1","volume-title":"Large-scale midi-based composer classification. arXiv preprint arXiv:2010.14805","author":"Kong Qiuqiang","year":"2020","unstructured":"Qiuqiang Kong, Keunwoo Choi, and Yuxuan Wang. 2020. Large-scale midi-based composer classification. arXiv preprint arXiv:2010.14805 (2020)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3121991"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i4.25650"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3571294"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3277290"},{"key":"e_1_3_2_1_30_1","volume-title":"An Optimized Method for Large-Scale Pre-Training in Symbolic Music. In 2022 IEEE 16th International Conference on Anti-counterfeiting, Security, and Identification (ASID). IEEE, 105-109","author":"Liu Shike","year":"2022","unstructured":"Shike Liu, Hongguang Xu, and Ke Xu. 2022. An Optimized Method for Large-Scale Pre-Training in Symbolic Music. In 2022 IEEE 16th International Conference on Anti-counterfeiting, Security, and Identification (ASID). IEEE, 105-109."},{"key":"e_1_3_2_1_31_1","volume-title":"Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101","author":"Loshchilov I","year":"2017","unstructured":"I Loshchilov. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)."},{"key":"e_1_3_2_1_32_1","volume-title":"Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781","author":"Mikolov Tomas","year":"2013","unstructured":"Tomas Mikolov, Kai Chen, Greg Corrado, and Jeffrey Dean. 2013a. Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781 (2013)."},{"key":"e_1_3_2_1_33_1","volume-title":"Distributed representations of words and phrases and their compositionality. Advances in neural information processing systems","author":"Mikolov Tomas","year":"2013","unstructured":"Tomas Mikolov, Ilya Sutskever, Kai Chen, Greg S Corrado, and Jeff Dean. 2013b. Distributed representations of words and phrases and their compositionality. Advances in neural information processing systems, Vol. 26 (2013)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-018-3758-9"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096516"},{"key":"e_1_3_2_1_36_1","first-page":"383","volume-title":"19th International Society for Music Information Retrieval Conference (ISMIR","author":"Panda Renato","year":"2018","unstructured":"Renato Panda, Ricardo Malheiro, and Rui Pedro Paiva. 2018. Musical texture and expressivity features for music emotion recognition. In 19th International Society for Music Information Retrieval Conference (ISMIR 2018). 383-391."},{"key":"e_1_3_2_1_37_1","volume-title":"A novel multi-task learning method for symbolic music emotion recognition. arXiv preprint arXiv:2201.05782","author":"Qiu Jibao","year":"2022","unstructured":"Jibao Qiu, CL Chen, and Tong Zhang. 2022. A novel multi-task learning method for symbolic music emotion recognition. arXiv preprint arXiv:2201.05782 (2022)."},{"key":"e_1_3_2_1_38_1","unstructured":"Alec Radford Karthik Narasimhan Tim Salimans Ilya Sutskever et al. 2018. Improving language understanding by generative pre-training. (2018)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2013.2271648"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/E17-2043"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-70542-0_5"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i15.17626"},{"key":"e_1_3_2_1_43_1","volume-title":"A convolutional approach to melody line identification in symbolic scores. arXiv preprint arXiv:1906.10547","author":"Simonetta Federico","year":"2019","unstructured":"Federico Simonetta, Carlos Cancino-Chac\u00f3n, Stavros Ntalampiras, and Gerhard Widmer. 2019. A convolutional approach to melody line identification in symbolic scores. arXiv preprint arXiv:1906.10547 (2019)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i14.29461"},{"key":"e_1_3_2_1_46_1","volume-title":"Composer style classification of piano sheet music images using language model pretraining. arXiv preprint arXiv:2007.14587","author":"Tsai TJ","year":"2020","unstructured":"TJ Tsai and Kevin Ji. 2020. Composer style classification of piano sheet music images using language model pretraining. arXiv preprint arXiv:2007.14587 (2020)."},{"key":"e_1_3_2_1_47_1","volume-title":"Attention is all you need. Advances in Neural Information Processing Systems","author":"Vaswani A","year":"2017","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/2393347.2393368"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654940"},{"key":"e_1_3_2_1_50_1","volume-title":"Pop909: A pop-song dataset for music arrangement generation. arXiv preprint arXiv:2008.07142","author":"Wang Ziyu","year":"2020","unstructured":"Ziyu Wang, Ke Chen, Junyan Jiang, Yiyi Zhang, Maoran Xu, Shuqi Dai, Xianbin Gu, and Gus Xia. 2020. Pop909: A pop-song dataset for music arrangement generation. arXiv preprint arXiv:2008.07142 (2020)."},{"key":"e_1_3_2_1_51_1","volume-title":"Huggingface's transformers: State-of-the-art natural language processing. arXiv preprint arXiv:1910.03771","author":"Wolf T","year":"2019","unstructured":"T Wolf. 2019. Huggingface's transformers: State-of-the-art natural language processing. arXiv preprint arXiv:1910.03771 (2019)."},{"key":"e_1_3_2_1_52_1","volume-title":"International conference on machine learning. PMLR, 11863-11874","author":"Yang Lingxiao","year":"2021","unstructured":"Lingxiao Yang, Ru-Yuan Zhang, Lida Li, and Xiaohua Xie. 2021. Simam: A simple, parameter-free attention module for convolutional neural networks. In International conference on machine learning. PMLR, 11863-11874."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Yi-Hsuan Yang and Homer H Chen. 2011. Music emotion recognition. (2011).","DOI":"10.1201\/b10731"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681288"},{"key":"e_1_3_2_1_55_1","volume-title":"Musicbert: Symbolic music understanding with large-scale pre-training. arXiv preprint arXiv:2106.05630","author":"Zeng Mingliang","year":"2021","unstructured":"Mingliang Zeng, Xu Tan, Rui Wang, Zeqian Ju, Tao Qin, and Tie-Yan Liu. 2021. Musicbert: Symbolic music understanding with large-scale pre-training. arXiv preprint arXiv:2106.05630 (2021)."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6510"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754590","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:25:42Z","timestamp":1765308342000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754590"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":56,"alternative-id":["10.1145\/3746027.3754590","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754590","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}