{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T02:02:42Z","timestamp":1782352962908,"version":"3.54.5"},"reference-count":47,"publisher":"Tsinghua University Press","issue":"4","license":[{"start":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T00:00:00Z","timestamp":1722470400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2024,7,24]],"date-time":"2024-07-24T00:00:00Z","timestamp":1721779200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Comp. Visual. Med."],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1007\/s41095-024-0417-1","type":"journal-article","created":{"date-parts":[[2024,7,23]],"date-time":"2024-07-23T22:05:36Z","timestamp":1721772336000},"page":"791-802","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Dance2MIDI: Dance-driven multi-instrument music generation"],"prefix":"10.26599","volume":"10","author":[{"given":"Bo","family":"Han","sequence":"first","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou 310058, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuheng","family":"Li","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou 310058, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yixuan","family":"Shen","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore 119077, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Ren","sequence":"additional","affiliation":[{"name":"Speech &#x0026; Audio Team, Bytedance AI Lab, Singapore 048583, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feilin","family":"Han","sequence":"additional","affiliation":[{"name":"Department of Film and TV Technology, Beijing Film Academy, Beijing 100088, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"11138","reference":[{"issue":"1","key":"417_CR1","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1145\/602421.602425","volume":"46","author":"M Cannataro","year":"2003","unstructured":"Cannataro, M.; Talia, D. The knowledge grid. Communications of the ACM Vol. 46, No. 1, 89\u201393, 2003.","journal-title":"Communications of the ACM"},{"issue":"8","key":"417_CR2","doi-asserted-by":"publisher","first-page":"1235","DOI":"10.1016\/j.future.2005.06.001","volume":"21","author":"C T D V O Mastroianni","year":"2005","unstructured":"Mastroianni, C.; Talia, D.; Verta, O. A super-peer model for resource discovery services in large-scale grids. Future Generation Computer Systems Vol. 21, No. 8, 1235\u20131248, 2005.","journal-title":"Future Generation Computer Systems"},{"key":"417_CR3","unstructured":"Aggarwal, G.; Parikh, D. Dance2Music: Automatic dance-driven music generation. arXiv preprint arXiv:2107.06252, 2021."},{"key":"417_CR4","doi-asserted-by":"crossref","unstructured":"Di, S.; Jiang, Z.; Liu, S.; Wang, Z.; Zhu, L.; He, Z.; Liu, H.; Yan, S. Video background music generation with controllable music transformer. In: Proceedings of the 29th ACM International Conference on Multimedia, 2037\u20132045, 2021.","DOI":"10.1145\/3474085.3475195"},{"key":"417_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"758","DOI":"10.1007\/978-3-030-58621-8_44","volume-title":"Computer Vision\u2013ECCV 2020","author":"C Gan","year":"2020","unstructured":"Gan, C.; Huang, D.; Chen, P.; Tenenbaum, J. B.; Torralba, A. Foley music: Learning to generate music from videos. In: Computer Vision\u2013ECCV 2020. Lecture Notes in Computer Science, Vol. 12356. Vedaldi, A.; Bischof, H.; Brox, T.; Frahm, J. Eds. Springer Cham, 758\u2013775, 2020."},{"key":"417_CR6","doi-asserted-by":"crossref","unstructured":"Kao, H. K.; Su, L. Temporally guided music-to-body-movement generation. In: Proceedings of the 28th ACM International Conference on Multimedia, 147\u2013155, 2020.","DOI":"10.1145\/3394171.3413848"},{"key":"417_CR7","doi-asserted-by":"crossref","unstructured":"Li, R.; Yang, S.; Ross, D. A.; Kanazawa, A. AI choreographer: Music conditioned 3D dance generation with AIST. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 13401\u201313412, 2021.","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"417_CR8","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"182","DOI":"10.1007\/978-3-031-19836-6_11","volume-title":"Computer Vision\u2013ECCV 2022","author":"Y Zhu","year":"2022","unstructured":"Zhu, Y.; Olszewski, K.; Wu, Y.; Achlioptas, P.; Chai, M.; Yan, Y.; Tulyakov, S. Quantized GAN for complex music generation from dance videos. In: Computer Vision\u2013ECCV 2022. Lecture Notes in Computer Science, Vol. 13697. Avidan, S.; Brostow, G.; Ciss\u00e9, M.; Farinella, G. M.; Hassner, T. Eds. Springer Cham, 182\u2013199, 2022."},{"key":"417_CR9","unstructured":"Han, B.; Peng, H.; Dong, M.; Ren, Y.; Shen, Y.; Xu, C. AMD: Autoregressive motion diffusion. arXiv preprint arXiv:2305.09381, 2023."},{"key":"417_CR10","doi-asserted-by":"crossref","unstructured":"Kim, J.; Oh, H.; Kim, S.; Tong, H.; Lee, S. A brand new dance partner: Music-conditioned pluralistic dancing controlled by multiple dance genres. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 3490\u20133500, 2022.","DOI":"10.1109\/CVPR52688.2022.00348"},{"key":"417_CR11","unstructured":"Lee, H.; Yang, X.; Liu, M.; Wang, T.; Lu, Y.; Yang, M.; Kautz, J. Dancing to music. In: Proceedings of the 33rd Conference on Neural Information Processing Systems, 3581\u20133591, 2019."},{"key":"417_CR12","doi-asserted-by":"crossref","unstructured":"Li, B.; Zhao, Y.; Shi, Z.; Sheng, L. DanceFormer: Music conditioned 3D dance generation with parametric motion transformer. In: Proceedings of the 36th AAAI Conference on Artificial Intelligence, 1272\u20131279, 2022.","DOI":"10.1609\/aaai.v36i2.20014"},{"key":"417_CR13","unstructured":"Li, S.; Yu, W.; Gu, T.; Lin, C.; Wang, Q.; Qian, C.; Loy, C. C.; Liu, Z. Bailando: 3D dance generation by actor-critic GPT with choreographic memory. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 11050\u201311059, 2022."},{"issue":"4","key":"417_CR14","doi-asserted-by":"publisher","first-page":"981","DOI":"10.1007\/s00521-018-3813-6","volume":"32","author":"J P Briot","year":"2020","unstructured":"Briot, J. P.; Pachet, F. Deep learning for music generation: Challenges and directions. Neural Computing and Applications Vol. 32, No. 4, 981\u2013993, 2020.","journal-title":"Neural Computing and Applications"},{"key":"417_CR15","unstructured":"Ji, S.; Luo, J.; Yang, X. A comprehensive survey on deep music generation: Multi-level representations, algorithms, evaluations, and future directions. arXiv preprint arXiv:2011.06801, 2020."},{"key":"417_CR16","unstructured":"Su, K.; Liu, X.; Shlizerman, E. How does it sound? In: Proceedings of the 35th Conference on Neural Information Processing Systems, 29258\u201329273, 2021."},{"key":"417_CR17","unstructured":"Wang, Z.; Ma, L.; Zhang, C.; Han, B.; Xu, Y.; Wang, Y.; Chen, X.; Hong, H.; Liu, W.; Wu, X.; et al. REMAST: Real-time emotion-based music arrangement with soft transition. arXiv preprint arXiv:2305.08029, 2023."},{"key":"417_CR18","doi-asserted-by":"crossref","unstructured":"Yan, S.; Xiong, Y.; Lin, D. Spatial temporal graph convolutional networks for skeleton-based action recognition. In: Proceedings of the 32nd AAAI Conference on Article Intelligence, 7444\u20137452, 2018.","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"417_CR19","unstructured":"Devlin, J.; Chang, M.; Lee, K.; Toutanova, K. BERT: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, 4171\u20134186, 2019."},{"key":"417_CR20","unstructured":"Engel, J.; Agrawal, K. K.; Chen, S.; Gulrajani, I.; Donahue, C.; Roberts, A. GANSynth: Adversarial neural audio synthesis. arXiv preprint arXiv: 1902.08710, 2019."},{"key":"417_CR21","unstructured":"Goel, K.; Gu, A.; Donahue, C.; R\u00e9, C. It\u2019s raw! Audio generation with state-space models. arXiv preprint arXiv:2202.09729, 2022."},{"key":"417_CR22","unstructured":"Van den Oord, A.; Dieleman, S.; Zen, H.; Simonyan, K.; Vinyals, O.; Graves, A.; Kalchbrenner, N.; Senior, A.; Kavukcuoglu, K. WaveNet: A generative model for raw audio. arXiv preprint arXiv:1609.03499, 2016."},{"key":"417_CR23","unstructured":"Dhariwal, P.; Jun, H.; Payne, C.; Kim, J. W.; Radford, A.; Sutskever, I. Jukebox: A generative model for music. arXiv preprint arXiv:2005.00341, 2020."},{"key":"417_CR24","unstructured":"Kumar, K.; Kumar, R.; de Boissiere, T.; Gestin, L.; Teoh, W. Z.; Sotelo, J.; de Brebisson, A.; Bengio, Y.; Courville, A. MelGAN: Generative adversarial networks for conditional waveform synthesis. In: Proceedings of the 33rd Conference on Neural Information Procesing Systems, 14881\u201314892, 2019."},{"key":"417_CR25","unstructured":"Vasquez, S.; Lewis, M. MelNet: A generative model for audio in the frequency domain. arXiv preprint arXiv:1906.01083, 2019."},{"key":"417_CR26","doi-asserted-by":"crossref","unstructured":"Dong, H. W.; Hsiao, W. Y.; Yang, L. C.; Yang, Y. H. MuseGAN: Multi-track sequential generative adversarial networks for symbolic music generation and accompaniment. In: Proceedings of the 32nd AAAI Conference on Artificial Intelligence, 34\u201341, 2018.","DOI":"10.1609\/aaai.v32i1.11312"},{"key":"417_CR27","unstructured":"Huang, C. Z. A.; Vaswani, A.; Uszkoreit, J.; Shazeer, N.; Simon, I.; Hawthorne, C.; Dai, A. M.; Hoffman, M. D.; Dinculescu, M.; Eck, D. Music transformer. arXiv preprint arXiv:1809.04281, 2018."},{"key":"417_CR28","doi-asserted-by":"crossref","unstructured":"Muhamed, A.; Li, L.; Shi, X.; Yaddanapudi, S.; Chi, W.; Jackson, D.; Suresh, R.; Lipton, Z. C.; Smola, A. J. Symbolic music generation with transformer-GANs. In: Proceedings of the 35th AAAI Conference on Artificial Intelligence, 408\u2013417, 2021.","DOI":"10.1609\/aaai.v35i1.16117"},{"key":"417_CR29","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1162\/tacl_a_00300","volume":"8","author":"M Joshi","year":"2020","unstructured":"Joshi, M.; Chen, D.; Liu, Y.; Weld, D. S.; Zettlemoyer, L.; Levy, O. SpanBERT: Improving pre-training by representing and predicting spans. Transactions of the Association for Computational Linguistics Vol. 8, 64\u201377, 2020.","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"417_CR30","doi-asserted-by":"crossref","unstructured":"Ren, Y.; He, J.; Tan, X.; Qin, T.; Zhao, Z.; Liu, T. Y. PopMAG: Pop music accompaniment generation. In: Proceedings of the 28th ACM International Conference on Multimedia, 1198\u20131206, 2020.","DOI":"10.1145\/3394171.3413721"},{"key":"417_CR31","unstructured":"Liu, J.; Dong, Y.; Cheng, Z.; Zhang, X.; Li, X.; Yu, F.; Sun, M. Symphony generation with permutation invariant language model. In: Proceedings of the 24th International Society for Music Information Retrieval Conference, 551\u2013558, 2022."},{"key":"417_CR32","unstructured":"Pedersoli, F.; Goto, M. Dance beat tracking from visual information alone. In: Proceedings of the 21st International Society for Music Information Retrieval Conference, 400\u2013408, 2020."},{"key":"417_CR33","unstructured":"Gillick, J.; Roberts, A.; Engel, J.; Eck, D.; Bamman, D. Learning to groove with inverse sequence transformations. In: Proceedings of the 36th International Conference on Machine Learning, 2269\u20132279, 2019."},{"key":"417_CR34","doi-asserted-by":"crossref","unstructured":"Raffel, C. Learning-based methods for comparing sequences, with applications to Audio-to-MIDI alignment and matching. Ph.D. Thesis. Columbia University, 2016.","DOI":"10.1109\/ICASSP.2016.7471641"},{"key":"417_CR35","unstructured":"Hawthorne, C.; Stasyuk, A.; Roberts, A.; Simon, I.; Huang, C. Z. A.; Dieleman, S.; Elsen, E.; Engel, J.; Eck, D. Enabling factorized piano music modeling and generation with the MAESTRO dataset. arXiv preprint arXiv:1810.12247, 2018."},{"key":"417_CR36","doi-asserted-by":"crossref","unstructured":"Ferreira, L. N.; Lelis, L. H. S.; Whitehead, J. Computer-generated music for tabletop role-playing games. In: Proceedings of the 16th AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment, 59\u201365, 2020.","DOI":"10.1609\/aiide.v16i1.7408"},{"key":"417_CR37","unstructured":"Tsuchida, S.; Fukayama, S.; Hamasaki, M.; Goto, M. AIST dance video database: Multi-genre, multi-dancer, and multi-camera database for dance information processing. In: Proceedings of the 20th International Society for Music Information Retrieval Conference, 501\u2013510, 2019."},{"key":"417_CR38","unstructured":"Gardner, J.; Simon, I.; Manilow, E.; Hawthorne, C.; Engel, J. MT3: Multi-task multitrack music transcription. arXiv preprint arXiv:2111.03017, 2021."},{"key":"417_CR39","doi-asserted-by":"crossref","unstructured":"Cao, Z.; Simon, T.; Wei, S. E.; Sheikh, Y. Realtime multi-person 2D pose estimation using part affinity fields. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 7291\u20137299, 2017.","DOI":"10.1109\/CVPR.2017.143"},{"key":"417_CR40","doi-asserted-by":"crossref","unstructured":"Sun, K.; Xiao, B.; Liu, D.; Wang, J. Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 5693\u20135703, 2019.","DOI":"10.1109\/CVPR.2019.00584"},{"key":"417_CR41","unstructured":"Lugaresi, C.; Tang, J.; Nash, H.; McClanahan, C.; Uboweja, E.; Hays, M.; Zhang, F.; Chang, C. L.; Yong, M. G.; Lee, J.; et al. MediaPipe: A framework for building perception pipelines. arXiv preprint arXiv:1906.08172, 2019."},{"key":"417_CR42","doi-asserted-by":"crossref","unstructured":"Chen, K.; Tan, Z.; Lei, J.; Zhang, S. H.; Guo, Y. C.; Zhang, W.; Hu, S. M. ChoreoMaster: Choreography-oriented music-driven dance synthesis. ACM Transactions on Graphics Vol. 40, No. 4, Article No. 145, 2021.","DOI":"10.1145\/3476576.3476724"},{"key":"417_CR43","doi-asserted-by":"crossref","unstructured":"Chen, C. F. R.; Fan, Q.; Panda, R. CrossViT: Cross-attention multi-scale vision transformer for image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 357\u2013366, 2021.","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"417_CR44","doi-asserted-by":"crossref","unstructured":"Davis, A.; Agrawala, M. Visual rhythm and beat. ACM Transactions on Graphics Vol. 37, No. 4, Article No. 122, 2018.","DOI":"10.1145\/3197517.3201371"},{"key":"417_CR45","unstructured":"Wu, S. L.; Yang, Y. H. The jazz transformer on the front line: Exploring the shortcomings of AI-composed music through quantitative measures. In: Proceedings of the 21st International Society for Music Information Retrieval Conference, 142\u2013149, 2020."},{"issue":"3","key":"417_CR46","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1002\/j.1538-7305.1948.tb01338.x","volume":"27","author":"C E Shannon","year":"1948","unstructured":"Shannon, C. E. A mathematical theory of communication. The Bell System Technical Journal Vol. 27, No. 3, 379\u2013423, 1948.","journal-title":"The Bell System Technical Journal"},{"key":"417_CR47","doi-asserted-by":"publisher","first-page":"351","DOI":"10.1007\/978-1-4842-2496-0_20","volume-title":"Linux Sound Programming","author":"J Newmarch","year":"2017","unstructured":"Newmarch, J. FluidSynth. In: Linux Sound Programming. Newmarch, CA, USA: Apress, 351\u2013353, 2017."}],"container-title":["Computational Visual Media"],"original-title":[],"link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41095-024-0417-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s41095-024-0417-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41095-024-0417-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10750449\/10897710\/10897723.pdf?arnumber=10897723","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T18:38:53Z","timestamp":1762367933000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10897723\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8]]},"references-count":47,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1007\/s41095-024-0417-1","relation":{},"ISSN":["2096-0662","2096-0433"],"issn-type":[{"value":"2096-0662","type":"electronic"},{"value":"2096-0433","type":"print"}],"subject":[],"published":{"date-parts":[[2024,8]]},"assertion":[{"value":"4 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 July 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration of competing interest"}}]}}