{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T02:02:42Z","timestamp":1782352962914,"version":"3.54.5"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61702466"],"award-info":[{"award-number":["61702466"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["CUC230B028"],"award-info":[{"award-number":["CUC230B028"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["CUCAI24002"],"award-info":[{"award-number":["CUCAI24002"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J AUDIO SPEECH MUSIC PROC."],"DOI":"10.1186\/s13636-024-00370-6","type":"journal-article","created":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T13:02:43Z","timestamp":1727701363000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Dance2Music-Diffusion: leveraging latent diffusion models for music generation from dance videos"],"prefix":"10.1186","volume":"2024","author":[{"given":"Chaoyang","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5845-1857","authenticated-orcid":false,"given":"Yan","family":"Hua","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,9,30]]},"reference":[{"key":"370_CR1","unstructured":"H.Y. Lee, X. Yang, M.Y. Liu, T.C. Wang, Y.D. Lu, M.H. Yang, J. Kautz, Dancing to music. Adv. Neural Inform. Process. Syst. 32, 3586\u20133596 (2019)"},{"key":"370_CR2","doi-asserted-by":"crossref","unstructured":"R. Li, S. Yang, D.A. Ross, A. Kanazawa, in Proceedings of the IEEE\/CVF International Conference on Computer Vision. Ai choreographer: Music conditioned 3d dance generation with aist++ (IEEE, Piscataway, NJ, 2021), pp. 13401\u201313412","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"370_CR3","unstructured":"G. Aggarwal, D. Parikh, Dance2music: Automatic dance-driven music generation. arXiv preprint arXiv:2107.06252 (2021)"},{"key":"370_CR4","doi-asserted-by":"crossref","unstructured":"C. Gan, D. Huang, P. Chen, J.B. Tenenbaum, A. Torralba, in European Conference on Computer Vision. Foley music: Learning to generate music from videos (Springer, Cham, 2020), pp. 758\u2013775","DOI":"10.1007\/978-3-030-58621-8_44"},{"key":"370_CR5","doi-asserted-by":"crossref","unstructured":"H.K. Kao, L. Su, in Proceedings of the 28th ACM International Conference on Multimedia. Temporally guided music-to-body-movement generation (ACM, New York, 2020), pp. 147\u2013155","DOI":"10.1145\/3394171.3413848"},{"key":"370_CR6","doi-asserted-by":"crossref","unstructured":"B. Han, Y. Ren, Y. Li, Dance2midi: Dance-driven multi-instruments music generation. arXiv preprint arXiv:2301.09080 (2023)","DOI":"10.1007\/s41095-024-0417-1"},{"issue":"4","key":"370_CR7","doi-asserted-by":"publisher","first-page":"8","DOI":"10.2307\/3679619","volume":"9","author":"G Loy","year":"1985","unstructured":"G. Loy, Musicians make a standard: The midi phenomenon. Comput. Music. J. 9(4), 8\u201326 (1985)","journal-title":"Comput. Music. J."},{"key":"370_CR8","doi-asserted-by":"crossref","unstructured":"Y. Zhu, K. Olszewski, Y. Wu, P. Achlioptas, M. Chai, Y. Yan, S. Tulyakov, in European Conference on Computer Vision. Quantized gan for complex music generation from dance videos (Springer, Cham, 2022), pp. 182\u2013199","DOI":"10.1007\/978-3-031-19836-6_11"},{"key":"370_CR9","unstructured":"S. Li, W. Dong, Y. Zhang, F. Tang, C. Ma, O. Deussen, T.Y. Lee, C. Xu, Dance-to-music generation with encoder-based textual inversion of diffusion models. arXiv preprint arXiv:2401.17800 (2024)"},{"key":"370_CR10","doi-asserted-by":"crossref","unstructured":"V. Tan, J. Nam, J. Nam, J. Noh, in SIGGRAPH Asia 2023 Technical Communications. Motion to dance music generation using latent diffusion mode (ACM, New York, 2023), pp. 1\u20134","DOI":"10.1145\/3610543.3626164"},{"key":"370_CR11","doi-asserted-by":"crossref","unstructured":"R. Rombach, A. Blattmann, D. Lorenz, P. Esser, B. Ommer, in Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. High-resolution image synthesis with latent diffusion models (IEEE, Piscataway, NJ, 2022), pp. 10684\u201310695","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"370_CR12","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"P. Dhariwal, A. Nichol, Diffusion models beat gans on image synthesis. Adv. Neural Inform. Process. Syst. 34, 8780\u20138794 (2021)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"370_CR13","doi-asserted-by":"crossref","unstructured":"L. Zhang, A. Rao, M. Agrawala, in Proceedings of the IEEE\/CVF International Conference on Computer Vision. Adding conditional control to text-to-image diffusion models (IEEE, Piscataway, NJ, 2022), pp. 3836\u20133847","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"370_CR14","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"C. Saharia, W. Chan, S. Saxena, L. Li, J. Whang, E.L. Denton, K. Ghasemipour, R. Gontijo Lopes, B. Karagol Ayan, T. Salimans et al., Photorealistic text-to-image diffusion models with deep language understanding. Adv. Neural Inform. Process. Syst. 35, 36479\u201336494 (2022)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"370_CR15","unstructured":"A. Razavi, A. Van den Oord, O. Vinyals, Generating diverse high-fidelity images with vq-vae-2. Adv. Neural Inform. Process. Syst. 32, 14866\u201314876 (2019)"},{"key":"370_CR16","unstructured":"F. Schneider, O. Kamal, Z. Jin, B. Sch\u00f6lkopf, Mo\\^ usai: Text-to-music generation with long-context latent diffusion. arXiv preprint arXiv:2301.11757 (2023)"},{"key":"370_CR17","unstructured":"Q. Huang, D.S. Park, T. Wang, T.I. Denk, A. Ly, N. Chen, Z. Zhang, Z. Zhang, J. Yu, C. Frank et al., Noise2music: Text-conditioned music generation with diffusion models. arXiv preprint arXiv:2302.03917 (2023)"},{"key":"370_CR18","unstructured":"A. Agostinelli, T.I. Denk, Z. Borsos, J. Engel, M. Verzetti, A. Caillon, Q. Huang, A. Jansen, A. Roberts, M. Tagliasacchi et al., Musiclm: Generating music from text. arXiv preprint arXiv:2301.11325 (2023)"},{"key":"370_CR19","unstructured":"H. Liu, Z. Chen, Y. Yuan, X. Mei, X. Liu, D. Mandic, W. Wang, M.D. Plumbley, Audioldm: Text-to-audio generation with latent diffusion models. H. Liu, Z. Chen, Y. Yuan et al., AudioLDM: text-to-audio generation with latent diffusion models, in Proceedings of the 40th International Conference on Machine Learning (PMLR, Brookline, MA, 2023), pp. 21450\u201321474"},{"key":"370_CR20","unstructured":"M. Morrison, R. Kumar, K. Kumar, P. Seetharaman, A. Courville, Y. Bengio, Chunked autoregressive gan for conditional waveform synthesis. arXiv preprint arXiv:2110.10139 (2021)"},{"key":"370_CR21","unstructured":"C. Donahue, J. McAuley, M. Puckette, Synthesizing audio with generative adversarial networks. arXiv preprint arXiv:1802.04208 1 (2018)"},{"key":"370_CR22","unstructured":"L.C. Yang, S.Y. Chou, Y.H. Yang, Midinet: A convolutional generative adversarial network for symbolic-domain music generation. arXiv preprint arXiv:1703.10847 (2017)"},{"key":"370_CR23","unstructured":"K. Deng, A. Bansal, D. Ramanan, Unsupervised audiovisual synthesis via exemplar autoencoders. arXiv preprint arXiv:2001.04463 (2020)"},{"key":"370_CR24","unstructured":"P. Dhariwal, H. Jun, C. Payne, J.W. Kim, A. Radford, I. Sutskever, Jukebox: A generative model for music. arXiv preprint arXiv:2005.00341 (2020)"},{"key":"370_CR25","first-page":"1376","volume":"35","author":"B Yu","year":"2022","unstructured":"B. Yu, P. Lu, R. Wang, W. Hu, X. Tan, W. Ye, S. Zhang, T. Qin, T.Y. Liu, Museformer: Transformer with fine-and coarse-grained attention for music generation. Adv. Neural Inform. Process. Syst. 35, 1376\u20131388 (2022)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"370_CR26","unstructured":"J. Ens, P. Pasquier, Mmm: Exploring conditional multi-track music generation with the transformer. arXiv preprint arXiv:2008.06048 (2020)"},{"key":"370_CR27","doi-asserted-by":"crossref","unstructured":"Y.J. Shih, S.L. Wu, F. Zalkow, M. M\u00fcller, Y.H. Yang, Theme transformer: Symbolic music generation with theme-conditioned transformer. IEEE Trans. Multimedia 25, 3495\u20133508 (2022)","DOI":"10.1109\/TMM.2022.3161851"},{"key":"370_CR28","unstructured":"Y. Zhu, Y. Wu, K. Olszewski, J. Ren, S. Tulyakov, Y. Yan, Discrete contrastive diffusion for cross-modal music and image generation. arXiv preprint arXiv:2206.07771 (2022)"},{"key":"370_CR29","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, \u0141. Kaiser, I. Polosukhin, Attention is all you need. Adv. Neural Inform. Process. Syst. 30, 5998\u20136008 (2017)"},{"key":"370_CR30","doi-asserted-by":"crossref","unstructured":"M. Loper, N. Mahmood, J. Romero, G. Pons-Moll, M.J. Black, in Seminal Graphics Papers: Pushing the Boundaries. Smpl: A skinned multi-person linear model, vol. 2 (ACM, New York, 2023), pp. 851\u2013866","DOI":"10.1145\/3596711.3596800"},{"key":"370_CR31","unstructured":"J. Devlin, M.W. Chang, K. Lee, K. Toutanova, Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"370_CR32","unstructured":"J. Song, C. Meng, S. Ermon, Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502 (2020)"},{"key":"370_CR33","unstructured":"T. Salimans, J. Ho, Progressive distillation for fast sampling of diffusion models. arXiv preprint arXiv:2202.00512 (2022)"},{"key":"370_CR34","unstructured":"F. Schneider, Archisound: Audio generation with diffusion. arXiv preprint arXiv:2301.13267 (2023)"},{"key":"370_CR35","unstructured":"I. Loshchilov, F. Hutter, Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"370_CR36","doi-asserted-by":"crossref","unstructured":"S. Di, Z. Jiang, S. Liu, Z. Wang, L. Zhu, Z. He, H. Liu, S. Yan, in Proceedings of the 29th ACM International Conference on Multimedia. Video background music generation with controllable music transformer (ACM, New York,  2021), pp. 2037\u20132045","DOI":"10.1145\/3474085.3475195"},{"key":"370_CR37","doi-asserted-by":"crossref","unstructured":"B. McFee, C. Raffel, D. Liang, D.P. Ellis, M. McVicar, E. Battenberg, O. Nieto, in SciPy. librosa: Audio and music signal analysis in python (SciPy, Austin, TX, 2015), pp. 18\u201324","DOI":"10.25080\/Majora-7b98e3ed-003"},{"key":"370_CR38","unstructured":"S. Tsuchida, S. Fukayama, M. Hamasaki, M. Goto, in ISMIR. Aist dance video database: Multi-genre, multi-dancer, and multi-camera database for dance information processing, vol. 1 (ISMIR, Delft, Netherlands, 2019), p. 6"}],"container-title":["EURASIP Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-024-00370-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13636-024-00370-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-024-00370-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T13:08:43Z","timestamp":1727701723000},"score":1,"resource":{"primary":{"URL":"https:\/\/asmp-eurasipjournals.springeropen.com\/articles\/10.1186\/s13636-024-00370-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,30]]},"references-count":38,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2024,12]]}},"alternative-id":["370"],"URL":"https:\/\/doi.org\/10.1186\/s13636-024-00370-6","relation":{},"ISSN":["1687-4722"],"issn-type":[{"value":"1687-4722","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,30]]},"assertion":[{"value":"21 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 September 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"48"}}