{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T06:17:17Z","timestamp":1782800237754,"version":"3.54.5"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T00:00:00Z","timestamp":1713052800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Key Areas Research and Development Program of Guangzhou","award":["2023B01J0029"],"award-info":[{"award-number":["2023B01J0029"]}]},{"name":"Science and technology research in key areas in Foshan","award":["2020001006832"],"award-info":[{"award-number":["2020001006832"]}]},{"DOI":"10.13039\/501100015956","name":"Special Project for Research and Development in Key areas of Guangdong Province","doi-asserted-by":"publisher","award":["2018B010109007"],"award-info":[{"award-number":["2018B010109007"]}],"id":[{"id":"10.13039\/501100015956","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Science and technology projects of Guangzhou","award":["202007040006"],"award-info":[{"award-number":["202007040006"]}]},{"name":"Guangdong Provincial Key Laboratory of Cyber-Physical System","award":["2020B1212060069"],"award-info":[{"award-number":["2020B1212060069"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation","doi-asserted-by":"crossref","award":["92267107"],"award-info":[{"award-number":["92267107"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100012245","name":"Science and Technology Planning Project of Guangdong Province","doi-asserted-by":"publisher","award":["2021B0101220006"],"award-info":[{"award-number":["2021B0101220006"]}],"id":[{"id":"10.13039\/501100012245","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Science and Technology Projects in Guangzhou","award":["202201011706"],"award-info":[{"award-number":["202201011706"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s00371-024-03376-5","type":"journal-article","created":{"date-parts":[[2024,4,14]],"date-time":"2024-04-14T13:01:26Z","timestamp":1713099686000},"page":"961-973","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["QEAN: quaternion-enhanced attention network for visual dance generation"],"prefix":"10.1007","volume":"41","author":[{"given":"Zhizhen","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yejing","family":"Huo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guoheng","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"An","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuhang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lian","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zinuo","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,4,14]]},"reference":[{"key":"3376_CR1","doi-asserted-by":"publisher","first-page":"315","DOI":"10.1590\/0101-3173.2023.v46n4.p315","volume":"46","author":"Y Yang","year":"2023","unstructured":"Yang, Y., Zhang, E.: Cultural thought and philosophical elements of singing and dancing in Indian films. Trans\/Form\/A \u00e7 \u00e3 o 46, 315\u2013328 (2023). https:\/\/doi.org\/10.1590\/0101-3173.2023.v46n4.p315","journal-title":"Trans\/Form\/A \u00e7 \u00e3 o"},{"key":"3376_CR2","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1080\/08963568.2017.1285747","volume":"22","author":"M Siciliano","year":"2017","unstructured":"Siciliano, M.: A citation analysis of business librarianship: examining the Journal of Business and Finance Librarianship from 1990\u20132014. J. Bus. Finance Librariansh. 22, 81\u201396 (2017)","journal-title":"J. Bus. Finance Librariansh."},{"key":"3376_CR3","doi-asserted-by":"publisher","first-page":"1725","DOI":"10.1007\/s00371-017-1452-z","volume":"34","author":"A Aristidou","year":"2018","unstructured":"Aristidou, A., Stavrakis, E., Papaefthimiou, M., Papagiannakis, G., Chrysanthou, Y.: Style-based motion analysis for dance composition. Vis. Comput. 34, 1725\u20131737 (2018)","journal-title":"Vis. Comput."},{"key":"3376_CR4","unstructured":"Li, Ji., Yin, Y., Chu, H., Zhou, Y., Wang, T., Fidler, S., Li, H.: Learning to generate diverse dance motions with transformer. In: arXiv:2008.08171. https:\/\/api.semanticscholar.org\/CorpusID:221173065 (2020)"},{"key":"3376_CR5","unstructured":"Huang, R., Hu, H., Wu, W., Sawada, K., Zhang, M., Jiang, D.: Dance revolution: long-term dance generation with music via curriculum learning. In: International Conference on Learning Representations. https:\/\/api.semanticscholar.org\/CorpusID:235614403 (2020)"},{"key":"3376_CR6","unstructured":"Zhang, X., Xu, Y., Yang, S., Gao, L., Sun, H.: Dance generation with style embedding: learning and transferring latent representations of dance styles. In: arXiv:1041.4802. https:\/\/api.semanticscholar.org\/CorpusID:233476346 (2021)"},{"key":"3376_CR7","unstructured":"Huang, R., Hu, H., Wu, W., Sawada, K., Zhang, M., Jiang, D.: Dance revolution: long-term dance generation with music via curriculum learning. In: International conference on learning representations. https:\/\/api.semanticscholar.org\/CorpusID:235614403 (2020)"},{"key":"3376_CR8","unstructured":"Bengio, S., Vinyals, O., Jaitly, N., Shazeer, N.M.: Scheduled sampling for sequence prediction with recurrent neural networks. In: arXiv:1506.03099. https:\/\/api.semanticscholar.org\/CorpusID:1820089 (2015)"},{"key":"3376_CR9","doi-asserted-by":"crossref","unstructured":"Ginosar, S., Bar, A., Kohavi, G., Chan, C., Owens, A., Malik, J.: Learning individual styles of conversational gesture. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3492\u20133501. https:\/\/api.semanticscholar.org\/CorpusID:182952539 (2019)","DOI":"10.1109\/CVPR.2019.00361"},{"issue":"7","key":"3376_CR10","doi-asserted-by":"publisher","first-page":"6662","DOI":"10.1109\/TCYB.2021.3079311","volume":"52","author":"B Sheng","year":"2022","unstructured":"Sheng, B., Li, P., Ali, R., Philip Chen, C.L.: Improving video temporal consistency via broad learning system. IEEE Trans. Cybern. 52(7), 6662\u20136675 (2022). https:\/\/doi.org\/10.1109\/TCYB.2021.3079311","journal-title":"IEEE Trans. Cybern."},{"key":"3376_CR11","doi-asserted-by":"publisher","first-page":"4499","DOI":"10.1109\/TNNLS.2021.3116209","volume":"34","author":"Z Xie","year":"2021","unstructured":"Xie, Z., Zhang, W., Sheng, B., Li, P., Chen, C.P.: BaGFN: broad attentive graph fusion network for high-order feature interactions. IEEE Trans. Neural Netw. Learn. Syst. 34, 4499\u20134513 (2021)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"3376_CR12","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth 16x16 words: transformers for image recognition at scale. In: arXiv:2010.11929. https:\/\/api.semanticscholar.org\/CorpusID:225039882 (2020)"},{"key":"3376_CR13","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: 2021 IEEE\/CVF international conference on computer vision (ICCV), pp. 9992\u201310002. https:\/\/api.semanticscholar.org\/CorpusID:232352874 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"3376_CR14","unstructured":"Vaswani, A., Shazeer, N.M., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. In: Neural Information Processing Systems. https:\/\/api.semanticscholar.org\/CorpusID:13756489 (2017)"},{"key":"3376_CR15","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1109\/TMM.2021.3120873","volume":"25","author":"X Lin","year":"2021","unstructured":"Lin, X., Sun, S., Huang, W., Sheng, B., Li, P., Feng, D.D.: EAPT: efficient attention pyramid transformer for image processing. IEEE Trans. Multimed. 25, 50\u201361 (2021)","journal-title":"IEEE Trans. Multimed."},{"key":"3376_CR16","doi-asserted-by":"crossref","unstructured":"Li, R., Yang, S., Ross, D.A., Kanazawa, A.: AI choreographer: music conditioned 3D dance generation with AIST++. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 13381\u201313392. https:\/\/api.semanticscholar.org\/CorpusID:236882798 (2021)","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"3376_CR17","doi-asserted-by":"crossref","unstructured":"Siyao, L., Yu, W., Gu, T., Lin, C., Wang, Q., Qian, C., Loy, C.C., Liu, Zi.: Bailando: 3D dance generation by actor-critic GPT with choreographic memory. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11040\u201311049. https:\/\/api.semanticscholar.org\/CorpusID:247627867 (2022)","DOI":"10.1109\/CVPR52688.2022.01077"},{"key":"3376_CR18","doi-asserted-by":"publisher","first-page":"855","DOI":"10.1007\/s11263-019-01245-6","volume":"128","author":"D Pavllo","year":"2019","unstructured":"Pavllo, D., Feichtenhofer, C., Auli, M., Grangier, D.: Modeling human motion with quaternion-based neural networks. Int. J. Comput. Vis. 128, 855\u2013872 (2019)","journal-title":"Int. J. Comput. Vis."},{"key":"3376_CR19","doi-asserted-by":"crossref","unstructured":"Ma, W., Yin, M., Li, G., Yang, F., Chang, K.: PCMG:3D point cloud human motion generation based on self-attention and transformer. In: The Visual Computer. https:\/\/api.semanticscholar.org\/CorpusID:261566852 (2023)","DOI":"10.1007\/s00371-023-03063-x"},{"key":"3376_CR20","doi-asserted-by":"crossref","unstructured":"Greenwood, D., Laycock, S.D., Matthews, I.: Predicting head pose from speech with a conditional variational autoencoder. In: Interspeech. https:\/\/api.semanticscholar.org\/CorpusID:11113871 (2017)","DOI":"10.21437\/Interspeech.2017-894"},{"key":"3376_CR21","doi-asserted-by":"crossref","unstructured":"Huang, Y., Zhang, J., Liu, S., Bao, Q., Zeng, D., Chen, Z., Liu, W.: Genre-conditioned long-term 3D dance generation driven by music. In: ICASSP 2022\u20142022 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 4858\u20134862. https:\/\/api.semanticscholar.org\/CorpusID:249437513 (2022)","DOI":"10.1109\/ICASSP43922.2022.9747838"},{"key":"3376_CR22","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9, 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"3376_CR23","unstructured":"Yu, Q., He, J., Deng, X., Shen, X., Chen, L.-C.: Convolutions die hard: open-vocabulary segmentation with single frozen convolutional CLIP. In: arXiv:2308.02487. https:\/\/api.semanticscholar.org\/CorpusID:260611350 (2023)"},{"key":"3376_CR24","doi-asserted-by":"crossref","unstructured":"Tsai, Y.H.H., Bai, S., Liang, P.P., Kolter, J.Z., Morency, L.P., Salakhutdinov, R.: Multimodal transformer for unaligned multimodal language sequences. In: Proceedings of the conference. Association for Computational Linguistics. Meeting 2019, pp. 6558\u20136569. https:\/\/api.semanticscholar.org\/CorpusID:173990158 (2019)","DOI":"10.18653\/v1\/P19-1656"},{"key":"3376_CR25","unstructured":"Wu, Z., Xu, J., Zou, X., Huang, K., Shi, X., Huang, J.: EasyPhoto: your smart AI photo generator. https:\/\/api.semanticscholar.org\/CorpusID:263829612 (2023)"},{"key":"3376_CR26","unstructured":"Tendulkar, P., Das, A., Kembhavi, A., Parikh, D.: Feel the music: automatically generating a dance for an input song. In: arXiv:2006.11905. https:\/\/api.semanticscholar.org\/CorpusID:219572850 (2020)"},{"key":"3376_CR27","doi-asserted-by":"crossref","unstructured":"Kundu, J.N., Buckchash, H., Mandikal, P., Jamkhandi, A., Radhakrishnan, V.B.: Cross-conditioned recurrent networks for long-term synthesis of inter-person human motion interactions. In: 2020 IEEE winter conference on applications of computer vision (WACV), pp. 2713\u20132722. https:\/\/api.semanticscholar.org\/CorpusID:214675800 (2020)","DOI":"10.1109\/WACV45572.2020.9093627"},{"key":"3376_CR28","unstructured":"Li, L., Lei, J., Gan, Z., Yu, L., Chen, Y.-C., Pillai, R.K., Cheng, Y., Zhou, L., Wang, X.E., Wang, W.Y., Berg, T.L., Bansal, M., Liu, J., Wang, L., Liu, Z.: VALUE: a multi-task benchmark for video-and-language understanding evaluation. In: arXiv:2106.04632. https:\/\/api.semanticscholar.org\/CorpusID:235377363 (2021)"},{"key":"3376_CR29","doi-asserted-by":"crossref","unstructured":"Ghosh, P., Song, J., Aksan, E., Hilliges, O.: Learning human motion models for long-term predictions. In: 2017 International Conference on 3D Vision (3DV), pp. 458\u2013466. https:\/\/api.semanticscholar.org\/CorpusID:13549534 (2017)","DOI":"10.1109\/3DV.2017.00059"},{"key":"3376_CR30","unstructured":"Wu, C., Yin, S.-K., Qi, W., Wang, X., Tang, Z., Duan, N.: Visual ChatGPT: talking, drawing and editing with visual foundation models. In: arXiv:2303.04671. https:\/\/api.semanticscholar.org\/CorpusID:257404891 (2023)"},{"key":"3376_CR31","doi-asserted-by":"crossref","unstructured":"Du, Z., Qian, Y., Liu, X., Ding, M., Qiu, J., Yang, Z., Tang, J.: GLM: general language model pretraining with autoregressive blank infilling. In: Annual Meeting of the Association for Computational Linguistics. https:\/\/api.semanticscholar.org\/CorpusID:247519241 (2021)","DOI":"10.18653\/v1\/2022.acl-long.26"},{"issue":"4","key":"3376_CR32","doi-asserted-by":"publisher","first-page":"913","DOI":"10.53106\/160792642021072204018","volume":"22","author":"Z Bai","year":"2021","unstructured":"Bai, Z., Chen, X., Zhou, M., Yi, T., Chien, W.-C.: Low-rank multimodal fusion algorithm based on context modeling. J. Internet Technol. 22(4), 913\u2013921 (2021)","journal-title":"J. Internet Technol."},{"key":"3376_CR33","doi-asserted-by":"crossref","unstructured":"Holden, D., Saito, J., Komura, T.: A deep learning framework for character motion synthesis and editing. ACM Trans. Graph. (TOG) 35, 1\u201311 (2016)","DOI":"10.1145\/2897824.2925975"},{"key":"3376_CR34","doi-asserted-by":"crossref","unstructured":"Qiu, H., Wang, C., Wang, J., Wang, N., Zeng, W.: Cross view fusion for 3D human pose estimation. In: 2019 IEEE\/CVF international conference on computer vision (ICCV), pp. 4341\u20134350. https:\/\/api.semanticscholar.org\/CorpusID:201891326 (2019)","DOI":"10.1109\/ICCV.2019.00444"},{"key":"3376_CR35","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Olszewski, K., Wu, Y., Achlioptas, P., Chai, M., Yan, Y., Tulyakov, S.: Quantized GAN for complex music generation from dance videos. In: arXiv:2204.00604. https:\/\/api.semanticscholar.org\/CorpusID:247922422 (2022)","DOI":"10.1007\/978-3-031-19836-6_11"},{"key":"3376_CR36","doi-asserted-by":"publisher","first-page":"2102","DOI":"10.1109\/TCSVT.2022.3223150","volume":"33","author":"Z Zheng","year":"2023","unstructured":"Zheng, Z., Huang, G., Yuan, X., Pun, C.-M., Liu, H., Ling, W.-K.: Quaternion-valued correlation learning for few-shot semantic segmentation. IEEE Trans. Circuits Syst. Video Technol. 33, 2102\u20132115 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3376_CR37","unstructured":"Su, J., Lu, Y., Pan, S., Wen, B., Liu, Y.: RoFormer: enhanced transformer with rotary position embedding. In: arXiv:2104.09864. https:\/\/api.semanticscholar.org\/CorpusID:233307138 (2021)"},{"key":"3376_CR38","unstructured":"Tsuchida, S., Fukayama, S., Hamasaki, M., Goto, M.: AIST dance video database: multi-genre, multi-dancer, and multi-camera database for dance information processing. In: International Society for Music Information Retrieval Conference. https:\/\/api.semanticscholar.org\/CorpusID:208334750 (2019)"},{"key":"3376_CR39","doi-asserted-by":"crossref","unstructured":"McFee, B., Raffel, C., Liang, D., Ellis, D.P.W., McVicar, M., Battenberg, E., Nieto, O.: librosa: audio and music signal analysis in python. In: SciPy. https:\/\/api.semanticscholar.org\/CorpusID:33504 (2015)","DOI":"10.25080\/Majora-7b98e3ed-003"},{"key":"3376_CR40","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: GANs trained by a two time-scale update rule converge to a local nash equilibrium. In: Neural Information Processing Systems. https:\/\/api.semanticscholar.org\/CorpusID:326772 (2017)"},{"key":"3376_CR41","unstructured":"Onuma, K., Faloutsos, C., Hodgins, J.K.: FMDistance: a fast and effective distance function for motion capture data. In: Eurographics. https:\/\/api.semanticscholar.org\/CorpusID:8323054 (2008)"},{"key":"3376_CR42","doi-asserted-by":"crossref","unstructured":"Tan, H.H., Bansal, M.: LXMERT: learning cross-modality encoder representations from transformers. In: Conference on Empirical Methods in Natural Language Processing. https:\/\/api.semanticscholar.org\/CorpusID:201103729 (2019)","DOI":"10.18653\/v1\/D19-1514"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03376-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-024-03376-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03376-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,3]],"date-time":"2025-02-03T12:36:46Z","timestamp":1738586206000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-024-03376-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,14]]},"references-count":42,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["3376"],"URL":"https:\/\/doi.org\/10.1007\/s00371-024-03376-5","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4,14]]},"assertion":[{"value":"16 March 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 April 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"We declare that we do not have any commercial or associative interest that represents a conflict of interest in connection with the work submitted.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}