{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T02:28:16Z","timestamp":1783823296231,"version":"3.55.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2024,10,16]],"date-time":"2024-10-16T00:00:00Z","timestamp":1729036800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,16]],"date-time":"2024-10-16T00:00:00Z","timestamp":1729036800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Chunhui Plan Cooperative Project of Ministry of Education under Grant","award":["HZKY20220424"],"award-info":[{"award-number":["HZKY20220424"]}]},{"name":"the Guangdong Basic and Applied Basic Research Foundation","award":["2022A1515140126, 2023A1515011172"],"award-info":[{"award-number":["2022A1515140126, 2023A1515011172"]}]},{"name":"the Young and Middle-aged Science and Technology Innovation Talent of Shenyang","award":["RC220485"],"award-info":[{"award-number":["RC220485"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s00530-024-01492-9","type":"journal-article","created":{"date-parts":[[2024,10,16]],"date-time":"2024-10-16T06:01:55Z","timestamp":1729058515000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Motion synthesis via distilled absorbing discrete diffusion model"],"prefix":"10.1007","volume":"30","author":[{"given":"Junyi","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chao","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bangli","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haibin","family":"Cai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinggang","family":"Meng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,16]]},"reference":[{"issue":"3","key":"1492_CR1","doi-asserted-by":"publisher","first-page":"1301","DOI":"10.1007\/s00530-023-01054-5","volume":"29","author":"X Chao","year":"2023","unstructured":"Chao, X., Hou, Z., Mo, Y., Shi, H., Yao, W.: Structural feature representation and fusion of human spatial cooperative motion for action recognition. Multimed. Syst. 29(3), 1301\u20131314 (2023)","journal-title":"Multimed. Syst."},{"issue":"6","key":"1492_CR2","doi-asserted-by":"publisher","first-page":"671","DOI":"10.1007\/s00530-020-00677-2","volume":"26","author":"P Verma","year":"2020","unstructured":"Verma, P., Sah, A., Srivastava, R.: Deep learning-based multi-modal approach using rgb and skeleton sequences for human activity recognition. Multimed. Syst. 26(6), 671\u2013685 (2020)","journal-title":"Multimed. Syst."},{"issue":"1","key":"1492_CR3","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1007\/s00530-022-00981-z","volume":"29","author":"S Liu","year":"2023","unstructured":"Liu, S., He, N., Wang, C., Yu, H., Han, W.: Lightweight human pose estimation algorithm based on polarized self-attention. Multimed. Syst. 29(1), 197\u2013210 (2023)","journal-title":"Multimed. Syst."},{"issue":"4","key":"1492_CR4","doi-asserted-by":"publisher","first-page":"2085","DOI":"10.1007\/s00530-023-01085-y","volume":"29","author":"H Yang","year":"2023","unstructured":"Yang, H., Liu, H., Zhang, Y., Wu, X.: Hsgnet: hierarchically stacked graph network with attention mechanism for 3d human pose estimation. Multimed. Syst. 29(4), 2085\u20132097 (2023)","journal-title":"Multimed. Syst."},{"key":"1492_CR5","doi-asserted-by":"crossref","unstructured":"Guo, C., Zou, S., Zuo, X., Wang, S., Ji, W., Li, X., Cheng, L.: Generating diverse and natural 3d human motions from text. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5152\u20135161 (2022)","DOI":"10.1109\/CVPR52688.2022.00509"},{"issue":"1","key":"1492_CR6","doi-asserted-by":"publisher","first-page":"681","DOI":"10.1109\/TPAMI.2021.3139918","volume":"45","author":"Z Liu","year":"2022","unstructured":"Liu, Z., Wu, S., Jin, S., Ji, S., Liu, Q., Lu, S., Cheng, L.: Investigating pose representations and motion contexts modeling for 3d motion prediction. IEEE Trans. Pattern Anal. Mach. Intell. 45(1), 681\u2013697 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1492_CR7","doi-asserted-by":"crossref","unstructured":"Mao, W., Liu, M., Salzmann, M., Li, H.: Learning trajectory dependencies for human motion prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9489\u20139497 (2019)","DOI":"10.1109\/ICCV.2019.00958"},{"issue":"2","key":"1492_CR8","doi-asserted-by":"publisher","first-page":"413","DOI":"10.1007\/s00530-021-00807-4","volume":"28","author":"R Zhang","year":"2022","unstructured":"Zhang, R., Shu, X., Yan, R., Zhang, J., Song, Y.: Skip-attention encoder-decoder framework for human motion prediction. Multimed. Syst. 28(2), 413\u2013422 (2022)","journal-title":"Multimed. Syst."},{"issue":"6","key":"1492_CR9","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/s00138-023-01447-6","volume":"34","author":"L Geng","year":"2023","unstructured":"Geng, L., Yang, W., Jiao, Y., Zeng, S., Chen, X.: A multilayer human motion prediction perceptron by aggregating repetitive motion. Mach. Vis. Appl. 34(6), 98 (2023)","journal-title":"Mach. Vis. Appl."},{"key":"1492_CR10","doi-asserted-by":"crossref","unstructured":"Siyao, L., Yu, W., Gu, T., Lin, C., Wang, Q., Qian, C., Loy, C.C., Liu, Z.: Bailando: 3d dance generation by actor-critic gpt with choreographic memory. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11050\u201311059 (2022)","DOI":"10.1109\/CVPR52688.2022.01077"},{"key":"1492_CR11","doi-asserted-by":"crossref","unstructured":"Tseng, J., Castellon, R., Liu, K.: Edge: Editable dance generation from music. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 448\u2013458 (2023)","DOI":"10.1109\/CVPR52729.2023.00051"},{"key":"1492_CR12","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Wang, B.: Ude: A unified driving engine for human motion generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5632\u20135641 (2023)","DOI":"10.1109\/CVPR52729.2023.00545"},{"key":"1492_CR13","doi-asserted-by":"crossref","unstructured":"Guo, C., Zuo, X., Wang, S., Zou, S., Sun, Q., Deng, A., Gong, M., Cheng, L.: Action2motion: Conditioned generation of 3d human motions. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 2021\u20132029 (2020)","DOI":"10.1145\/3394171.3413635"},{"key":"1492_CR14","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: Action-conditioned 3d human motion synthesis with transformer vae. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10985\u201310995 (2021)","DOI":"10.1109\/ICCV48922.2021.01080"},{"key":"1492_CR15","doi-asserted-by":"crossref","unstructured":"Cervantes, P., Sekikawa, Y., Sato, I., Shinoda, K.: Implicit neural representations for variable length human motion generation. In: European Conference on Computer Vision, pp. 356\u2013372 (2022)","DOI":"10.1007\/978-3-031-19790-1_22"},{"key":"1492_CR16","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: Temos: Generating diverse human motions from textual descriptions. In: European Conference on Computer Vision, pp. 480\u2013497 (2022)","DOI":"10.1007\/978-3-031-20047-2_28"},{"key":"1492_CR17","doi-asserted-by":"crossref","unstructured":"Cai, H., Bai, C., Tai, Y.-W., Tang, C.-K.: Deep video generation, prediction and completion of human action sequences. In: European Conference on Computer Vision, pp. 374\u2013390 (2018)","DOI":"10.1007\/978-3-030-01216-8_23"},{"key":"1492_CR18","first-page":"12281","volume":"34","author":"Z Wang","year":"2020","unstructured":"Wang, Z., Yu, P., Zhao, Y., Zhang, R., Zhou, Y., Yuan, J., Chen, C.: Learning diverse stochastic human-action generators by learning smooth latent transitions. Proceed. AAAI Conf. Artif. Intell. 34, 12281\u201312288 (2020)","journal-title":"Proceed. AAAI Conf. Artif. Intell."},{"key":"1492_CR19","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., Bengio, Y.: Generative adversarial nets. Adv. Neural Inform. Process. Syst. 27 (2014)"},{"key":"1492_CR20","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"1492_CR21","doi-asserted-by":"crossref","unstructured":"Zhang, M., Cai, Z., Pan, L., Hong, F., Guo, X., Yang, L., Liu, Z.: Motiondiffuse: Text-driven human motion generation with diffusion model. IEEE Transactions on Pattern Analysis and Machine Intelligence, 1\u201315 (2024)","DOI":"10.1109\/TPAMI.2024.3355414"},{"key":"1492_CR22","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1492_CR23","doi-asserted-by":"crossref","unstructured":"Chen, X., Jiang, B., Liu, W., Huang, Z., Fu, B., Chen, T., Yu, G.: Executing your commands via motion diffusion in latent space. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18000\u201318010 (2023)","DOI":"10.1109\/CVPR52729.2023.01726"},{"key":"1492_CR24","doi-asserted-by":"crossref","unstructured":"Zhang, M., Guo, X., Pan, L., Cai, Z., Hong, F., Li, H., Yang, L., Liu, Z.: Remodiffuse: Retrieval-augmented motion diffusion model. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 364\u2013373 (2023)","DOI":"10.1109\/ICCV51070.2023.00040"},{"key":"1492_CR25","doi-asserted-by":"crossref","unstructured":"Guo, C., Zuo, X., Wang, S., Cheng, L.: Tm2t: Stochastic and tokenized modeling for the reciprocal generation of 3d human motions and texts. In: European Conference on Computer Vision, pp. 580\u2013597 (2022)","DOI":"10.1007\/978-3-031-19833-5_34"},{"key":"1492_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, J., Zhang, Y., Cun, X., Zhang, Y., Zhao, H., Lu, H., Shen, X., Shan, Y.: Generating human motion from textual descriptions with discrete representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14730\u201314740 (2023)","DOI":"10.1109\/CVPR52729.2023.01415"},{"key":"1492_CR27","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Adv. Neural Inform. Process. Syst. 30 (2017)"},{"key":"1492_CR28","doi-asserted-by":"crossref","unstructured":"Zhong, C., Hu, L., Zhang, Z., Xia, S.: Attt2m: Text-driven human motion generation with multi-perspective attention mechanism. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 509\u2013519 (2023)","DOI":"10.1109\/ICCV51070.2023.00053"},{"key":"1492_CR29","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. In: International Conference on Machine Learning, pp. 2256\u20132265 (2015)"},{"key":"1492_CR30","first-page":"12454","volume":"34","author":"E Hoogeboom","year":"2021","unstructured":"Hoogeboom, E., Nielsen, D., Jaini, P., Forr\u00e9, P., Welling, M.: Argmax flows and multinomial diffusion: Learning categorical distributions. Adv. Neural. Inf. Process. Syst. 34, 12454\u201312465 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1492_CR31","first-page":"17981","volume":"34","author":"J Austin","year":"2021","unstructured":"Austin, J., Johnson, D.D., Ho, J., Tarlow, D., Van Den Berg, R.: Structured denoising diffusion models in discrete state-spaces. Adv. Neural. Inf. Process. Syst. 34, 17981\u201317993 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1492_CR32","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"1492_CR33","unstructured":"Lan, Z., Chen, M., Goodman, S., Gimpel, K., Sharma, P., Soricut, R.: Albert: A lite bert for self-supervised learning of language representations. arXiv preprint arXiv:1909.11942 (2019)"},{"key":"1492_CR34","unstructured":"Van Den\u00a0Oord, A., Vinyals, O., et al.: Neural discrete representation learning. Adv. Neural Inform. Process. Syst. 30 (2017)"},{"key":"1492_CR35","doi-asserted-by":"crossref","unstructured":"Bond-Taylor, S., Hessey, P., Sasaki, H., Breckon, T.P., Willcocks, C.G.: Unleashing transformers: Parallel token prediction with discrete absorbing diffusion for fast high-resolution image generation from vector-quantized codes. In: European Conference on Computer Vision, pp. 170\u2013188 (2022)","DOI":"10.1007\/978-3-031-20050-2_11"},{"key":"1492_CR36","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021)"},{"key":"1492_CR37","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)"},{"key":"1492_CR38","unstructured":"Lin, A.S., Wu, L., Corona, R., Tai, K., Huang, Q., Mooney, R.J.: Generating animated videos of human activities from natural language descriptions. Adv. Neural Inform. Process. Syst. (2018)"},{"key":"1492_CR39","doi-asserted-by":"crossref","unstructured":"Ahuja, C., Morency, L.-P.: Language2pose: Natural language grounded pose forecasting. In: 2019 International Conference on 3D Vision (3DV), pp. 719\u2013728 (2019)","DOI":"10.1109\/3DV.2019.00084"},{"key":"1492_CR40","doi-asserted-by":"crossref","unstructured":"Bhattacharya, U., Rewkowski, N., Banerjee, A., Guhan, P., Bera, A., Manocha, D.: Text2gestures: A transformer-based network for generating emotive body gestures for virtual agents. In: 2021 IEEE Virtual Reality and 3D User Interfaces (VR), pp. 1\u201310 (2021)","DOI":"10.1109\/VR50410.2021.00037"},{"key":"1492_CR41","doi-asserted-by":"crossref","unstructured":"Ghosh, A., Cheema, N., Oguz, C., Theobalt, C., Slusallek, P.: Synthesis of compositional animations from textual descriptions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1396\u20131406 (2021)","DOI":"10.1109\/ICCV48922.2021.00143"},{"key":"1492_CR42","doi-asserted-by":"crossref","unstructured":"Tulyakov, S., Liu, M.-Y., Yang, X., Kautz, J.: Mocogan: Decomposing motion and content for video generation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1526\u20131535 (2018)","DOI":"10.1109\/CVPR.2018.00165"},{"key":"1492_CR43","unstructured":"Lee, H.-Y., Yang, X., Liu, M.-Y., Wang, T.-C., Lu, Y.-D., Yang, M.-H., Kautz, J.: Dancing to music. Adv. Neural Inform. Process. Syst. 32 (2019)"},{"key":"1492_CR44","unstructured":"Tevet, G., Raab, S., Gordon, B., Shafir, Y., Cohen-Or, D., Bermano, A.H.: Human motion diffusion model. arXiv preprint arXiv:2209.14916 (2022)"},{"key":"1492_CR45","doi-asserted-by":"crossref","unstructured":"Mahmood, N., Ghorbani, N., Troje, N.F., Pons-Moll, G., Black, M.J.: Amass: Archive of motion capture as surface shapes. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5442\u20135451 (2019)","DOI":"10.1109\/ICCV.2019.00554"},{"issue":"4","key":"1492_CR46","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1089\/big.2016.0028","volume":"4","author":"C.M. Matthias Plappert","year":"2016","unstructured":"Plappert, C.M. Matthias., Asfour, T.: The kit motion-language dataset. Big Data 4(4), 236\u2013252 (2016)","journal-title":"Big Data"},{"key":"1492_CR47","unstructured":"Zheng, L., Yuan, J., Yu, L., Kong, L. : A reparameterized discrete diffusion model for text generation. arXiv preprint arXiv:2302.05737 (2023)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01492-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01492-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01492-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T09:07:42Z","timestamp":1734340062000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01492-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,16]]},"references-count":47,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["1492"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01492-9","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,16]]},"assertion":[{"value":"22 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"320"}}