{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:44:18Z","timestamp":1783439058294,"version":"3.54.6"},"publisher-location":"Cham","reference-count":106,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726262","type":"print"},{"value":"9783031726279","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T00:00:00Z","timestamp":1729382400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T00:00:00Z","timestamp":1729382400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72627-9_2","type":"book-chapter","created":{"date-parts":[[2024,10,19]],"date-time":"2024-10-19T21:02:10Z","timestamp":1729371730000},"page":"18-38","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":47,"title":["EMDM: Efficient Motion Diffusion Model for\u00a0Fast and\u00a0High-Quality Motion Generation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-0575-0523","authenticated-orcid":false,"given":"Wenyang","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0186-8269","authenticated-orcid":false,"given":"Zhiyang","family":"Dou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8269-3602","authenticated-orcid":false,"given":"Zeyu","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6525-1372","authenticated-orcid":false,"given":"Zhouyingcheng","family":"Liao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0740-8548","authenticated-orcid":false,"given":"Jingbo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0121-3852","authenticated-orcid":false,"given":"Wenjia","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2933-5667","authenticated-orcid":false,"given":"Yuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2729-5860","authenticated-orcid":false,"given":"Taku","family":"Komura","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2284-3952","authenticated-orcid":false,"given":"Wenping","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4301-1474","authenticated-orcid":false,"given":"Lingjie","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,20]]},"reference":[{"key":"2_CR1","doi-asserted-by":"crossref","unstructured":"Ahuja, C., Morency, L.P.: Language2pose: natural language grounded pose forecasting. In: 2019 International Conference on 3D Vision (3DV), pp. 719\u2013728. IEEE (2019)","DOI":"10.1109\/3DV.2019.00084"},{"issue":"4","key":"2_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592458","volume":"42","author":"S Alexanderson","year":"2023","unstructured":"Alexanderson, S., Nagy, R., Beskow, J., Henter, G.E.: Listen, denoise, action! audio-driven motion synthesis with diffusion models. ACM Trans. Graph. (TOG) 42(4), 1\u201320 (2023)","journal-title":"ACM Trans. Graph. (TOG)"},{"issue":"6","key":"2_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3550454.3555435","volume":"41","author":"T Ao","year":"2022","unstructured":"Ao, T., Gao, Q., Lou, Y., Chen, B., Liu, L.: Rhythmic gesticulator: rhythm-aware co-speech gesture synthesis with hierarchical neural embeddings. ACM Trans. Graph. (TOG) 41(6), 1\u201319 (2022)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Ao, T., Zhang, Z., Liu, L.: GestureDiffuClip: gesture diffusion model with clip latents. arXiv preprint arXiv:2303.14613 (2023)","DOI":"10.1145\/3592097"},{"key":"2_CR5","doi-asserted-by":"publisher","unstructured":"Cervantes, P., Sekikawa, Y., Sato, I., Shinoda, K.: Implicit neural representations for variable length human motion generation. In: European Conference on Computer Vision,. pp. 356\u2013372. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19790-1_22","DOI":"10.1007\/978-3-031-19790-1_22"},{"key":"2_CR6","doi-asserted-by":"crossref","unstructured":"Chen, R., Shi, M., Huang, S., Tan, P., Komura, T., Chen, X.: Taming diffusion probabilistic models for character control. arXiv preprint arXiv:2404.15121 (2024)","DOI":"10.1145\/3641519.3657440"},{"key":"2_CR7","doi-asserted-by":"crossref","unstructured":"Chen, X., et al.: Executing your commands via motion diffusion in latent space. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18000\u201318010 (2023)","DOI":"10.1109\/CVPR52729.2023.01726"},{"key":"2_CR8","unstructured":"Chen, X., et al.: Learning variational motion prior for video-based motion capture. arXiv preprint arXiv:2210.15134 (2022)"},{"issue":"1","key":"2_CR9","doi-asserted-by":"publisher","first-page":"6386","DOI":"10.1038\/s41467-020-19712-x","volume":"11","author":"E Chong","year":"2020","unstructured":"Chong, E., et al.: Detection of eye contact with deep neural networks is as accurate as human experts. Nat. Commun. 11(1), 6386 (2020)","journal-title":"Nat. Commun."},{"key":"2_CR10","doi-asserted-by":"crossref","unstructured":"Chou, G., Bahat, Y., Heide, F.: Diffusion-SDF: conditional generative modeling of signed distance functions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2262\u20132272 (2023)","DOI":"10.1109\/ICCV51070.2023.00215"},{"key":"2_CR11","doi-asserted-by":"crossref","unstructured":"Christen, S., et al.: Learning human-to-robot handovers from point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9654\u20139664 (2023)","DOI":"10.1109\/CVPR52729.2023.00931"},{"key":"2_CR12","unstructured":"Chung, H.W., et\u00a0al.: Scaling instruction-finetuned language models. arXiv preprint arXiv:2210.11416 (2022)"},{"key":"2_CR13","unstructured":"Cong, P., et al.: LaserHuman: language-guided scene-aware human motion generation in free environment. arXiv preprint arXiv:2403.13307 (2024)"},{"key":"2_CR14","doi-asserted-by":"crossref","unstructured":"Crawford, F.W., et\u00a0al.: Impact of close interpersonal contact on COVID-19 incidence: evidence from 1 year of mobile device data. Sci. Adv. 8(1), eabi5499 (2022)","DOI":"10.1126\/sciadv.abi5499"},{"key":"2_CR15","doi-asserted-by":"crossref","unstructured":"Dabral, R., Mughal, M.H., Golyanik, V., Theobalt, C.: MoFusion: a framework for denoising-diffusion-based motion synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9760\u20139770 (2023)","DOI":"10.1109\/CVPR52729.2023.00941"},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Dou, Z., Chen, X., Fan, Q., Komura, T., Wang, W.: C$$\\cdot $$ASE: learning conditional adversarial skill embeddings for physics-based characters. arXiv preprint arXiv:2309.11351 (2023)","DOI":"10.1145\/3610548.3618205"},{"key":"2_CR17","doi-asserted-by":"crossref","unstructured":"Dou, Z., et al.: TORE: token reduction for efficient human mesh recovery with transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15143\u201315155 (2023)","DOI":"10.1109\/ICCV51070.2023.01390"},{"key":"2_CR18","unstructured":"Duan, Y., et al.: Single-shot motion completion with transformer. arXiv preprint arXiv:2103.00776 (2021)"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Guo, C., et al.: Generating diverse and natural 3D human motions from text. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5152\u20135161 (2022)","DOI":"10.1109\/CVPR52688.2022.00509"},{"key":"2_CR20","doi-asserted-by":"crossref","unstructured":"Guo, C., Zuo, X., Wang, S., Cheng, L.: TM2T: stochastic and tokenized modeling for the reciprocal generation of 3D human motions and texts. In: ECCV (2022)","DOI":"10.1007\/978-3-031-19833-5_34"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Guo, C., et al.: Action2Motion: conditioned generation of 3D human motions. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 2021\u20132029 (2020)","DOI":"10.1145\/3394171.3413635"},{"key":"2_CR22","doi-asserted-by":"crossref","unstructured":"Guo, Y., et al.: Student close contact behavior and COVID-19 transmission in China\u2019s classrooms. PNAS Nexus 2(5), pgad142 (2023)","DOI":"10.1093\/pnasnexus\/pgad142"},{"issue":"4","key":"2_CR23","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1145\/3386569.3392480","volume":"39","author":"FG Harvey","year":"2020","unstructured":"Harvey, F.G., Yurick, M., Nowrouzezahrai, D., Pal, C.: Robust motion in-betweening. ACM Trans. Graph. (TOG) 39(4), 60\u20131 (2020)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR24","unstructured":"Ho, J., et\u00a0al.: Imagen Video: high definition video generation with diffusion models. arXiv preprint arXiv:2210.02303 (2022)"},{"key":"2_CR25","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2_CR26","unstructured":"Ho, J., Salimans, T.: Classifier-free diffusion guidance. arXiv preprint arXiv:2207.12598 (2022)"},{"issue":"4","key":"2_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073663","volume":"36","author":"D Holden","year":"2017","unstructured":"Holden, D., Komura, T., Saito, J.: Phase-functioned neural networks for character control. ACM Trans. Graph. (TOG) 36(4), 1\u201313 (2017)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR28","unstructured":"Jiang, B., Chen, X., Liu, W., Yu, J., Yu, G., Chen, T.: MotionGPT: human motion as a foreign language. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Jiang, Y., Won, J., Ye, Y., Liu, C.K.: DROP: dynamics responses from human motion prior and projective dynamics. arXiv preprint arXiv:2309.13742 (2023)","DOI":"10.1145\/3610548.3618175"},{"key":"2_CR30","doi-asserted-by":"crossref","unstructured":"Karunratanakul, K., Preechakul, K., Suwajanakorn, S., Tang, S.: Guided motion diffusion for controllable human motion synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2151\u20132162 (2023)","DOI":"10.1109\/ICCV51070.2023.00205"},{"key":"2_CR31","unstructured":"Kim, J., Kim, J., Choi, S.: FLAME: free-form language-based motion synthesis & editing. arXiv preprint arXiv:2209.00349 (2022)"},{"key":"2_CR32","doi-asserted-by":"crossref","unstructured":"Kong, H., Gong, K., Lian, D., Mi, M.B., Wang, X.: Priority-centric human motion generation in discrete latent space. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14806\u201314816 (2023)","DOI":"10.1109\/ICCV51070.2023.01360"},{"key":"2_CR33","unstructured":"Lee, H.Y., et al.: Dancing to music. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"2_CR34","doi-asserted-by":"crossref","unstructured":"Lee, S., Starke, S., Ye, Y., Won, J., Winkler, A.: QuestEnvSim: environment-aware simulated motion tracking from sparse sensors. arXiv preprint arXiv:2306.05666 (2023)","DOI":"10.1145\/3588432.3591504"},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Lee, T., Moon, G., Lee, K.M.: MultiAct: long-term 3D human motion generation from multiple action labels. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 1231\u20131239 (2023)","DOI":"10.1609\/aaai.v37i1.25206"},{"key":"2_CR36","doi-asserted-by":"crossref","unstructured":"Li, B., Zhao, Y., Zhelun, S., Sheng, L.: DanceFormer: music conditioned 3D dance generation with parametric motion transformer. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 1272\u20131279 (2022)","DOI":"10.1609\/aaai.v36i2.20014"},{"key":"2_CR37","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, C., Chen, Z., Bian, S., Yang, L., Lu, C.: HybriK: a hybrid analytical-neural inverse kinematics solution for 3D human pose and shape estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3383\u20133393 (2021)","DOI":"10.1109\/CVPR46437.2021.00339"},{"key":"2_CR38","doi-asserted-by":"crossref","unstructured":"Li, R., Yang, S., Ross, D.A., Kanazawa, A.: AI choreographer: music conditioned 3D dance generation with AIST++. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13401\u201313412 (2021)","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"2_CR39","doi-asserted-by":"crossref","unstructured":"Li, T., Qiao, C., Ren, G., Yin, K., Ha, S.: AAMDM: accelerated auto-regressive motion diffusion model. arXiv preprint arXiv:2401.06146 (2023)","DOI":"10.1109\/CVPR52733.2024.00178"},{"key":"2_CR40","doi-asserted-by":"crossref","unstructured":"Li, Z., Peng, X.B., Abbeel, P., Levine, S., Berseth, G., Sreenath, K.: Robust and versatile bipedal jumping control through reinforcement learning. Science and Systems XIX, Daegu, Republic of Korea, Robotics (2023)","DOI":"10.15607\/RSS.2023.XIX.052"},{"key":"2_CR41","doi-asserted-by":"crossref","unstructured":"Liao, Z., Golyanik, V., Habermann, M., Theobalt, C.: VINECS: video-based neural character skinning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1377\u20131387 (2024)","DOI":"10.1109\/CVPR52733.2024.00137"},{"key":"2_CR42","doi-asserted-by":"publisher","first-page":"640","DOI":"10.1007\/978-3-031-20086-1_37","volume-title":"Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part II","author":"Z Liao","year":"2022","unstructured":"Liao, Z., Yang, J., Saito, J., Pons-Moll, G., Zhou, Y.: Skeleton-free pose transfer for\u00a0stylized 3D characters. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part II, pp. 640\u2013656. Springer Nature Switzerland, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20086-1_37"},{"key":"2_CR43","doi-asserted-by":"publisher","DOI":"10.1016\/j.jhazmat.2022.129233","volume":"436","author":"X Liu","year":"2022","unstructured":"Liu, X., et al.: Close contact behavior-based COVID-19 transmission and interventions in a subway system. J. Hazard. Mater. 436, 129233 (2022)","journal-title":"J. Hazard. Mater."},{"key":"2_CR44","unstructured":"Liu, Y., et al.: SyncDreamer: generating multiview-consistent images from a single-view image. arXiv preprint arXiv:2309.03453 (2023)"},{"key":"2_CR45","doi-asserted-by":"crossref","unstructured":"Long, X., et\u00a0al.: Wonder3D: single image to 3D using cross-domain diffusion. arXiv preprint arXiv:2310.15008 (2023)","DOI":"10.1109\/CVPR52733.2024.00951"},{"key":"2_CR46","doi-asserted-by":"crossref","unstructured":"Mahmood, N., Ghorbani, N., Troje, N.F., Pons-Moll, G., Black, M.J.: AMASS: archive of motion capture as surface shapes. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00554"},{"key":"2_CR47","doi-asserted-by":"crossref","unstructured":"M\u00fcller, N., Siddiqui, Y., Porzi, L., Bulo, S.R., Kontschieder, P., Nie\u00dfner, M.: DiffRF: rendering-guided 3D radiance field diffusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4328\u20134338 (2023)","DOI":"10.1109\/CVPR52729.2023.00421"},{"issue":"4","key":"2_CR48","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592456","volume":"42","author":"K Pang","year":"2023","unstructured":"Pang, K., et al.: Bodyformer: semantics-guided 3D body gesture synthesis with transformer. ACM Trans. Graph. (TOG) 42(4), 1\u201312 (2023)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR49","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., et al.: Expressive body capture: 3D hands, face, and body from a single image. In: Proceedings IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.01123"},{"issue":"4","key":"2_CR50","first-page":"1","volume":"37","author":"XB Peng","year":"2018","unstructured":"Peng, X.B., Abbeel, P., Levine, S., Van de Panne, M.: DeepMimic: example-guided deep reinforcement learning of physics-based character skills. ACM Trans. Graph. (TOG) 37(4), 1\u201314 (2018)","journal-title":"ACM Trans. Graph. (TOG)"},{"issue":"4","key":"2_CR51","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3528223.3530110","volume":"41","author":"XB Peng","year":"2022","unstructured":"Peng, X.B., Guo, Y., Halper, L., Levine, S., Fidler, S.: ASE: large-scale reusable adversarial skill embeddings for physically simulated characters. ACM Trans. Graph. (TOG) 41(4), 1\u201317 (2022)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR52","doi-asserted-by":"publisher","unstructured":"Peng, X.B., Ma, Z., Abbeel, P., Levine, S., Kanazawa, A.: AMP: adversarial motion priors for stylized physics-based character control. ACM Trans. Graph. 40(4) (2021). https:\/\/doi.org\/10.1145\/3450626.3459670","DOI":"10.1145\/3450626.3459670"},{"key":"2_CR53","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: Action-conditioned 3D human motion synthesis with transformer VAE. In: International Conference on Computer Vision (ICCV) (2021)","DOI":"10.1109\/ICCV48922.2021.01080"},{"key":"2_CR54","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: TEMOS: generating diverse human motions from textual descriptions. In: European Conference on Computer Vision (ECCV) (2022)","DOI":"10.1007\/978-3-031-20047-2_28"},{"key":"2_CR55","doi-asserted-by":"publisher","unstructured":"Petrovich, M., Black, M.J., Varol, G.: TEMOS: generating diverse human motions from textual descriptions. In: European Conference on Computer Vision, pp. 480\u2013497. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-20047-2_28","DOI":"10.1007\/978-3-031-20047-2_28"},{"key":"2_CR56","doi-asserted-by":"crossref","unstructured":"Pi, H., Peng, S., Yang, M., Zhou, X., Bao, H.: Hierarchical generation of human-object interactions with diffusion probabilistic models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15061\u201315073 (2023)","DOI":"10.1109\/ICCV51070.2023.01383"},{"key":"2_CR57","doi-asserted-by":"publisher","unstructured":"Plappert, M., Mandery, C., Asfour, T.: The kit motion-language dataset. Big Data 4(4), 236\u2013252 (2016). https:\/\/doi.org\/10.1089\/big.2016.0028","DOI":"10.1089\/big.2016.0028"},{"key":"2_CR58","unstructured":"Po, R., et\u00a0al.: State of the art on diffusion models for visual computing. arXiv preprint arXiv:2310.07204 (2023)"},{"key":"2_CR59","doi-asserted-by":"crossref","unstructured":"Raab, S., Leibovitch, I., Li, P., Aberman, K., Sorkine-Hornung, O., Cohen-Or, D.: MoDi: unconditional motion synthesis from diverse data. arXiv preprint arXiv:2206.08010 (2022)","DOI":"10.1109\/CVPR52729.2023.01333"},{"key":"2_CR60","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"issue":"1","key":"2_CR61","first-page":"5485","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(1), 5485\u20135551 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"2_CR62","doi-asserted-by":"crossref","unstructured":"Rempe, D., Birdal, T., Hertzmann, A., Yang, J., Sridhar, S., Guibas, L.J.: HuMoR: 3D human motion model for robust pose estimation. In: International Conference on Computer Vision (ICCV) (2021)","DOI":"10.1109\/ICCV48922.2021.01129"},{"key":"2_CR63","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2022). https:\/\/github.com\/CompVis\/latent-diffusionhttps:\/\/arxiv.org\/abs\/2112.10752","DOI":"10.1109\/CVPR52688.2022.01042"},{"issue":"1","key":"2_CR64","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3407659","volume":"40","author":"M Shi","year":"2020","unstructured":"Shi, M., et al.: MotioNet: 3D human motion reconstruction from monocular video with skeleton consistency. ACM Trans. Graph. (TOG) 40(1), 1\u201315 (2020)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR65","doi-asserted-by":"crossref","unstructured":"Shi, M., Starke, S., Ye, Y., Komura, T., Won, J.: PhaseMP: robust 3D pose estimation via phase-conditioned human motion prior. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14725\u201314737 (2023)","DOI":"10.1109\/ICCV51070.2023.01353"},{"key":"2_CR66","unstructured":"Shi, Y., Wang, J., Jiang, X., Dai, B.: Controllable motion diffusion model. arXiv preprint arXiv:2306.00416 (2023)"},{"key":"2_CR67","doi-asserted-by":"crossref","unstructured":"Smith, L., et al.: Learning and adapting agile locomotion skills by transferring experience. arXiv preprint arXiv:2304.09834 (2023)","DOI":"10.15607\/RSS.2023.XIX.051"},{"key":"2_CR68","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. In: International Conference on Machine Learning, pp. 2256\u20132265. PMLR (2015)"},{"key":"2_CR69","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502 (2020)"},{"issue":"4","key":"2_CR70","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3528223.3530178","volume":"41","author":"S Starke","year":"2022","unstructured":"Starke, S., Mason, I., Komura, T.: DeepPhase: periodic autoencoders for learning motion phase manifolds. ACM Trans. Graph. (TOG) 41(4), 1\u201313 (2022)","journal-title":"ACM Trans. Graph. (TOG)"},{"issue":"6","key":"2_CR71","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1145\/3355089.3356505","volume":"38","author":"S Starke","year":"2019","unstructured":"Starke, S., Zhang, H., Komura, T., Saito, J.: Neural state machine for character-scene interactions. ACM Trans. Graph. 38(6), 209\u20131 (2019)","journal-title":"ACM Trans. Graph."},{"issue":"4","key":"2_CR72","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3450626.3459881","volume":"40","author":"S Starke","year":"2021","unstructured":"Starke, S., Zhao, Y., Zinno, F., Komura, T.: Neural animation layering for synthesizing martial arts movements. ACM Trans. Graph. (TOG) 40(4), 1\u201316 (2021)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR73","doi-asserted-by":"crossref","unstructured":"Sun, Q., et\u00a0al.: AIOS: all-in-one-stage expressive human pose and shape estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1834\u20131843 (2024)","DOI":"10.1109\/CVPR52733.2024.00180"},{"key":"2_CR74","doi-asserted-by":"crossref","unstructured":"Tessler, C., Kasten, Y., Guo, Y., Mannor, S., Chechik, G., Peng, X.B.: CALM: conditional adversarial latent models for directable virtual characters. In: ACM SIGGRAPH 2023 Conference Proceedings, pp.\u00a01\u20139 (2023)","DOI":"10.1145\/3588432.3591541"},{"key":"2_CR75","doi-asserted-by":"crossref","unstructured":"Tevet, G., Gordon, B., Hertz, A., Bermano, A.H., Cohen-Or, D.: MotionCLIP: exposing human motion generation to clip space. arXiv preprint arXiv:2203.08063 (2022)","DOI":"10.1007\/978-3-031-20047-2_21"},{"key":"2_CR76","unstructured":"Tevet, G., Raab, S., Gordon, B., Shafir, Y., Bermano, A.H., Cohen-Or, D.: Human motion diffusion model. arXiv preprint arXiv:2209.14916 (2022)"},{"key":"2_CR77","unstructured":"Touvron, H., et\u00a0al.: LLaMA: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"2_CR78","doi-asserted-by":"crossref","unstructured":"Voas, J.: What is the best automated metric for text to motion generation? Arxiv (2023). arXiv preprint arXiv:2309.10248","DOI":"10.1145\/3610548.3618185"},{"key":"2_CR79","unstructured":"Wan, W., Dou, Z., Komura, T., Wang, W., Jayaraman, D., Liu, L.: TLControl: trajectory and language control for human motion synthesis. arXiv preprint arXiv:2311.17135 (2023)"},{"key":"2_CR80","unstructured":"Wan, W., et al.: DiffusionPhase: motion diffusion in frequency domain. arXiv preprint arXiv:2312.04036 (2023)"},{"issue":"2","key":"2_CR81","doi-asserted-by":"publisher","first-page":"4702","DOI":"10.1109\/LRA.2022.3151614","volume":"7","author":"W Wan","year":"2022","unstructured":"Wan, W., et al.: Learn to predict how humans manipulate large-sized objects from interactive motions. IEEE Robot. Autom. Lett. 7(2), 4702\u20134709 (2022)","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"2","key":"2_CR82","doi-asserted-by":"publisher","first-page":"4702","DOI":"10.1109\/LRA.2022.3151614","volume":"7","author":"W Wan","year":"2022","unstructured":"Wan, W., et al.: Learn to predict how humans manipulate large-sized objects from interactive motions. IEEE Robot. Autom. Lett. 7(2), 4702\u20134709 (2022). https:\/\/doi.org\/10.1109\/LRA.2022.3151614","journal-title":"IEEE Robot. Autom. Lett."},{"key":"2_CR83","doi-asserted-by":"crossref","unstructured":"Wang, W., et al.: Zolly: zoom focal length correctly for perspective-distorted human mesh reconstruction. arXiv preprint arXiv:2303.13796 (2023)","DOI":"10.1109\/ICCV51070.2023.00363"},{"key":"2_CR84","doi-asserted-by":"crossref","unstructured":"Winkler, A., Won, J., Ye, Y.: QuestSim: human motion tracking from sparse sensors with simulated avatars. In: SIGGRAPH Asia 2022 Conference Papers, pp.\u00a01\u20138 (2022)","DOI":"10.1145\/3550469.3555411"},{"key":"2_CR85","unstructured":"Xiao, Z., Kreis, K., Vahdat, A.: Tackling the generative learning trilemma with denoising diffusion GANs. arXiv preprint arXiv:2112.07804 (2021)"},{"key":"2_CR86","unstructured":"Xie, Y., Jampani, V., Zhong, L., Sun, D., Jiang, H.: OmniControl: control any joint at any time for human motion generation. arXiv preprint arXiv:2310.08580 (2023)"},{"key":"2_CR87","doi-asserted-by":"crossref","unstructured":"Xu, L., et\u00a0al.: ActFormer: a GAN-based transformer towards general action-conditioned 3D human motion generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2228\u20132238 (2023)","DOI":"10.1109\/ICCV51070.2023.00212"},{"key":"2_CR88","doi-asserted-by":"crossref","unstructured":"Yamane, K., Revfi, M., Asfour, T.: Synthesizing object receiving motions of humanoid robots with human motion database. In: 2013 IEEE International Conference on Robotics and Automation, pp. 1629\u20131636. IEEE (2013)","DOI":"10.1109\/ICRA.2013.6630788"},{"key":"2_CR89","doi-asserted-by":"crossref","unstructured":"Yan, S., Li, Z., Xiong, Y., Yan, H., Lin, D.: Convolutional sequence generation for skeleton-based action synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4394\u20134402 (2019)","DOI":"10.1109\/ICCV.2019.00449"},{"key":"2_CR90","doi-asserted-by":"crossref","unstructured":"Yang, X., Dou, Z., Ding, Y., Su, B., Qian, H., Zhang, N.: Analysis of SARS-CoV-2 transmission in airports based on real human close contact behaviors. J. Build. Eng., 108299 (2023)","DOI":"10.1016\/j.jobe.2023.108299"},{"key":"2_CR91","doi-asserted-by":"crossref","unstructured":"Ye, Y., Liu, L., Hu, L., Xia, S.: Neural3Points: learning to generate physically realistic full-body motion for virtual reality users. In: Computer Graphics Forum, vol.\u00a041, pp. 183\u2013194. Wiley Online Library (2022)","DOI":"10.1111\/cgf.14634"},{"key":"2_CR92","unstructured":"Yu, Z., et\u00a0al.: Surf-D: high-quality surface generation for arbitrary topologies using diffusion models. arXiv preprint arXiv:2311.17050 (2023)"},{"key":"2_CR93","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Song, J., Iqbal, U., Vahdat, A., Kautz, J.: PhysDiff: physics-guided human motion diffusion model. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16010\u201316021 (2023)","DOI":"10.1109\/ICCV51070.2023.01467"},{"issue":"4","key":"2_CR94","first-page":"1","volume":"42","author":"H Zhang","year":"2023","unstructured":"Zhang, H., et al.: Learning physically simulated tennis skills from broadcast videos. ACM Trans. Graph. (TOG) 42(4), 1\u201314 (2023)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"2_CR95","doi-asserted-by":"crossref","unstructured":"Zhang, J., et al.: T2M-GPT: generating human motion from textual descriptions with discrete representations. arXiv preprint arXiv:2301.06052 (2023)","DOI":"10.1109\/CVPR52729.2023.01415"},{"key":"2_CR96","unstructured":"Zhang, J., et al.: TapMo: shape-aware motion generation of skeleton-free characters. arXiv preprint arXiv:2310.12678 (2023)"},{"key":"2_CR97","doi-asserted-by":"crossref","unstructured":"Zhang, J., et al.: Skinned motion retargeting with residual perception of motion semantics & geometry. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13864\u201313872 (2023)","DOI":"10.1109\/CVPR52729.2023.01332"},{"key":"2_CR98","unstructured":"Zhang, M., et al.: MotionDiffuse: text-driven human motion generation with diffusion model. arXiv preprint arXiv:2208.15001 (2022)"},{"key":"2_CR99","doi-asserted-by":"crossref","unstructured":"Zhang, M., et al.: RemoDiffuse: retrieval-augmented motion diffusion model. arXiv preprint arXiv:2304.01116 (2023)","DOI":"10.1109\/ICCV51070.2023.00040"},{"key":"2_CR100","doi-asserted-by":"publisher","DOI":"10.1016\/j.jhazmat.2023.132069","volume":"458","author":"N Zhang","year":"2023","unstructured":"Zhang, N., et al.: Close contact behaviors of university and school students in 10 indoor environments. J. Hazard. Mater. 458, 132069 (2023)","journal-title":"J. Hazard. Mater."},{"key":"2_CR101","doi-asserted-by":"crossref","DOI":"10.1007\/978-981-99-2792-0","volume":"99","author":"N Zhang","year":"2023","unstructured":"Zhang, N., Liu, X., Gao, S., Su, B., Dou, Z.: Popularization of high-speed railway reduces the infection risk via close contact route during journey. Sustain. Urban Areas 99, 104979 (2023)","journal-title":"Sustain. Urban Areas"},{"key":"2_CR102","unstructured":"Zhang, Y., Black, M.J., Tang, S.: Perpetual motion: generating unbounded human motion. arXiv preprint arXiv:2007.13886 (2020)"},{"key":"2_CR103","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Black, M.J., Tang, S.: We are more than our joints: predicting how 3D bodies move. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3372\u20133382 (2021)","DOI":"10.1109\/CVPR46437.2021.00338"},{"key":"2_CR104","unstructured":"Zhang, Y., et al.: MotionGPT: finetuned LLMs are general-purpose motion generators. arXiv preprint arXiv:2306.10900 (2023)"},{"key":"2_CR105","doi-asserted-by":"crossref","unstructured":"Zhao, R., Su, H., Ji, Q.: Bayesian adversarial human motion synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6225\u20136234 (2020)","DOI":"10.1109\/CVPR42600.2020.00626"},{"key":"2_CR106","doi-asserted-by":"crossref","unstructured":"Zhu, L., Liu, X., Liu, X., Qian, R., Liu, Z., Yu, L.: Taming diffusion models for audio-driven co-speech gesture generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10544\u201310553 (2023)","DOI":"10.1109\/CVPR52729.2023.01016"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72627-9_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,19]],"date-time":"2024-10-19T21:02:49Z","timestamp":1729371769000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72627-9_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,20]]},"ISBN":["9783031726262","9783031726279"],"references-count":106,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72627-9_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,20]]},"assertion":[{"value":"20 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}