{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:10:47Z","timestamp":1784268647392,"version":"3.55.0"},"reference-count":97,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,9,28]],"date-time":"2025-09-28T00:00:00Z","timestamp":1759017600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,28]],"date-time":"2025-09-28T00:00:00Z","timestamp":1759017600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s11432-024-4592-3","type":"journal-article","created":{"date-parts":[[2025,10,3]],"date-time":"2025-10-03T06:10:32Z","timestamp":1759471832000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":25,"title":["UniAnimate: taming unified video diffusion models for consistent human image animation"],"prefix":"10.1007","volume":"68","author":[{"given":"Xiang","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiwei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Changxin","family":"Gao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiayu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoqiang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingya","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luxin","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nong","family":"Sang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,28]]},"reference":[{"key":"4592_CR1","first-page":"201","volume-title":"Proceedings of European Conference on Computer Vision","author":"C Yang","year":"2018","unstructured":"Yang C, Wang Z, Zhu X, et al. Pose guided human video generation. In: Proceedings of European Conference on Computer Vision, 2018. 201\u2013216"},{"key":"4592_CR2","volume-title":"Dwnet: dense warp-based network for pose-guided human video generation","author":"P Zablotskaia","year":"2019","unstructured":"Zablotskaia P, Siarohin A, Zhao B, et al. Dwnet: dense warp-based network for pose-guided human video generation. 2019. ArXiv:1910.09139"},{"key":"4592_CR3","first-page":"8153","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"L Hu","year":"2024","unstructured":"Hu L. Animate Anyone: consistent and controllable image-to-video synthesis for character animation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 8153\u20138163"},{"key":"4592_CR4","volume-title":"MagicAnimate: temporally consistent human image animation using diffusion model","author":"Z Xu","year":"2023","unstructured":"Xu Z, Zhang J, Liew J H, et al. MagicAnimate: temporally consistent human image animation using diffusion model. 2023. ArXiv:2311.16498"},{"key":"4592_CR5","volume-title":"Magicdance: realistic human dance video generation with motions & facial expressions transfer","author":"D Chang","year":"2023","unstructured":"Chang D, Shi Y, Gao Q, et al. Magicdance: realistic human dance video generation with motions & facial expressions transfer. 2023. ArXiv:2311.12052"},{"key":"4592_CR6","first-page":"1","volume":"41","author":"Y Jiang","year":"2022","unstructured":"Jiang Y, Yang S, Qiu H, et al. Text2human: text-driven controllable human image generation. ACM Trans Graphics, 2022, 41: 1\u201311","journal-title":"ACM Trans Graphics"},{"key":"4592_CR7","volume-title":"Proceedings of International Conference on Learning Representations","author":"W Hong","year":"2023","unstructured":"Hong W, Ding M, Zheng W, et al. Cogvideo: large-scale pretraining for text-to-video generation via Transformers. In: Proceedings of International Conference on Learning Representations, 2023"},{"key":"4592_CR8","volume-title":"Proceedings of International Conference on Learning Representations","author":"U Singer","year":"2023","unstructured":"Singer U, Polyak A, Hayes T, et al. Make-A-Video: text-to-video generation without text-video data. In: Proceedings of International Conference on Learning Representations, 2023"},{"key":"4592_CR9","volume-title":"Imagen video: high definition video generation with diffusion models","author":"J Ho","year":"2022","unstructured":"Ho J, Chan W, Saharia C, et al. Imagen video: high definition video generation with diffusion models. 2022. ArXiv:2210.02303"},{"key":"4592_CR10","first-page":"10209","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Luo","year":"2023","unstructured":"Luo Z, Chen D, Zhang Y, et al. Videofusion: decomposed diffusion models for high-quality video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 10209\u201310218"},{"key":"4592_CR11","volume-title":"I2VGen-XL: high-quality image-to-video synthesis via cascaded diffusion models","author":"S Zhang","year":"2023","unstructured":"Zhang S, Wang J, Zhang Y, et al. I2VGen-XL: high-quality image-to-video synthesis via cascaded diffusion models. 2023. ArXiv:2311.04145"},{"key":"4592_CR12","first-page":"1526","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"S Tulyakov","year":"2018","unstructured":"Tulyakov S, Liu M Y, Yang X, et al. MocoGAN: decomposing motion and content for video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2018. 1526\u20131535"},{"key":"4592_CR13","volume-title":"Dreamtalk: when expressive talking head generation meets diffusion probabilistic models","author":"Y Ma","year":"2023","unstructured":"Ma Y, Zhang S, Wang J, et al. Dreamtalk: when expressive talking head generation meets diffusion probabilistic models. 2023. ArXiv:2312.09767"},{"key":"4592_CR14","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho J, Jain A, Abbeel P. Denoising diffusion probabilistic models. In: Proceedings of Advances in Neural Information Processing Systems, 2020. 33: 6840\u20136851","journal-title":"Proceedings of Advances in Neural Information Processing Systems"},{"key":"4592_CR15","volume-title":"Proceedings of International Conference on Learning Representations","author":"Y Guo","year":"2024","unstructured":"Guo Y, Yang C, Rao A, et al. Animatediff: animate your personalized text-to-image diffusion models without specific tuning. In: Proceedings of International Conference on Learning Representations, 2024"},{"key":"4592_CR16","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"X Wang","year":"2023","unstructured":"Wang X, Yuan H, Zhang S, et al. Videocomposer: compositional video synthesis with motion controllability. In: Proceedings of Advances in Neural Information Processing Systems, 2023"},{"key":"4592_CR17","volume-title":"Videocrafter1: open diffusion models for high-quality video generation","author":"H Chen","year":"2023","unstructured":"Chen H, Xia M, He Y, et al. Videocrafter1: open diffusion models for high-quality video generation. 2023. ArXiv:2310.19512"},{"key":"4592_CR18","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Wang","year":"2024","unstructured":"Wang X, Zhang S, Yuan H, et al. A recipe for scaling up text-to-video generation with text-free videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024"},{"key":"4592_CR19","volume-title":"Videolcm: video latent consistency model","author":"X Wang","year":"2023","unstructured":"Wang X, Zhang S, Zhang H, et al. Videolcm: video latent consistency model. 2023. ArXiv:2312.09109"},{"key":"4592_CR20","volume-title":"Controlvideo: training-free controllable text-to-video generation","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Wei Y, Jiang D, et al. Controlvideo: training-free controllable text-to-video generation. 2023. ArXiv:2305.13077"},{"key":"4592_CR21","volume-title":"Controlvideo: adding conditional control for one shot text-to-video editing","author":"M Zhao","year":"2023","unstructured":"Zhao M, Wang R, Bao F, et al. Controlvideo: adding conditional control for one shot text-to-video editing. 2023. ArXiv:2305.17098"},{"key":"4592_CR22","first-page":"5264","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Wang","year":"2020","unstructured":"Wang Y, Bilinski P, Bremond F, et al. G3an: disentangling appearance and motion for video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020. 5264\u20135273"},{"key":"4592_CR23","first-page":"7502","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"W Y Yu","year":"2023","unstructured":"Yu W Y, Po L M, Cheung R C, et al. Bidirectionally deformable motion modulation for video-based human pose transfer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 7502\u20137512"},{"key":"4592_CR24","first-page":"7713","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"P Zhang","year":"2022","unstructured":"Zhang P, Yang L, Lai J H, et al. Exploring dual-task correlation for pose guided person image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 7713\u20137722"},{"key":"4592_CR25","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"I Goodfellow","year":"2014","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, et al. Generative adversarial nets. In: Proceedings of Advances in Neural Information Processing Systems, 2014"},{"key":"4592_CR26","volume-title":"Proceedings of International Conference on Learning Representations","author":"T Wang","year":"2024","unstructured":"Wang T, Li L, Lin K, et al. Disco: disentangled control for referring human dance generation in real world. In: Proceedings of International Conference on Learning Representations, 2024"},{"key":"4592_CR27","volume-title":"Champ: controllable and consistent human image animation with 3D parametric guidance","author":"S Zhu","year":"2024","unstructured":"Zhu S, Chen J L, Dai Z, et al. Champ: controllable and consistent human image animation with 3D parametric guidance. 2024. ArXiv:2403.14781"},{"key":"4592_CR28","first-page":"4117","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"Y Ma","year":"2024","unstructured":"Ma Y, He Y, Cun X, et al. Follow your pose: pose-guided text-to-video generation using pose-free videos. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2024. 4117\u20134125"},{"key":"4592_CR29","first-page":"22680","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"J Karras","year":"2023","unstructured":"Karras J, Holynski A, Wang T C, et al. Dreampose: fashion video synthesis with stable diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 22680\u201322690"},{"key":"4592_CR30","first-page":"3836","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"L Zhang","year":"2023","unstructured":"Zhang L, Rao A, Agrawala M. Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 3836\u20133847"},{"key":"4592_CR31","first-page":"22563","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"A Blattmann","year":"2023","unstructured":"Blattmann A, Rombach R, Ling H, et al. Align your latents: high-resolution video synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 22563\u201322575"},{"key":"4592_CR32","volume-title":"Mamba: linear-time sequence modeling with selective state spaces","author":"A Gu","year":"2023","unstructured":"Gu A, Dao T. Mamba: linear-time sequence modeling with selective state spaces. 2023. ArXiv:2312.00752"},{"key":"4592_CR33","volume-title":"Vision mamba: efficient visual representation learning with bidirectional state space model","author":"L Zhu","year":"2024","unstructured":"Zhu L, Liao B, Zhang Q, et al. Vision mamba: efficient visual representation learning with bidirectional state space model. 2024. ArXiv:2401.09417"},{"key":"4592_CR34","volume-title":"Videomamba: state space model for efficient video understanding","author":"K Li","year":"2024","unstructured":"Li K, Li X, Wang Y, et al. Videomamba: state space model for efficient video understanding. 2024. ArXiv:2403.06977"},{"key":"4592_CR35","first-page":"10684","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"R Rombach","year":"2022","unstructured":"Rombach R, Blattmann A, Lorenz D, et al. High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 10684\u201310695"},{"key":"4592_CR36","first-page":"16784","volume-title":"Proceedings of International Conference on Machine Learning","author":"A Q Nichol","year":"2022","unstructured":"Nichol A Q, Dhariwal P, Ramesh A, et al. Glide: towards photorealistic image generation and editing with text-guided diffusion models. In: Proceedings of International Conference on Machine Learning, 2022. 16784\u201316804"},{"key":"4592_CR37","volume-title":"Hierarchical text-conditional image generation with clip latents","author":"A Ramesh","year":"2022","unstructured":"Ramesh A, Dhariwal P, Nichol A, et al. Hierarchical text-conditional image generation with clip latents. 2022. ArXiv:2204.06125"},{"key":"4592_CR38","volume-title":"T2I-adapter: learning adapters to dig out more controllable ability for text-to-image diffusion models","author":"C Mou","year":"2023","unstructured":"Mou C, Wang X, Xie L, et al. T2I-adapter: learning adapters to dig out more controllable ability for text-to-image diffusion models. 2023. ArXiv:2302.08453"},{"key":"4592_CR39","volume-title":"Proceedings of International Conference on Machine Learning","author":"L Huang","year":"2023","unstructured":"Huang L, Chen D, Liu Y, et al. Composer: creative and controllable image synthesis with composable conditions. In: Proceedings of International Conference on Machine Learning, 2023"},{"key":"4592_CR40","volume-title":"Proceedings of International Conference on Learning Representations","author":"J Song","year":"2021","unstructured":"Song J, Meng C, Ermon S. Denoising diffusion implicit models. In: Proceedings of International Conference on Learning Representations, 2021"},{"key":"4592_CR41","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia C, Chan W, Saxena S, et al. Photorealistic text-to-image diffusion models with deep language understanding. In: Proceedings of Advances in Neural Information Processing Systems, 2022. 35: 36479\u201336494","journal-title":"Proceedings of Advances in Neural Information Processing Systems"},{"key":"4592_CR42","doi-asserted-by":"publisher","first-page":"151101","DOI":"10.1007\/s11432-022-3679-0","volume":"66","author":"M Liu","year":"2023","unstructured":"Liu M, Wei Y X, Wu X H, et al. Survey on leveraging pre-trained generative adversarial networks for image editing and restoration. Sci China Inf Sci, 2023, 66: 151101","journal-title":"Sci China Inf Sci"},{"key":"4592_CR43","volume-title":"Modelscope text-to-video technical report","author":"J Wang","year":"2023","unstructured":"Wang J, Yuan H, Chen D, et al. Modelscope text-to-video technical report. 2023. ArXiv:2308.06571"},{"key":"4592_CR44","first-page":"234","volume-title":"Proceedings of International Conference on Medical Image Computing and Computer-Assisted Intervention","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger O, Fischer P, Brox T. U-net: convolutional networks for biomedical image segmentation. In: Proceedings of International Conference on Medical Image Computing and Computer-Assisted Intervention, 2015. 234\u2013241"},{"key":"4592_CR45","first-page":"7623","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"J Z Wu","year":"2023","unstructured":"Wu J Z, Ge Y, Wang X, et al. Tune-a-video: one-shot tuning of image diffusion models for text-to-video generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 7623\u20137633"},{"key":"4592_CR46","first-page":"23040","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"W Chai","year":"2023","unstructured":"Chai W, Guo X, Wang G, et al. Stablevideo: text-driven consistency-aware diffusion video editing. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 23040\u201323050"},{"key":"4592_CR47","first-page":"23206","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"D Ceylan","year":"2023","unstructured":"Ceylan D, Huang C H P, Mitra N J. Pix2video: video editing using image diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 23206\u201323217"},{"key":"4592_CR48","volume-title":"Magicvideo: efficient video generation with latent diffusion models","author":"D Zhou","year":"2022","unstructured":"Zhou D, Wang W, Yan H, et al. Magicvideo: efficient video generation with latent diffusion models. 2022. ArXiv:2211.11018"},{"key":"4592_CR49","volume-title":"Latent-shift: latent diffusion with temporal shift for efficient text-to-video generation","author":"J An","year":"2023","unstructured":"An J, Zhang S, Yang H, et al. Latent-shift: latent diffusion with temporal shift for efficient text-to-video generation. 2023. ArXiv:2304.08477"},{"key":"4592_CR50","volume-title":"Simda: simple diffusion adapter for efficient video generation","author":"Z Xing","year":"2023","unstructured":"Xing Z, Dai Q, Hu H, et al. Simda: simple diffusion adapter for efficient video generation. 2023. ArXiv:2308.09710"},{"key":"4592_CR51","volume-title":"Hierarchical spatio-temporal decoupling for text-to-video generation","author":"Z Qing","year":"2023","unstructured":"Qing Z, Zhang S, Wang J, et al. Hierarchical spatio-temporal decoupling for text-to-video generation. 2023. ArXiv:2312.04483"},{"key":"4592_CR52","volume-title":"Instructvideo: instructing video diffusion models with human feedback","author":"H Yuan","year":"2023","unstructured":"Yuan H, Zhang S, Wang X, et al. Instructvideo: instructing video diffusion models with human feedback. 2023. ArXiv:2312.12490"},{"key":"4592_CR53","volume-title":"Dreamvideo: composing your dream videos with customized subject and motion","author":"Y Wei","year":"2023","unstructured":"Wei Y, Zhang S, Qing Z, et al. Dreamvideo: composing your dream videos with customized subject and motion. 2023. ArXiv:2312.04433"},{"key":"4592_CR54","first-page":"7310","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"H Chen","year":"2024","unstructured":"Chen H, Zhang Y, Cun X, et al. Videocrafter2: overcoming data limitations for high-quality video diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024. 7310\u20137320"},{"key":"4592_CR55","volume-title":"Lavie: high-quality video generation with cascaded latent diffusion models","author":"Y Wang","year":"2023","unstructured":"Wang Y, Chen X, Ma X, et al. Lavie: high-quality video generation with cascaded latent diffusion models. 2023. ArXiv:2309.15103"},{"key":"4592_CR56","doi-asserted-by":"publisher","first-page":"1879","DOI":"10.1007\/s11263-024-02271-9","volume":"133","author":"D J Zhang","year":"2025","unstructured":"Zhang D J, Wu J Z, Liu J W, et al. Show-1: marrying pixel and latent diffusion models for text-to-video generation. Int J Comput Vis, 2025, 133: 1879\u20131893","journal-title":"Int J Comput Vis"},{"key":"4592_CR57","volume-title":"Proceedings of International Conference on Learning Representations","author":"X Chen","year":"2024","unstructured":"Chen X, Wang Y, Zhang L, et al. Seine: short-to-long video diffusion model for generative transition and prediction. In: Proceedings of International Conference on Learning Representations, 2024"},{"key":"4592_CR58","first-page":"7346","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"P Esser","year":"2023","unstructured":"Esser P, Chiu J, Atighehchian P, et al. Structure and content-guided video synthesis with diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 7346\u20137356"},{"key":"4592_CR59","volume-title":"Make-your-video: customized video generation using textual and structural guidance","author":"J Xing","year":"2023","unstructured":"Xing J, Xia M, Liu Y, et al. Make-your-video: customized video generation using textual and structural guidance. 2023. ArXiv:2306.00943"},{"key":"4592_CR60","volume-title":"Dragnuwa: fine-grained control in video generation by integrating text, image, and trajectory","author":"S Yin","year":"2023","unstructured":"Yin S, Wu C, Liang J, et al. Dragnuwa: fine-grained control in video generation by integrating text, image, and trajectory. 2023. ArXiv:2308.08089"},{"key":"4592_CR61","volume-title":"Motion-conditioned diffusion model for controllable video synthesis","author":"T S Chen","year":"2023","unstructured":"Chen T S, Lin C H, Tseng H Y, et al. Motion-conditioned diffusion model for controllable video synthesis. 2023. ArXiv:2304.14404"},{"key":"4592_CR62","first-page":"32","volume-title":"Proceedings of Advances in Neural Information Processing Systems","author":"A Siarohin","year":"2019","unstructured":"Siarohin A, Lathuili\u00e8re S, Tulyakov S, et al. First order motion model for image animation. In: Proceedings of Advances in Neural Information Processing Systems, 2019. 32"},{"key":"4592_CR63","first-page":"3693","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Li","year":"2019","unstructured":"Li Y, Huang C, Loy C C. Dense intrinsic appearance flow for human pose transfer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019. 3693\u20133702"},{"key":"4592_CR64","first-page":"13653","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"A Siarohin","year":"2021","unstructured":"Siarohin A, Woodford O J, Ren J, et al. Motion representations for articulated animation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021. 13653\u201313662"},{"key":"4592_CR65","first-page":"3657","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Zhao","year":"2022","unstructured":"Zhao J, Zhang H. Thin-plate spline motion model for image animation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 3657\u20133666"},{"key":"4592_CR66","volume-title":"Do you guys want to dance: zero-shot compositional human dance generation with multiple persons","author":"Z Xu","year":"2024","unstructured":"Xu Z, Wei K, Yang X, et al. Do you guys want to dance: zero-shot compositional human dance generation with multiple persons. 2024. ArXiv:2401.13363"},{"key":"4592_CR67","volume-title":"Poseanimate: zero-shot high fidelity pose controllable character animation","author":"B Zhu","year":"2024","unstructured":"Zhu B, Wang F, Lu T, et al. Poseanimate: zero-shot high fidelity pose controllable character animation. 2024. ArXiv:2404.13680"},{"key":"4592_CR68","first-page":"156","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"C Lea","year":"2017","unstructured":"Lea C, Flynn M D, Vidal R, et al. Temporal convolutional networks for action segmentation and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2017. 156\u2013165"},{"key":"4592_CR69","first-page":"7565","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"X Wang","year":"2021","unstructured":"Wang X, Zhang S, Qing Z, et al. Oadtr: online action detection with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021. 7565\u20137575"},{"key":"4592_CR70","first-page":"6836","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"A Arnab","year":"2021","unstructured":"Arnab A, Dehghani M, Heigold G, et al. Vivit: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021. 6836\u20136846"},{"key":"4592_CR71","first-page":"1905","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X Wang","year":"2021","unstructured":"Wang X, Zhang S, Qing Z, et al. Self-supervised learning for semi-supervised temporal action proposal. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021. 1905\u20131914"},{"key":"4592_CR72","first-page":"5533","volume-title":"In: Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Z Qiu","year":"2017","unstructured":"Qiu Z, Yao T, Mei T. Learning spatio-temporal representation with pseudo-3D residual networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2017. 5533\u20135541"},{"key":"4592_CR73","first-page":"2024","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"S Gupta","year":"2022","unstructured":"Gupta S, Keshari A, Das S. RV-GAN: recurrent gan for unconditional video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 2024\u20132033"},{"key":"4592_CR74","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","author":"Y Li","year":"2018","unstructured":"Li Y, Min M, Shen D, et al. Video generation from text. In: Proceedings of the AAAI Conference on Artificial Intelligence, 2018"},{"key":"4592_CR75","first-page":"4","volume-title":"Proceedings of International Conference on Machine Learning","author":"G Bertasius","year":"2021","unstructured":"Bertasius G, Wang H, Torresani L. Is space-time attention all you need for video understanding? In: Proceedings of International Conference on Machine Learning, 2021. 4"},{"key":"4592_CR76","doi-asserted-by":"publisher","first-page":"160104","DOI":"10.1007\/s11432-021-3445-y","volume":"65","author":"D K Liang","year":"2022","unstructured":"Liang D K, Chen X W, Xu W, et al. TransCrowd: weakly-supervised crowd counting with transformers. Sci China Inf Sci, 2022, 65: 160104","journal-title":"Sci China Inf Sci"},{"key":"4592_CR77","doi-asserted-by":"publisher","first-page":"202102","DOI":"10.1007\/s11432-022-3783-3","volume":"66","author":"K Li","year":"2023","unstructured":"Li K, Guo D, Wang M. ViGT: proposal-free video grounding with a learnable token in the transformer. Sci China Inf Sci, 2023, 66: 202102","journal-title":"Sci China Inf Sci"},{"key":"4592_CR78","doi-asserted-by":"publisher","first-page":"152102","DOI":"10.1007\/s11432-021-3536-5","volume":"67","author":"Y F Shao","year":"2024","unstructured":"Shao Y F, Geng Z C, Liu Y T, et al. CPT: a pre-trained unbalanced transformer for both Chinese language understanding and generation. Sci China Inf Sci, 2024, 67: 152102","journal-title":"Sci China Inf Sci"},{"key":"4592_CR79","doi-asserted-by":"publisher","first-page":"210102","DOI":"10.1007\/s11432-022-3700-8","volume":"66","author":"H X Chen","year":"2023","unstructured":"Chen H X, Li H X, Li Y H, et al. Sparse spatial transformers for few-shot learning. Sci China Inf Sci, 2023, 66: 210102","journal-title":"Sci China Inf Sci"},{"key":"4592_CR80","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Li","year":"2024","unstructured":"Li Z, Yang B, Liu Q, et al. Monkey: image resolution and text label are important things for large multi-modal models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024"},{"key":"4592_CR81","volume-title":"Textmonkey: an OCR-free large multimodal model for understanding document","author":"Y Liu","year":"2024","unstructured":"Liu Y, Yang B, Liu Q, et al. Textmonkey: an OCR-free large multimodal model for understanding document. 2024. ArXiv:2403.04473"},{"key":"4592_CR82","volume-title":"Few-shot action recognition with captioning foundation models","author":"X Wang","year":"2023","unstructured":"Wang X, Zhang S, Yuan H, et al. Few-shot action recognition with captioning foundation models. 2023. ArXiv:2310.10125"},{"key":"4592_CR83","doi-asserted-by":"publisher","first-page":"1899","DOI":"10.1007\/s11263-023-01917-4","volume":"132","author":"X Wang","year":"2024","unstructured":"Wang X, Zhang S, Cen J, et al. CLIP-guided prototype modulating for few-shot action recognition. Int J Comput Vis, 2024, 132: 1899\u20131912","journal-title":"Int J Comput Vis"},{"key":"4592_CR84","volume-title":"Efficiently modeling long sequences with structured state spaces","author":"A Gu","year":"2021","unstructured":"Gu A, Goel K, R\u00e9 C. Efficiently modeling long sequences with structured state spaces. 2021. ArXiv:2111.00396"},{"key":"4592_CR85","volume-title":"Vmamba: visual state space model","author":"Y Liu","year":"2024","unstructured":"Liu Y, Tian Y, Zhao Y, et al. Vmamba: visual state space model. 2024. ArXiv:2401.10166"},{"key":"4592_CR86","volume-title":"Plainmamba: improving non-hierarchical mamba in visual recognition","author":"C Yang","year":"2024","unstructured":"Yang C, Chen Z, Espinosa M, et al. Plainmamba: improving non-hierarchical mamba in visual recognition. 2024. ArXiv:2403.17695"},{"key":"4592_CR87","volume-title":"Video mamba suite: state space model as a versatile alternative for video understanding","author":"G Chen","year":"2024","unstructured":"Chen G, Huang Y, Xu J, et al. Video mamba suite: state space model as a versatile alternative for video understanding. 2024. ArXiv:2403.09626"},{"key":"4592_CR88","first-page":"12753","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Jafarian","year":"2021","unstructured":"Jafarian Y, Park H S. Learning high fidelity depths of dressed humans by watching social media dance videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021. 12753\u201312762"},{"key":"4592_CR89","first-page":"4210","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Z Yang","year":"2023","unstructured":"Yang Z, Zeng A, Yuan C, et al. Effective whole-body pose estimation with two-stages distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 4210\u20134220"},{"key":"4592_CR90","first-page":"8748","volume-title":"Proceedings of International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford A, Kim J W, Hallacy C, et al. Learning transferable visual models from natural language supervision. In: Proceedings of International Conference on Machine Learning, 2021. 8748\u20138763"},{"key":"4592_CR91","volume-title":"Decoupled weight decay regularization","author":"I Loshchilov","year":"2017","unstructured":"Loshchilov I, Hutter F. Decoupled weight decay regularization. 2017. ArXiv:1711.05101"},{"key":"4592_CR92","first-page":"2366","volume-title":"Proceedings of International Conference on Pattern Recognition","author":"A Hore","year":"2010","unstructured":"Hore A, Ziou D. Image quality metrics: PSNR vs. SSIM. In: Proceedings of International Conference on Pattern Recognition, 2010. 2366\u20132369"},{"key":"4592_CR93","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang Z, Bovik A C, Sheikh H R, et al. Image quality assessment: from error visibility to structural similarity. IEEE Trans Image Process, 2004, 13: 600\u2013612","journal-title":"IEEE Trans Image Process"},{"key":"4592_CR94","first-page":"586","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"R Zhang","year":"2018","unstructured":"Zhang R, Isola P, Efros A A, et al. The unreasonable effectiveness of deep features as a perceptual metric. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2018. 586\u2013595"},{"key":"4592_CR95","volume-title":"Towards accurate generative models of video: a new metric & challenges","author":"T Unterthiner","year":"2018","unstructured":"Unterthiner T, van Steenkiste S, Kurach K, et al. Towards accurate generative models of video: a new metric & challenges. 2018. ArXiv:1812.01717"},{"key":"4592_CR96","first-page":"13535","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Ren","year":"2022","unstructured":"Ren Y, Fan X, Li G, et al. Neural texture extraction and distribution for controllable person image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 13535\u201313544"},{"key":"4592_CR97","first-page":"5968","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"A K Bhunia","year":"2023","unstructured":"Bhunia A K, Khan S, Cholakkal H, et al. Person image synthesis via denoising diffusion model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 5968\u20135976"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4592-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4592-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4592-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,3]],"date-time":"2025-10-03T07:03:57Z","timestamp":1759475037000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4592-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,28]]},"references-count":97,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["4592"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4592-3","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,28]]},"assertion":[{"value":"15 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 November 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 March 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 September 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"200103"}}