{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:50:55Z","timestamp":1778082655072,"version":"3.51.4"},"publisher-location":"Cham","reference-count":55,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726484","type":"print"},{"value":"9783031726491","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72649-1_27","type":"book-chapter","created":{"date-parts":[[2024,9,29]],"date-time":"2024-09-29T07:01:50Z","timestamp":1727593310000},"page":"475-491","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":20,"title":["LivePhoto: Real Image Animation with\u00a0Text-Guided Motion Control"],"prefix":"10.1007","author":[{"given":"Xi","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiheng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mengting","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yutong","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yujun","family":"Shen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hengshuang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,30]]},"reference":[{"key":"27_CR1","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: a joint video and image encoder for end-to-end retrieval. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"27_CR2","doi-asserted-by":"crossref","unstructured":"Blattmann, A., et al.: Align your latents: high-resolution video synthesis with latent diffusion models. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02161"},{"key":"27_CR3","doi-asserted-by":"crossref","unstructured":"Chai, W., Guo, X., Wang, G., Lu, Y.: StableVideo: text-driven consistency-aware diffusion video editing. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.02106"},{"key":"27_CR4","unstructured":"Chen, J., et\u00a0al.: PixArt: fast training of diffusion transformer for photorealistic text-to-image synthesis. arXiv:2310.00426 (2023)"},{"key":"27_CR5","unstructured":"Chen, T.S., Lin, C.H., Tseng, H.Y., Lin, T.Y., Yang, M.H.: Motion-conditioned diffusion model for controllable video synthesis. arXiv:2304.14404 (2023)"},{"key":"27_CR6","doi-asserted-by":"crossref","unstructured":"Chen, X., Huang, L., Liu, Y., Shen, Y., Zhao, D., Zhao, H.: AnyDoor: zero-shot object-level image customization. arXiv:2307.09481 (2023)","DOI":"10.1109\/CVPR52733.2024.00630"},{"key":"27_CR7","doi-asserted-by":"crossref","unstructured":"Cheng, C.C., Chen, H.Y., Chiu, W.C.: Time flies: animating a still image with time-lapse video as reference. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00568"},{"key":"27_CR8","doi-asserted-by":"crossref","unstructured":"Esser, P., Chiu, J., Atighehchian, P., Granskog, J., Germanidis, A.: Structure and content-guided video synthesis with diffusion models. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00675"},{"key":"27_CR9","unstructured":"Gal, R., et al.: An image is worth one word: personalizing text-to-image generation using textual inversion. arXiv:2208.01618 (2022)"},{"key":"27_CR10","unstructured":"Guo, Y., et al.: AnimateDiff: animate your personalized text-to-image diffusion models without specific tuning. arXiv:2307.04725 (2023)"},{"key":"27_CR11","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: NeurIPS (2020)"},{"key":"27_CR12","unstructured":"Ho, J., Salimans, T., Gritsenko, A., Chan, W., Norouzi, M., Fleet, D.J.: Video diffusion models. arXiv:2204.03458 (2022)"},{"key":"27_CR13","doi-asserted-by":"crossref","unstructured":"Holynski, A., Curless, B.L., Seitz, S.M., Szeliski, R.: Animating pictures with Eulerian motion fields. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00575"},{"key":"27_CR14","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models. arXiv:2106.09685 (2021)"},{"key":"27_CR15","doi-asserted-by":"crossref","unstructured":"Hu, Y., Luo, C., Chen, Z.: Make it move: controllable image-to-video generation with text descriptions. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01768"},{"issue":"1","key":"27_CR16","first-page":"4","volume":"18","author":"WC Jhou","year":"2015","unstructured":"Jhou, W.C., Cheng, W.H.: Animating still landscape photographs through cloud motion creation. TMM 18(1), 4\u201313 (2015)","journal-title":"TMM"},{"key":"27_CR17","doi-asserted-by":"crossref","unstructured":"Karras, J., Holynski, A., Wang, T.C., Kemelmacher-Shlizerman, I.: DreamPose: fashion image-to-video synthesis via stable diffusion. arXiv:2304.06025 (2023)","DOI":"10.1109\/ICCV51070.2023.02073"},{"key":"27_CR18","doi-asserted-by":"crossref","unstructured":"Kawar, B., et al.: Imagic: text-based real image editing with diffusion models. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"27_CR19","doi-asserted-by":"crossref","unstructured":"Khachatryan, L., et al.: Text2Video-Zero: text-to-image diffusion models are zero-shot video generators. arXiv:2303.13439 (2023)","DOI":"10.1109\/ICCV51070.2023.01462"},{"key":"27_CR20","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational Bayes. arXiv:1312.6114 (2013)"},{"key":"27_CR21","doi-asserted-by":"crossref","unstructured":"Li, Z., Tucker, R., Snavely, N., Holynski, A.: Generative image dynamics. arXiv:2309.07906 (2023)","DOI":"10.1109\/CVPR52733.2024.02279"},{"key":"27_CR22","unstructured":"Liew, J.H., Yan, H., Zhang, J., Xu, Z., Feng, J.: MagicEdit: high-fidelity and temporally coherent video editing. arXiv:2308.14749 (2023)"},{"key":"27_CR23","unstructured":"Liu, Z., et al.: Cones: concept neurons in diffusion models for customized generation. arXiv:2303.05125 (2023)"},{"key":"27_CR24","unstructured":"Liu, Z., et al.: Cones 2: customizable image synthesis with multiple subjects. arXiv:2305.19327 (2023)"},{"key":"27_CR25","unstructured":"Luan, T.: AnimateDiff-I2V (2023). https:\/\/github.com\/ykk648\/AnimateDiff-I2V"},{"key":"27_CR26","doi-asserted-by":"crossref","unstructured":"Mahapatra, A., Kulkarni, K.: Controllable animation of fluid elements in still images. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00365"},{"key":"27_CR27","unstructured":"Meng, C., et al.: SDEdit: guided image synthesis and editing with stochastic differential equations. arXiv:2108.01073 (2021)"},{"key":"27_CR28","doi-asserted-by":"crossref","unstructured":"Mou, C., et al.: T2I-Adapter: learning adapters to dig out more controllable ability for text-to-image diffusion models. arXiv:2302.08453 (2023)","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"27_CR29","doi-asserted-by":"crossref","unstructured":"Okabe, M., Anjyo, K., Igarashi, T., Seidel, H.P.: Animating pictures of fluid using video examples. In: Computer Graphics Forum. Wiley Online Library (2009)","DOI":"10.1111\/j.1467-8659.2009.01408.x"},{"key":"27_CR30","unstructured":"Oquab, M., et\u00a0al.: DINOv2: learning robust visual features without supervision. arXiv:2304.07193 (2023)"},{"key":"27_CR31","unstructured":"Podell, D., et al.: SDXL: improving latent diffusion models for high-resolution image synthesis. arXiv:2307.01952 (2023)"},{"key":"27_CR32","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"27_CR33","unstructured":"Researchers, P.: PikaLabs: An innovative text-to-video platform, October 2023. https:\/\/www.pika.art\/"},{"key":"27_CR34","unstructured":"Researchers, R.: Gen-2: The next step forward for generative AI, October 2023. https:\/\/research.runwayml.com\/gen2"},{"key":"27_CR35","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"27_CR36","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: DreamBooth: fine tuning text-to-image diffusion models for subject-driven generation. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"27_CR37","doi-asserted-by":"crossref","unstructured":"Saharia, C., T., et\u00a0al.: Photorealistic text-to-image diffusion models with deep language understanding. In: NeurIPS (2022)","DOI":"10.1145\/3528233.3530757"},{"key":"27_CR38","doi-asserted-by":"crossref","unstructured":"Shalev, Y., Wolf, L.: Image animation with perturbed masks. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00363"},{"key":"27_CR39","unstructured":"Siarohin, A., Lathuili\u00e8re, S., Tulyakov, S., Ricci, E., Sebe, N.: First order motion model for image animation. In: NeurIPS (2019)"},{"key":"27_CR40","unstructured":"Singer, U., et\u00a0al.: Make-a-Video: text-to-video generation without text-video data. arXiv:2209.14792 (2022)"},{"key":"27_CR41","unstructured":"talesofai: AnimateDiff talesofai (2023). https:\/\/github.com\/talesofai\/AnimateDiff"},{"key":"27_CR42","unstructured":"Wang, T., et al.: DISCO: disentangled control for referring human dance generation in real world. arXiv:2307.00040 (2023)"},{"key":"27_CR43","unstructured":"Wang, X., et al.: VideoComposer: compositional video synthesis with motion controllability. In: NeurIPS (2023)"},{"issue":"4","key":"27_CR44","first-page":"600","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A.C., Sheikh, H.R., Simoncelli, E.P.: Image quality assessment: from error visibility to structural similarity. TIP 13(4), 600\u2013612 (2004)","journal-title":"TIP"},{"key":"27_CR45","doi-asserted-by":"crossref","unstructured":"Wu, J.Z., et al.: Tune-a-video: one-shot tuning of image diffusion models for text-to-video generation. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00701"},{"key":"27_CR46","doi-asserted-by":"crossref","unstructured":"Xing, J., et al.: DynamiCrafter: animating open-domain images with video diffusion priors. arXiv:2310.12190 (2023)","DOI":"10.1007\/978-3-031-72952-2_23"},{"key":"27_CR47","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., Rui, Y.: MSR-VTT: a large video description dataset for bridging video and language. In: CVPR, pp. 5288\u20135296 (2016)","DOI":"10.1109\/CVPR.2016.571"},{"key":"27_CR48","unstructured":"Xue, Z., et al.: RAPHAEL: text-to-image generation via large mixture of diffusion paths. In: NeurIPS (2023)"},{"key":"27_CR49","doi-asserted-by":"crossref","unstructured":"Yin, S., et\u00a0al.: NUWA-XL: diffusion over diffusion for extremely long video generation. arXiv:2303.12346 (2023)","DOI":"10.18653\/v1\/2023.acl-long.73"},{"key":"27_CR50","unstructured":"Zhang, J., Yan, H., Xu, Z., Feng, J., Liew, J.H.: MagicAvatar: multimodal avatar generation and animation. arXiv:2308.14748 (2023)"},{"key":"27_CR51","doi-asserted-by":"crossref","unstructured":"Zhang, L., Agrawala, M.: Adding conditional control to text-to-image diffusion models. arXiv:2302.05543 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"27_CR52","unstructured":"Zhang, S., et al.: I2VGen-XL: high-quality image-to-video synthesis via cascaded diffusion models. arXiv:2311.04145 (2023)"},{"key":"27_CR53","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Xing, Z., Zeng, Y., Fang, Y., Chen, K.: PIA: your personalized image animator via plug-and-play modules in text-to-image models. In: CVPR (2023)","DOI":"10.1109\/CVPR52733.2024.00740"},{"key":"27_CR54","doi-asserted-by":"crossref","unstructured":"Zhao, J., Zhang, H.: Thin-plate spline motion model for image animation. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00364"},{"key":"27_CR55","doi-asserted-by":"crossref","unstructured":"Zhao, R., Wu, T., Guo, G.: Sparse to dense motion transfer for face image animation. In: ICCV (2021)","DOI":"10.1109\/ICCVW54120.2021.00226"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72649-1_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T21:19:27Z","timestamp":1732828767000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72649-1_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,30]]},"ISBN":["9783031726484","9783031726491"],"references-count":55,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72649-1_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,30]]},"assertion":[{"value":"30 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}