{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,8]],"date-time":"2026-05-08T22:56:42Z","timestamp":1778281002251,"version":"3.51.4"},"reference-count":30,"publisher":"Tsinghua University Press","issue":"6","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2024,8,28]],"date-time":"2024-08-28T00:00:00Z","timestamp":1724803200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Comp. Visual. Med."],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s41095-023-0375-z","type":"journal-article","created":{"date-parts":[[2024,8,28]],"date-time":"2024-08-28T09:02:19Z","timestamp":1724835739000},"page":"1157-1168","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["CLIP-Flow: Decoding images encoded in CLIP space"],"prefix":"10.26599","volume":"10","author":[{"given":"Hao","family":"Ma","sequence":"first","affiliation":[{"name":"Visual Computing Research Center, College of Computer Science and Software Engineering, Shenzhen University, Shenzhen 518060, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Li","sequence":"additional","affiliation":[{"name":"Visual Computing Research Center, College of Computer Science and Software Engineering, Shenzhen University, Shenzhen 518060, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingyuan","family":"Yang","sequence":"additional","affiliation":[{"name":"Visual Computing Research Center, College of Computer Science and Software Engineering, Shenzhen University, Shenzhen 518060, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Or","family":"Patashnik","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Tel Aviv University, Tel Aviv 6997801, Israel"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dani","family":"Lischinski","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, the Hebrew University of Jerusalem, Jerusalem 91904, Israel"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daniel","family":"Cohen-Or","sequence":"additional","affiliation":[{"name":"Visual Computing Research Center, College of Computer Science and Software Engineering, Shenzhen University, Shenzhen 518060, China; Department of Computer Science, Tel Aviv University, Tel Aviv 6997801, Israel"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hui","family":"Huang","sequence":"additional","affiliation":[{"name":"Visual Computing Research Center, College of Computer Science and Software Engineering, Shenzhen University, Shenzhen 518060, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"11138","reference":[{"key":"375_CR1","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"Radford, A.; Kim, J.; Hallacy, C.; Ramesh, A.; Goh, G.; Agarwal, S.; Sastry, G.; Askell, A.; Mishkin, P.; Clark, J.; et al. Learning transferable visual models from natural language supervision. In: Proceedings of the 38th International Conference on Machine Learning, 8748\u20138763, 2021."},{"key":"375_CR2","first-page":"2065","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"O Patashnik","year":"2021","unstructured":"Patashnik, O.; Wu, Z.; Shechtman, E.; Cohen-Or, D.; Lischinski, D. StyleCLIP: Text-driven manipulation of StyleGAN imagery. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2065\u20132074, 2021."},{"key":"375_CR3","doi-asserted-by":"crossref","unstructured":"Gal, R.; Patashnik, O.; Maron, H.; Bermano, A. H.; Chechik, G.; Cohen-Or, D. StyleGAN-NADA: CLIP-guided domain adaptation of image generators. ACM Transactions on Graphics Vol. 41, No. 4, Article No. 141, 2022.","DOI":"10.1145\/3528223.3530164"},{"key":"375_CR4","first-page":"5207","volume-title":"Proceedings of the 36th Conference on Neural Information Processing System","author":"K Frans","year":"2022","unstructured":"Frans, K.; Soros, L.; Witkowski, O. CLIPDraw: Exploring text-to-drawing synthesis through language-image encoders. In: Proceedings of the 36th Conference on Neural Information Processing System, 5207\u20135218, 2022."},{"key":"375_CR5","first-page":"3835","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"C Wang","year":"2022","unstructured":"Wang, C.; Chai, M.; He, M.; Chen, D.; Liao, J. CLIP-NeRF: Text-and-image driven manipulation of neural radiance fields. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 3835\u20133844, 2022."},{"key":"375_CR6","first-page":"13492","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"O Michel","year":"2022","unstructured":"Michel, O.; Bar-On, R.; Liu, R.; Benaim, S.; Hanocka, R. Text2Mesh: Text-driven neural stylization for meshes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 13492\u201313502, 2022."},{"key":"375_CR7","first-page":"18208","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"O Avrahami","year":"2022","unstructured":"Avrahami, O.; Lischinski, D.; Fried, O. Blended diffusion for text-driven editing of natural images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 18208\u201318218, 2022."},{"key":"375_CR8","first-page":"8821","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"A Ramesh","year":"2021","unstructured":"Ramesh, A.; Pavlov, M.; Goh, G.; Gray, S.; Voss, C.; Radford, A.; Chen, M.; Sutskever, I. Zero-shot text-to-image generation. In: Proceedings of the 38th International Conference on Machine Learning, 8821\u20138831, 2021."},{"key":"375_CR9","unstructured":"Nichol, A.; Dhariwal, P.; Ramesh, A.; Shyam, P.; Mishkin, P.; McGrew, B.; Sutskever, I.; Chen, M. GLIDE: Towards photorealistic image generation and editing with text-guided diffusion models. arXiv preprint arXiv:2112.10741, 2021."},{"key":"375_CR10","first-page":"1316","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"T Xu","year":"2018","unstructured":"Xu, T.; Zhang, P.; Huang, Q.; Zhang, H.; Gan, Z.; Huang, X.; He, X. AttnGAN: Fine-grained text to image generation with attentional generative adversarial networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 1316\u20131324, 2018."},{"key":"375_CR11","first-page":"2063","volume-title":"Proceedings of the 33rd Conference on Neural Information Processing Systems","author":"B Li","year":"2019","unstructured":"Li, B.; Qi, X.; Lukasiewicz, T.; Torr, P. Controllable text-to-image generation. In: Proceedings of the 33rd Conference on Neural Information Processing Systems, 2063\u20132073, 2019."},{"key":"375_CR12","first-page":"5802","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"M Zhu","year":"2019","unstructured":"Zhu, M.; Pan, P.; Chen, W.; Yang, Y. DM-GAN: Dynamic memory generative adversarial networks for text-to-image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 5802\u20135810, 2019."},{"key":"375_CR13","first-page":"16515","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"M Tao","year":"2022","unstructured":"Tao, M.; Tang, H.; Wu, F.; Jing, X.; Bao, B. K.; Xu, C. DF-GAN: A simple and effective baseline for text-to-image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 16515\u201316525, 2022."},{"key":"375_CR14","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"707","DOI":"10.1007\/978-3-031-19784-0_41","volume-title":"Computer Vision\u2013ECCV 2022","author":"O Bar-Tal","year":"2022","unstructured":"Bar-Tal, O.; Ofri-Amar, D.; Fridman, R.; Kasten, Y.; Dekel, T. Text2LIVE: Text-driven layered image and video editing. In: Computer Vision\u2013ECCV 2022. Lecture Notes in Computer Science, Vol. 13675. Avidan, S.; Brostow, G.; Ciss\u00e9, S.; Farinella, G. M.; Hassner, T. Eds. Springer Cham, 707\u2013723, 2022."},{"key":"375_CR15","first-page":"17907","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Zhou","year":"2022","unstructured":"Zhou, Y.; Zhang, R.; Chen, C.; Li, C.; Tensmeyer, C.; Yu, T.; Gu, J.; Xu, J.; Sun, T. Towards language-free training for text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 17907\u201317917, 2022."},{"key":"375_CR16","first-page":"2426","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"G Kim","year":"2022","unstructured":"Kim, G.; Kwon, T.; Ye, J. C. DiffusionCLIP: Text-guided diffusion models for robust image manipulation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2426\u20132435, 2022."},{"key":"375_CR17","first-page":"289","volume-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","author":"X Liu","year":"2023","unstructured":"Liu, X.; Park, D. H.; Azadi, S.; Zhang, G.; Chopikyan, A.; Hu, Y.; Shi, H.; Rohrbach, A.; Darrell, T. More control for free! image synthesis with semantic diffusion guidance. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, 289\u2013299, 2023."},{"key":"375_CR18","unstructured":"Ramesh, A.; Dhariwal, P.; Nichol, A.; Chu, C.; Chen, M. Hierarchical text-conditional image generation with CLIP latents. arXiv preprint arXiv:2204.06125, 2022."},{"key":"375_CR19","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"88","DOI":"10.1007\/978-3-031-19836-6_6","volume-title":"Computer Vision\u2013ECCV 2022","author":"K Crowson","year":"2022","unstructured":"Crowson, K.; Biderman, S.; Kornis, D.; Stander, D.; Hallahan, E.; Castricato, L.; Raff, E. VQGAN-clip: Open domain image generation and editing with natural language guidance. In: Computer Vision\u2013ECCV 2022. Lecture Notes in Computer Science, Vol. 13697. Avidan, S.; Brostow, G.; Cisse, S.; Farinella, G. M.; Hassner, T. Eds. Springer Cham, 88\u2013105, 2022."},{"key":"375_CR20","unstructured":"Crowson, K. CLIP guided diffusion HQ 256x256. 2021. Available at https:\/\/colab.research.google.com\/drive\/12a_Wrfi2_gwwAuN3VvMTwVMz9TfqctNj"},{"key":"375_CR21","first-page":"18603","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"A Sanghi","year":"2022","unstructured":"Sanghi, A.; Chu, H.; Lambourne, J. G.; Wang, Y.; Cheng, C. Y.; Fumero, M.; Malekshan, K. R. CLIP-forge: Towards zero-shot text-to-shape generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 18603\u201318613, 2022."},{"key":"375_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"358","DOI":"10.1007\/978-3-031-20047-2_21","volume-title":"Computer Vision\u2013ECCV 2022","author":"G Tevet","year":"2022","unstructured":"Tevet, G.; Gordon, B.; Hertz, A.; Bermano, A. H.; Cohen-Or, D. MotionCLIP: Exposing human motion generation to CLIP space. In: Computer Vision\u2013ECCV 2022. Lecture Notes in Computer Science, Vol. 13682. Avidan, S.; Brostow, G.; Ciss\u00e9, S.; Farinella, G. M.; Hassner, T. Eds. Springer Cham, 358\u2013374, 2022."},{"key":"375_CR23","unstructured":"Pinkney, J. N. M.; Li, C. clip2latent: Text driven sampling of a pre-trained StyleGAN using denoising diffusion and CLIP. arXiv preprint arXiv:2210.02347, 2022."},{"key":"375_CR24","unstructured":"Dinh, L.; Sohl-Dickstein, J.; Bengio, S. Density estimation using real nvp. arXiv preprint arXiv:1605.08803, 2016."},{"key":"375_CR25","volume-title":"Proceedings of the 32nd Conference on Neural Information Processing Systems","author":"D P Kingma","year":"2018","unstructured":"Kingma, D. P.; Dhariwal, P. Glow: Generative flow with invertible 1 \u00d7 1 convolutions. In: Proceedings of the 32nd Conference on Neural Information Processing Systems, 2018."},{"key":"375_CR26","first-page":"2256","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"W Xia","year":"2021","unstructured":"Xia, W.; Yang, Y.; Xue, J. H.; Wu, B. TediGAN: Text-guided diverse face image generation and manipulation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2256\u20132265, 2021."},{"key":"375_CR27","unstructured":"Wah, C.; Branson, S.; Welinder, P.; Perona, P.; Belongie, S. J. The caltech-UCSD birds-200-2011 dataset. 2011."},{"key":"375_CR28","first-page":"3730","volume-title":"Proceedings of the IEEE International Conference on Computer Vision","author":"Z Liu","year":"2015","unstructured":"Liu, Z.; Luo, P.; Wang, X.; Tang, X. Deep learning face attributes in the wild. In: Proceedings of the IEEE International Conference on Computer Vision, 3730\u20133738, 2015."},{"key":"375_CR29","first-page":"27517","volume-title":"Proceedings of the 35th Conference on Neural Information Processing Systems","author":"A Casanova","year":"2021","unstructured":"Casanova, A.; Careil, M.; Verbeek, J.; Drozdzal, M.; Romero-Soriano, A. Instance-conditioned GAN. In: Proceedings of the 35th Conference on Neural Information Processing Systems, 27517\u201327529, 2021."},{"key":"375_CR30","first-page":"8088","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"A Blattmann","year":"2022","unstructured":"Blattmann, A.; Rombach, R.; Oktay, K.; Muller, J.; Ommer, B. Semi-parametric neural image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 8088\u20138816, 2022."}],"container-title":["Computational Visual Media"],"original-title":[],"link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41095-023-0375-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s41095-023-0375-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41095-023-0375-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10750449\/10884987\/10884996.pdf?arnumber=10884996","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T18:38:14Z","timestamp":1762367894000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10884996\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":30,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1007\/s41095-023-0375-z","relation":{},"ISSN":["2096-0662","2096-0433"],"issn-type":[{"value":"2096-0662","type":"electronic"},{"value":"2096-0433","type":"print"}],"subject":[],"published":{"date-parts":[[2024,12]]},"assertion":[{"value":"16 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 August 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 August 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration of competing interest"}}]}}