{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T14:56:32Z","timestamp":1782312992009,"version":"3.54.5"},"reference-count":190,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T00:00:00Z","timestamp":1729209600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T00:00:00Z","timestamp":1729209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62377040, 62207007"],"award-info":[{"award-number":["62377040, 62207007"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62377040, 62207007"],"award-info":[{"award-number":["62377040, 62207007"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022YFB3303302"],"award-info":[{"award-number":["2022YFB3303302"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-024-10987-w","type":"journal-article","created":{"date-parts":[[2024,10,18]],"date-time":"2024-10-18T08:02:38Z","timestamp":1729238558000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Controllable image synthesis methods, applications and challenges: a comprehensive survey"],"prefix":"10.1007","volume":"57","author":[{"given":"Shanshan","family":"Huang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingsong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Liao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lian","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,18]]},"reference":[{"issue":"3","key":"10987_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3447648","volume":"40","author":"R Abdal","year":"2021","unstructured":"Abdal R, Zhu P, Mitra NJ, Wonka P (2021) Styleflow: attribute-conditioned exploration of stylegan-generated images using conditional continuous normalizing flows. ACM Trans Graph (ToG) 40(3):1\u201321","journal-title":"ACM Trans Graph (ToG)"},{"key":"10987_CR2","doi-asserted-by":"crossref","unstructured":"Abdal R, Qin Y, Wonka P (2019) Image2stylegan: How to embed images into the stylegan latent space? In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4432\u20134441","DOI":"10.1109\/ICCV.2019.00453"},{"key":"10987_CR3","doi-asserted-by":"crossref","unstructured":"Abdal R, Qin Y, Wonka P (2020) Image2stylegan++: How to edit the embedded images? In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8296\u20138305","DOI":"10.1109\/CVPR42600.2020.00832"},{"issue":"8","key":"10987_CR4","doi-asserted-by":"publisher","first-page":"5847","DOI":"10.1007\/s10462-020-09835-4","volume":"53","author":"M Abdolahnejad","year":"2020","unstructured":"Abdolahnejad M, Liu PX (2020) Deep learning for face image synthesis and semantic manipulations: a review and future perspectives. Artif Intell Rev 53(8):5847\u20135880","journal-title":"Artif Intell Rev"},{"issue":"4","key":"10987_CR5","first-page":"1345","volume":"10","author":"J Agnese","year":"2020","unstructured":"Agnese J, Herrera J, Tao H, Zhu X (2020) A survey and taxonomy of adversarial neural networks for text-to-image synthesis. Wiley Interdiscipl Rev: Data Min Knowl Discov 10(4):1345","journal-title":"Wiley Interdiscipl Rev: Data Min Knowl Discov"},{"key":"10987_CR6","doi-asserted-by":"crossref","unstructured":"Alaluf Y, Patashnik O, Cohen-Or D (2021) Restyle: A residual-based stylegan encoder via iterative refinement. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 6711\u20136720","DOI":"10.1109\/ICCV48922.2021.00664"},{"key":"10987_CR7","doi-asserted-by":"crossref","unstructured":"Alghamdi MM, Wang H, Bulpitt AJ, Hogg DC (2022) Talking head from speech audio using a pre-trained image generator. In: Proceedings of the 30th ACM international conference on multimedia. MM \u201922. Association for Computing Machinery, New York, pp 5228\u20135236","DOI":"10.1145\/3503161.3548101"},{"key":"10987_CR8","doi-asserted-by":"crossref","unstructured":"Avrahami O, Lischinski D, Fried O (2022) Blended diffusion for text-driven editing of natural images. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 18208\u201318218","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"10987_CR9","doi-asserted-by":"crossref","unstructured":"Bai J, Dong Z, Feng A, Zhang X, Ye T, Zhou K, Shou MZ (2023) Integrating view conditions for image synthesis. arXiv preprint arXiv:2310.16002","DOI":"10.24963\/ijcai.2024\/840"},{"key":"10987_CR10","unstructured":"Bai J, Liu C, Ni F, Wang H, Hu M, Guo X, Cheng L (2022) Lat: latent translation with cycle-consistency for video-text retrieval. arXiv preprint arXiv:2207.04858"},{"key":"10987_CR11","doi-asserted-by":"crossref","unstructured":"Bao J, Chen D, Wen F, Li H, Hua G (2017) Cvae-gan: fine-grained image generation through asymmetric training. In: Proceedings of the IEEE international conference on computer vision, pp 2745\u20132754","DOI":"10.1109\/ICCV.2017.299"},{"key":"10987_CR12","unstructured":"Batzolis G, Stanczuk J, Sch\u00f6nlieb C-B, Etmann C (2021) Conditional image generation with score-based diffusion models. arXiv preprint arXiv:2111.13606"},{"issue":"4","key":"10987_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3306346.3323023","volume":"38","author":"D Bau","year":"2019","unstructured":"Bau D, Strobelt H, Peebles W, Wulff J, Zhou B, Zhu J-Y, Torralba A (2019) Semantic photo manipulation with a generative image prior. ACM Trans Graph (TOG) 38(4):1\u201311","journal-title":"ACM Trans Graph (TOG)"},{"key":"10987_CR14","doi-asserted-by":"crossref","unstructured":"Bau D, Liu S, Wang T, Zhu J-Y, Torralba A (2020) Rewriting a deep generative model. In: European conference on computer vision. Springer, pp 351\u2013369","DOI":"10.1007\/978-3-030-58452-8_21"},{"key":"10987_CR15","unstructured":"Bau D, Zhu J-Y, Strobelt H, Zhou B, Tenenbaum JB, Freeman WT, Torralba A (2019) Gan dissection: Visualizing and understanding generative adversarial networks. In: Proceedings of the international conference on learning representations (ICLR)"},{"key":"10987_CR16","doi-asserted-by":"crossref","unstructured":"Bhunia AK, Khan S, Cholakkal H, Anwer RM, Laaksonen J, Shah M, Khan FS (2023) Person image synthesis via denoising diffusion model. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5968\u20135976","DOI":"10.1109\/CVPR52729.2023.00578"},{"key":"10987_CR17","unstructured":"Brock A, Donahue J, Simonyan K (2018) Large scale gan training for high fidelity natural image synthesis. In: International conference on learning representations"},{"key":"10987_CR18","doi-asserted-by":"crossref","unstructured":"Chen S-Y, Liu F-L, Lai Y-K, Rosin PL, Li C, Fu H, Gao L (2021) Deepfaceediting: deep face generation and editing with disentangled geometry and appearance control. arXiv preprint arXiv:2105.08935","DOI":"10.1145\/3476576.3476648"},{"key":"10987_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.dsp.2020.102866","volume":"107","author":"Q Cheng","year":"2020","unstructured":"Cheng Q, Gu X (2020) Cross-modal feature alignment based hybrid attentional generative adversarial networks for text-to-image synthesis. Dig Signal Process 107:102866","journal-title":"Dig Signal Process"},{"key":"10987_CR20","unstructured":"Cheng J, Liang X, Shi X, He T, Xiao T, Li M (2023) LayoutDiffuse: adapting foundational diffusion models for layout-to-image generation. arXiv preprint arXiv:2302.08908"},{"key":"10987_CR21","doi-asserted-by":"crossref","unstructured":"Chen S, Ye T, Bai J, Chen E, Shi J, Zhu L (2023) Sparse sampling transformer with uncertainty-driven ranking for unified removal of raindrops and rain streaks. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 13106\u201313117","DOI":"10.1109\/ICCV51070.2023.01205"},{"key":"10987_CR22","doi-asserted-by":"crossref","unstructured":"Cherepkov A, Voynov A, Babenko A (2021) Navigating the gan parameter space for semantic image editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3671\u20133680","DOI":"10.1109\/CVPR46437.2021.00367"},{"key":"10987_CR23","doi-asserted-by":"crossref","unstructured":"Choi J, Kim S, Jeong Y, Gwon Y, Yoon S (2021) Ilvr: Conditioning method for denoising diffusion probabilistic models. In: 2021 IEEE. In: CVF international conference on computer vision (ICCV), pp 14347\u201314356","DOI":"10.1109\/ICCV48922.2021.01410"},{"key":"10987_CR24","unstructured":"Chung H, Kim J-K (2023) C-supcongan: using contrastive learning and trained data features for audio-to-image generation. AICCC \u201922. Association for Computing Machinery, New York"},{"key":"10987_CR25","doi-asserted-by":"crossref","unstructured":"Collins E, Bala R, Price B, Susstrunk S (2020) Editing in style: uncovering the local semantics of gans. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5771\u20135780","DOI":"10.1109\/CVPR42600.2020.00581"},{"key":"10987_CR26","doi-asserted-by":"crossref","unstructured":"Deng Y, Yang J, Chen D, Wen F, Tong X (2020) Disentangled and controllable face image generation via 3d imitative-contrastive learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5154\u20135163","DOI":"10.1109\/CVPR42600.2020.00520"},{"key":"10987_CR27","doi-asserted-by":"crossref","unstructured":"Dhamo H, Farshad A, Laina I, Navab N, Hager GD, Tombari F, Rupprecht C (2020) Semantic image manipulation using scene graphs. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5213\u20135222","DOI":"10.1109\/CVPR42600.2020.00526"},{"key":"10987_CR28","first-page":"8780","volume":"33","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal P, Nichol A (2021) Diffusion models beat gans on image synthesis. Adv Neural Inf Process Syst 33:8780\u20138794","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR29","doi-asserted-by":"crossref","unstructured":"Ding Z, Xu Y, Xu W, Parmar G, Yang Y, Welling M, Tu Z (2020) Guided variational autoencoder for disentanglement learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7920\u20137929","DOI":"10.1109\/CVPR42600.2020.00794"},{"key":"10987_CR30","first-page":"19822","volume":"34","author":"M Ding","year":"2021","unstructured":"Ding M, Yang Z, Hong W, Zheng W, Zhou C, Yin D, Lin J, Zou X, Shao Z, Yang H et al (2021) Cogview: mastering text-to-image generation via transformers. Adv Neural Inf Process Syst 34:19822\u201319835","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR31","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107573","volume":"110","author":"Y Dong","year":"2021","unstructured":"Dong Y, Zhang Y, Ma L, Wang Z, Luo J (2021) Unsupervised text-to-image synthesis. Pattern Recogn 110:107573","journal-title":"Pattern Recogn"},{"key":"10987_CR32","doi-asserted-by":"crossref","unstructured":"Dorta G, Vicente S, Campbell ND, Simpson IJ (2020) The gan that warped: Semantic attribute editing with unpaired data. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5356\u20135365","DOI":"10.1109\/CVPR42600.2020.00540"},{"key":"10987_CR33","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, et al (2020) An image is worth $$16\\times 16$$ words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929"},{"key":"10987_CR34","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., Ommer, B.: Taming transformers for high-resolution image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12873\u201312883 (2021)","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"10987_CR35","doi-asserted-by":"crossref","unstructured":"Esser P, Sutter E, Ommer B (2018) A variational u-net for conditional appearance and shape generation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 8857\u20138866","DOI":"10.1109\/CVPR.2018.00923"},{"key":"10987_CR36","doi-asserted-by":"crossref","unstructured":"Fan W-C, Chen Y-C, Chen D, Cheng Y, Yuan L, Wang Y-CF (2022) Frido: feature pyramid diffusion for complex scene image synthesis. arXiv preprint arXiv:2208.13753","DOI":"10.1609\/aaai.v37i1.25133"},{"key":"10987_CR37","unstructured":"Fan D, Hou Y, Gao C (2023) Cf-vae: causal disentangled representation learning with vae and causal flows. arXiv preprint arXiv:2304.09010"},{"key":"10987_CR38","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1016\/j.neunet.2021.07.019","volume":"144","author":"S Frolov","year":"2021","unstructured":"Frolov S, Hinz T, Raue F, Hees J, Dengel A (2021) Adversarial text-to-image synthesis: a review. Neural Netw 144:187\u2013209","journal-title":"Neural Netw"},{"key":"10987_CR39","doi-asserted-by":"publisher","first-page":"2218","DOI":"10.1109\/TIFS.2021.3050065","volume":"16","author":"C Fu","year":"2021","unstructured":"Fu C, Hu Y, Wu X, Wang G, Zhang Q, He R (2021) High-fidelity face manipulation with extreme poses and expressions. IEEE Trans Inf Forens Secur 16:2218\u20132231","journal-title":"IEEE Trans Inf Forens Secur"},{"key":"10987_CR40","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107384","volume":"110","author":"L Gao","year":"2021","unstructured":"Gao L, Chen D, Zhao Z, Shao J, Shen HT (2021) Lightweight dynamic conditional gan with pyramid attention for text-to-image synthesis. Pattern Recogn 110:107384","journal-title":"Pattern Recogn"},{"key":"10987_CR42","doi-asserted-by":"crossref","unstructured":"Gao C, Liu Q, Xu Q, Wang L, Liu J, Zou C (2020) Sketchycoco: image generation from freehand scene sketches. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5174\u20135183","DOI":"10.1109\/CVPR42600.2020.00522"},{"key":"10987_CR43","unstructured":"Ge Y, Abu-El-Haija S, Xin G, Itti L (2020) Zero-shot synthesis with group-supervised learning. arXiv preprint arXiv:2009.06586"},{"key":"10987_CR44","doi-asserted-by":"crossref","unstructured":"Goetschalckx L, Andonian A, Oliva A, Isola P (2019) Ganalyze: toward visual definitions of cognitive image properties. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5744\u20135753","DOI":"10.1109\/ICCV.2019.00584"},{"key":"10987_CR45","first-page":"1","volume":"27","author":"I Goodfellow","year":"2014","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2014) Generative adversarial nets. Adv Neural Inf Process Syst 27:1\u20139","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR46","doi-asserted-by":"crossref","unstructured":"Gu S, Bao J, Yang H, Chen D, Wen F, Yuan L (2019) Mask-guided portrait editing with conditional gans. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3436\u20133445","DOI":"10.1109\/CVPR.2019.00355"},{"key":"10987_CR47","doi-asserted-by":"crossref","unstructured":"Gu S, Chen D, Bao J, Wen F, Zhang B, Chen D, Yuan L, Guo B (2022) Vector quantized diffusion model for text-to-image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10696\u201310706","DOI":"10.1109\/CVPR52688.2022.01043"},{"key":"10987_CR48","first-page":"9841","volume":"33","author":"E H\u00e4rk\u00f6nen","year":"2020","unstructured":"H\u00e4rk\u00f6nen E, Hertzmann A, Lehtinen J, Paris S (2020) Ganspace: discovering interpretable gan controls. Adv Neural Inf Process Syst 33:9841\u20139850","journal-title":"Adv Neural Inf Process Syst"},{"issue":"11","key":"10987_CR49","doi-asserted-by":"publisher","first-page":"5464","DOI":"10.1109\/TIP.2019.2916751","volume":"28","author":"Z He","year":"2019","unstructured":"He Z, Zuo W, Kan M, Shan S, Chen X (2019) Attgan: facial attribute editing by only changing what you want. IEEE Trans Image Process 28(11):5464\u20135478","journal-title":"IEEE Trans Image Process"},{"key":"10987_CR50","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho J, Jain A, Abbeel P (2020) Denoising diffusion probabilistic models. Adv Neural Inf Process Syst 33:6840\u20136851","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR51","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1016\/j.neunet.2021.10.017","volume":"145","author":"X Hou","year":"2022","unstructured":"Hou X, Zhang X, Liang H, Shen L, Lai Z, Wan J (2022) Guidedstyle: attribute knowledge guided style manipulation for semantic face editing. Neural Netw 145:209\u2013220","journal-title":"Neural Netw"},{"key":"10987_CR52","doi-asserted-by":"crossref","unstructured":"Hsiao W-L, Katsman I, Wu C-Y, Parikh D, Grauman K (2019) Fashion++: Minimal edits for outfit improvement. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5047\u20135056","DOI":"10.1109\/ICCV.2019.00515"},{"key":"10987_CR53","unstructured":"Hu EJ, et al (2021) Lora: low-rank adaptation of large language models. In: International conference on learning representations (ICLR)"},{"issue":"17","key":"10987_CR54","doi-asserted-by":"publisher","first-page":"26465","DOI":"10.1007\/s11042-021-10881-5","volume":"80","author":"S Huang","year":"2021","unstructured":"Huang S, Jin X, Jiang Q, Li J, Lee S-J, Wang P, Yao S (2021) A fully-automatic image colorization scheme using improved cyclegan with skip connections. Multimedia Tools Appl 80(17):26465\u201326492","journal-title":"Multimedia Tools Appl"},{"key":"10987_CR55","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.105006","volume":"114","author":"S Huang","year":"2022","unstructured":"Huang S, Jin X, Jiang Q, Liu L (2022) Deep learning for image colorization: current and future prospects. Eng Appl Artif Intell 114:105006","journal-title":"Eng Appl Artif Intell"},{"key":"10987_CR56","doi-asserted-by":"publisher","first-page":"272","DOI":"10.1016\/j.neunet.2022.11.016","volume":"158","author":"W Huang","year":"2023","unstructured":"Huang W, Tu S, Xu L (2023) Ia-faces: a bidirectional method for semantic face editing. Neural Netw 158:272\u2013292","journal-title":"Neural Netw"},{"issue":"1","key":"10987_CR57","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1631\/FITEE.2300303","volume":"25","author":"S Huang","year":"2024","unstructured":"Huang S, Wang Y, Gong Z, Liao J, Wang S, Liu L (2024) Controllable image generation based on causal representation learning. Front Inf Technol Electron Eng 25(1):135\u2013148","journal-title":"Front Inf Technol Electron Eng"},{"key":"10987_CR58","doi-asserted-by":"crossref","unstructured":"Huang Z, Chan KC, Jiang Y, Liu Z (2023) Collaborative diffusion for multi-modal face generation and editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6080\u20136090","DOI":"10.1109\/CVPR52729.2023.00589"},{"key":"10987_CR59","unstructured":"Huang L, Chen D, Liu Y, Shen Y, Zhao D, Zhou J (2023) Composer: creative and controllable image synthesis with composable conditions. In: Proceedings of the 40th international conference on machine learning, pp 13753\u201313773"},{"key":"10987_CR61","unstructured":"Jahanian A, Chai L, Isola P (2019) On the \u201csteerability\u201d of generative adversarial networks. arXiv preprint arXiv:1907.07171"},{"key":"10987_CR62","unstructured":"Jahn M, Rombach R, Ommer B (2021) High-resolution complex scene synthesis with transformers. arXiv preprint arXiv:2105.06458"},{"key":"10987_CR63","first-page":"14745","volume":"34","author":"Y Jiang","year":"2021","unstructured":"Jiang Y, Chang S, Wang Z (2021) Transgan: two pure transformers can make one strong gan, and that can scale up. Adv Neural Inf Process Syst 34:14745\u201314758","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR64","unstructured":"Jiang J, Ye T, Bai J, Chen S, Chai W, Jun S, Liu Y, Chen E (2023) Five a $$^{+}$$ network: you only need 9k parameters for underwater image enhancement. In: British machine vision conference. pp 1\u201316"},{"key":"10987_CR65","doi-asserted-by":"publisher","first-page":"7066","DOI":"10.1109\/JSTARS.2021.3090958","volume":"14","author":"X Jin","year":"2021","unstructured":"Jin X, Huang S, Jiang Q, Lee S-J, Wu L, Yao S (2021) Semisupervised remote sensing image fusion using multiscale conditional generative adversarial network with siamese structure. IEEE J Sel Top Appl Earth Observ Remote Sens 14:7066\u20137084","journal-title":"IEEE J Sel Top Appl Earth Observ Remote Sens"},{"key":"10987_CR66","doi-asserted-by":"crossref","unstructured":"Johnson J, Gupta A, Fei-Fei L (2018) Image generation from scene graphs. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1219\u20131228","DOI":"10.1109\/CVPR.2018.00133"},{"key":"10987_CR67","doi-asserted-by":"crossref","unstructured":"Kang M, Zhu J-Y, Zhang R, Park J, Shechtman E, Paris S, Park T (2023) Scaling up gans for text-to-image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10124\u201310134","DOI":"10.1109\/CVPR52729.2023.00976"},{"key":"10987_CR68","unstructured":"Karras T, Aila T, Laine S, Lehtinen J (2018) Progressive growing of gans for improved quality, stability, and variation. In: International conference on learning representations"},{"key":"10987_CR69","doi-asserted-by":"crossref","unstructured":"Karras T, Laine S, Aila T (2019) A style-based generator architecture for generative adversarial networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4401\u20134410","DOI":"10.1109\/CVPR.2019.00453"},{"key":"10987_CR70","doi-asserted-by":"crossref","unstructured":"Karras T, Laine S, Aittala M, Hellsten J, Lehtinen J, Aila T (2020) Analyzing and improving the image quality of stylegan. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8110\u20138119","DOI":"10.1109\/CVPR42600.2020.00813"},{"key":"10987_CR71","doi-asserted-by":"crossref","unstructured":"Kawar B, Zada S, Lang O, Tov O, Chang H, Dekel T, Mosseri I, Irani M (2023) Imagic: text-based real image editing with diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6007\u20136017","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"10987_CR72","doi-asserted-by":"crossref","unstructured":"Kim H, Choi Y, Kim J, Yoo S, Uh Y (2021) Exploiting spatial dimensions of latent in gan for real-time image editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 852\u2013861","DOI":"10.1109\/CVPR46437.2021.00091"},{"key":"10987_CR73","first-page":"10236","volume":"31","author":"DP Kingma","year":"2018","unstructured":"Kingma DP, Dhariwal P (2018) Glow: generative flow with invertible $$1\\times 1$$ convolutions. Adv Neural Inf Process Syst 31:10236\u201310245","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR74","unstructured":"Kingma DP, Welling M (2013) Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114"},{"key":"10987_CR75","unstructured":"Kocaoglu M, Snyder C, Dimakis AG, Vishwanath S (2018) Causalgan: learning causal implicit generative models with adversarial training. In: International conference on learning representations"},{"key":"10987_CR76","doi-asserted-by":"crossref","unstructured":"Koley S, Bhunia AK, Sain A, Chowdhury PN, Xiang T, Song Y-Z (2023) Picture that sketch: Photorealistic image generation from abstract sketches. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6850\u20136861","DOI":"10.1109\/CVPR52729.2023.00662"},{"key":"10987_CR77","doi-asserted-by":"crossref","unstructured":"Komanduri A, Wu Y, Chen F, Wu X (2024) Learning causally disentangled representations via the principle of independent causal mechanisms. In: Proceedings of the 33rd international joint conference on artificial intelligence","DOI":"10.24963\/ijcai.2024\/476"},{"key":"10987_CR78","doi-asserted-by":"crossref","unstructured":"Lee C-H, Liu Z, Wu L, Luo P (2020) Maskgan: towards diverse and interactive facial image manipulation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5549\u20135558","DOI":"10.1109\/CVPR42600.2020.00559"},{"key":"10987_CR79","doi-asserted-by":"crossref","unstructured":"Lee T, Kang J, Kim H, Kim T (2023) Generating realistic images from in-the-wild sounds. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 7160\u20137170","DOI":"10.1109\/ICCV51070.2023.00658"},{"key":"10987_CR80","doi-asserted-by":"crossref","unstructured":"Li W (2021) Image synthesis and editing with generative adversarial networks (gans): a review. In: 2021 5th world conference on smart trends in systems security and sustainability (WorldS4). IEEE, pp 65\u201370","DOI":"10.1109\/WorldS451998.2021.9514052"},{"key":"10987_CR81","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109072","volume":"134","author":"S Li","year":"2023","unstructured":"Li S, Liu L, Liu J, Song W, Hao A, Qin H (2023) Sc-gan: Subspace clustering based gan for automatic expression manipulation. Pattern Recogn 134:109072","journal-title":"Pattern Recogn"},{"key":"10987_CR82","doi-asserted-by":"crossref","unstructured":"Liang J, Pei W, Lu F (2023) Layout-bridging text-to-image synthesis. IEEE Trans Circuits Syst Video Technol 7438\u20137451","DOI":"10.1109\/TCSVT.2023.3274228"},{"key":"10987_CR83","doi-asserted-by":"crossref","unstructured":"Liao Y, Schwarz K, Mescheder L, Geiger A (2020) Towards unsupervised learning of generative models for 3d controllable image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5871\u20135880","DOI":"10.1109\/CVPR42600.2020.00591"},{"key":"10987_CR84","doi-asserted-by":"crossref","unstructured":"Li B, Deng S-H, Liu B, Li Y, He Z-F, Lai Y-K, Zhang C, Chen Z (2023) Controllable facial attribute editing via gaussian mixture model disentanglement. Dig Signal Process 103916","DOI":"10.1016\/j.dsp.2023.103916"},{"key":"10987_CR85","doi-asserted-by":"crossref","unstructured":"Li G, Liu Y, Wei X, Zhang Y, Wu S, Xu Y, Wong H-S (2021) Discovering density-preserving latent space walks in gans for semantic image transformations. In: Proceedings of the 29th ACM international conference on multimedia, pp 1562\u20131570","DOI":"10.1145\/3474085.3475293"},{"key":"10987_CR86","doi-asserted-by":"crossref","unstructured":"Li Y, Liu H, Wu Q, Mu F, Yang J, Gao J, Li C, Lee YJ (2023) Gligen: open-set grounded text-to-image generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 22511\u201322521","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"10987_CR87","first-page":"16331","volume":"34","author":"H Ling","year":"2021","unstructured":"Ling H, Kreis K, Li D, Kim SW, Torralba A, Fidler S (2021) Editgan: high-precision semantic image editing. Adv Neural Inf Process Syst 34:16331\u201316345","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR88","doi-asserted-by":"crossref","unstructured":"Lin J, Zhang R, Ganz F, Han S, Zhu J-Y (2021) Anycost gans for interactive image synthesis and editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 14986\u201314996","DOI":"10.1109\/CVPR46437.2021.01474"},{"key":"10987_CR89","doi-asserted-by":"crossref","unstructured":"Li X, Sun S, Feng R (2024) Causal representation learning via counterfactual intervention. In: Proceedings of the AAAI conference on artificial intelligence, vol 38, pp 3234\u20133242","DOI":"10.1609\/aaai.v38i4.28108"},{"key":"10987_CR90","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1016\/j.cag.2019.03.009","volume":"81","author":"C Liu","year":"2019","unstructured":"Liu C, Yang Z, Xu F, Yong J-H (2019) Image generation from bounding box-represented semantic labels. Comput Graph 81:32\u201340","journal-title":"Comput Graph"},{"issue":"6","key":"10987_CR91","doi-asserted-by":"publisher","first-page":"2733","DOI":"10.1109\/TNNLS.2020.3007790","volume":"32","author":"Y Liu","year":"2020","unstructured":"Liu Y, Sun Q, He X, Liu A-A, Su Y, Chua T-S (2020) Generating face images with attributes for free. IEEE Trans Neural Netw Learn Syst 32(6):2733\u20132743","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"10987_CR92","doi-asserted-by":"crossref","unstructured":"Liu R, Ge Y, Choi CL, Wang X, Li H (2021) Divco: diverse conditional image synthesis via contrastive generative adversarial network. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 16377\u201316386","DOI":"10.1109\/CVPR46437.2021.01611"},{"key":"10987_CR93","doi-asserted-by":"crossref","unstructured":"Liu R, Liu Y, Gong X, Wang X, Li H (2019) Conditional adversarial generative flow for controllable image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7992\u20138001","DOI":"10.1109\/CVPR.2019.00818"},{"key":"10987_CR94","doi-asserted-by":"crossref","unstructured":"Liu X, Park DH, Azadi S, Zhang G, Chopikyan A, Hu Y, Shi H, Rohrbach A, Darrell T (2023) More control for free! image synthesis with semantic diffusion guidance. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 289\u2013299","DOI":"10.1109\/WACV56688.2023.00037"},{"key":"10987_CR95","doi-asserted-by":"crossref","unstructured":"Liu B, Song K, Zhu Y, Melo G, Elgammal A (2021) Time: Text and image mutual-translation adversarial networks. In: Proceedings of the AAAI conference on artificial intelligence, vol 35, pp 2082\u20132090","DOI":"10.1609\/aaai.v35i3.16305"},{"key":"10987_CR96","unstructured":"Lu Y-D, Lee H-Y, Tseng H-Y, Yang M-H (2020) Unsupervised discovery of disentangled manifolds in gans. arXiv preprint arXiv:2011.11842"},{"key":"10987_CR97","doi-asserted-by":"crossref","unstructured":"Lugmayr A, Danelljan M, Romero A, Yu F, Timofte R, Van Gool L (2022) Repaint: inpainting using denoising diffusion probabilistic models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 11461\u201311471","DOI":"10.1109\/CVPR52688.2022.01117"},{"key":"10987_CR98","doi-asserted-by":"crossref","unstructured":"Mao Q, Lee H-Y, Tseng H-Y, Ma S, YangM-H (2019) Mode seeking generative adversarial networks for diverse image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1429\u20131437","DOI":"10.1109\/CVPR.2019.00152"},{"key":"10987_CR99","unstructured":"Meng C, He Y, Song Y, Song J, Wu J, Zhu J-Y, Ermon S (2022) Sdedit: guided image synthesis and editing with stochastic differential equations. In: International conference on learning representations"},{"key":"10987_CR100","doi-asserted-by":"crossref","unstructured":"Men Y, Mao Y, Jiang Y, Ma W-Y, Lian Z (2020) Controllable person image synthesis with attribute-decomposed gan. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5084\u20135093","DOI":"10.1109\/CVPR42600.2020.00513"},{"key":"10987_CR101","unstructured":"Mirza M, Osindero S (2014) Conditional generative adversarial nets. arXiv preprint arXiv:1411.1784"},{"key":"10987_CR102","unstructured":"Moraffah R, Moraffah B, Karami M, Raglin A, Liu H (2020) Can: a causal adversarial network for learning observational and interventional distributions. arXiv preprint arXiv:2008.11376"},{"key":"10987_CR103","doi-asserted-by":"crossref","unstructured":"Mou C, Wang X, Xie L, Wu Y, Zhang J, Qi Z, Shan Y (2024) T2i-adapter: learning adapters to dig out more controllable ability for text-to-image diffusion models. In: Proceedings of the AAAI conference on artificial intelligence, vol 38, pp 4296\u20134304","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"10987_CR104","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2021.104284","volume":"115","author":"S Naveen","year":"2021","unstructured":"Naveen S, Kiran MSR, Indupriya M, Manikanta T, Sudeep P (2021) Transformer models for enhancing attngan based text to image generation. Image Vis Comput 115:104284","journal-title":"Image Vis Comput"},{"key":"10987_CR105","unstructured":"Nichol AQ, Dhariwal P (2021) Improved denoising diffusion probabilistic models. In: International conference on machine learning. PMLR, pp 8162\u20138171"},{"key":"10987_CR106","unstructured":"Nichol AQ, Dhariwal P, Ramesh A, Shyam P, Mishkin P, Mcgrew B, Sutskever I, Chen M (2022) Glide: towards photorealistic image generation and editing with text-guided diffusion models. In: International conference on machine learning. PMLR, pp 16784\u201316804"},{"key":"10987_CR107","unstructured":"Odena A, Olah C, Shlens J (2017) Conditional image synthesis with auxiliary classifier gans. In: International conference on machine learning. PMLR, pp 2642\u20132651"},{"key":"10987_CR108","doi-asserted-by":"crossref","unstructured":"Pajouheshgar E, Zhang T, S\u00fcsstrunk S (2022) Optimizing latent space directions for gan-based local image editing. In: ICASSP 2022-2022 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 1740\u20131744","DOI":"10.1109\/ICASSP43922.2022.9747326"},{"key":"10987_CR109","doi-asserted-by":"crossref","unstructured":"Pang Y, Zhang Y, Quan W, Fan Y, Cun X, Shan Y, Yan D-m (2023) Dpe: disentanglement of pose and expression for general video portrait editing. arXiv preprint arXiv:2301.06281","DOI":"10.1109\/CVPR52729.2023.00049"},{"key":"10987_CR110","doi-asserted-by":"crossref","unstructured":"Park T, Efros AA, Zhang R, Zhu J-Y (2020) Contrastive learning for unpaired image-to-image translation. In: European conference on computer vision. Springer, pp 319\u2013345","DOI":"10.1007\/978-3-030-58545-7_19"},{"key":"10987_CR111","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2022.110026","volume":"259","author":"C Peng","year":"2023","unstructured":"Peng C, Zhang C, Liu D, Wang N, Gao X (2023) Face photo-sketch synthesis via intra-domain enhancement. Knowl-Based Syst 259:110026","journal-title":"Knowl-Based Syst"},{"key":"10987_CR112","doi-asserted-by":"crossref","unstructured":"Pidhorskyi S, Adjeroh DA, Doretto G (2020) Adversarial latent autoencoders. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 14104\u201314113","DOI":"10.1109\/CVPR42600.2020.01411"},{"key":"10987_CR60","unstructured":"Puzer (2022) Stylegan-encoder. https:\/\/github.com\/Puzer\/stylegan-encoder. Accessed Jan 2022"},{"key":"10987_CR113","doi-asserted-by":"crossref","unstructured":"Qiao T, Shao H, Xie S, Shi R (2024) Unsupervised generative fake image detector. IEEE Trans Circuits Syst Video Technol 8442\u20138455","DOI":"10.1109\/TCSVT.2024.3383833"},{"key":"10987_CR114","doi-asserted-by":"crossref","unstructured":"Qin C, Yu N, Xing C, Zhang S, Chen Z, Ermon S, Fu Y, Xiong C, Xu R (2023) Gluegen: plug and play multi-modal encoders for x-to-image generation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 23085\u201323096","DOI":"10.1109\/ICCV51070.2023.02110"},{"key":"10987_CR115","unstructured":"Radford A, Metz L, Chintala S (2015) Unsupervised representation learning with deep convolutional generative adversarial networks. arXiv preprint arXiv:1511.06434"},{"key":"10987_CR116","unstructured":"Ramesh A, Dhariwal P, Nichol A, Chu C, Chen M (2022) Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125"},{"key":"10987_CR117","unstructured":"Ramesh A, Pavlov M, Goh G, Gray S, Voss C, Radford A, Chen M, Sutskever I (2021) Zero-shot text-to-image generation. In: International conference on machine learning. PMLR, pp 8821\u20138831"},{"key":"10987_CR118","first-page":"14866","volume":"33","author":"A Razavi","year":"2019","unstructured":"Razavi A, Oord A, Vinyals O (2019) Generating diverse high-fidelity images with vq-vae-2. Adv Neural Inf Process Syst 33:14866\u201314876","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR119","doi-asserted-by":"publisher","first-page":"8622","DOI":"10.1109\/TIP.2020.3018224","volume":"29","author":"Y Ren","year":"2020","unstructured":"Ren Y, Li G, Liu S, Li TH (2020) Deep spatial transformation for pose-guided person image generation and animation. IEEE Trans Image Process 29:8622\u20138635","journal-title":"IEEE Trans Image Process"},{"key":"10987_CR120","doi-asserted-by":"crossref","unstructured":"Ren Y, Fan X, Li G, Liu S, Li TH (2022) Neural texture extraction and distribution for controllable person image synthesis. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 13535\u201313544","DOI":"10.1109\/CVPR52688.2022.01317"},{"key":"10987_CR121","doi-asserted-by":"crossref","unstructured":"Richardson E, Alaluf Y, Patashnik O, Nitzan Y, Azar Y, Shapiro S, Cohen-Or D (2021) Encoding in style: a stylegan encoder for image-to-image translation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2287\u20132296","DOI":"10.1109\/CVPR46437.2021.00232"},{"issue":"1","key":"10987_CR122","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3544777","volume":"42","author":"D Roich","year":"2022","unstructured":"Roich D, Mokady R, Bermano AH, Cohen-Or D (2022) Pivotal tuning for latent-based editing of real images. ACM Trans Graph (TOG) 42(1):1\u201313","journal-title":"ACM Trans Graph (TOG)"},{"key":"10987_CR123","doi-asserted-by":"crossref","unstructured":"Rombach R, Blattmann A, Lorenz D, Esser P, Ommer B (2022) High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 10684\u201310695","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"10987_CR124","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia C, Chan W, Saxena S, Li L, Whang J, Denton EL, Ghasemipour K, Gontijo Lopes R, Karagol Ayan B, Salimans T et al (2022) Photorealistic text-to-image diffusion models with deep language understanding. Adv Neural Inf Process Syst 35:36479\u201336494","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR125","unstructured":"Sanchez P, Tsaftaris SA (2022) Diffusion causal models for counterfactual estimation. arXiv preprint arXiv:2202.10166"},{"key":"10987_CR126","doi-asserted-by":"crossref","unstructured":"Sauer A, Schwarz K, Geiger A (2022) Stylegan-xl: Scaling stylegan to large diverse datasets. In: SIGGRAPH, pp 1\u201310","DOI":"10.1145\/3528233.3530738"},{"key":"10987_CR127","doi-asserted-by":"publisher","first-page":"126","DOI":"10.1016\/j.inffus.2021.02.014","volume":"72","author":"P Shamsolmoali","year":"2021","unstructured":"Shamsolmoali P, Zareapoor M, Granger E, Zhou H, Wang R, Celebi ME, Yang J (2021) Image synthesis with adversarial networks: a comprehensive survey and case studies. Inf Fus 72:126\u2013146","journal-title":"Inf Fus"},{"key":"10987_CR128","doi-asserted-by":"crossref","unstructured":"Shang W, Sohn K (2019) Attentive conditional channel-recurrent autoencoding for attribute-conditioned face synthesis. In: 2019 IEEE winter conference on applications of computer vision (WACV). IEEE, pp 1533\u20131542","DOI":"10.1109\/WACV.2019.00168"},{"key":"10987_CR129","first-page":"1","volume":"23","author":"X Shen","year":"2022","unstructured":"Shen X, Liu F, Dong H, Lian Q, Chen Z, Zhang T (2022) Weakly supervised disentangled generative causal representation learning. J Mach Learn Res 23:1\u201355","journal-title":"J Mach Learn Res"},{"key":"10987_CR130","doi-asserted-by":"crossref","unstructured":"Shen Y, Gu J, Tang X, Zhou B (2020) Interpreting the latent space of gans for semantic face editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9243\u20139252","DOI":"10.1109\/CVPR42600.2020.00926"},{"key":"10987_CR131","doi-asserted-by":"crossref","unstructured":"Shen Y, Yang C, Tang X, Zhou B (2020) Interfacegan: Interpreting the disentangled face representation learned by gans. IEEE Trans Pattern Anal Mach Intell 2004\u20132018","DOI":"10.1109\/TPAMI.2020.3034267"},{"key":"10987_CR132","doi-asserted-by":"crossref","unstructured":"Shen Y, Zhou B (2021) Closed-form factorization of latent semantics in gans. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 1532\u20131540","DOI":"10.1109\/CVPR46437.2021.00158"},{"key":"10987_CR133","first-page":"3483","volume":"28","author":"K Sohn","year":"2015","unstructured":"Sohn K, Lee H, Yan X (2015) Learning structured output representation using deep conditional generative models. Adv Neural Inf Process Syst 28:3483\u20133491","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR134","doi-asserted-by":"crossref","unstructured":"Song X, Cui J, Zhang H, Chen J, Hong R, Jiang Y-G (2024) Doubly abductive counterfactual inference for text-based image editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9162\u20139171","DOI":"10.1109\/CVPR52733.2024.00875"},{"key":"10987_CR135","unstructured":"Suzuki R, Koyama M, Miyato T, Yonetsuji T, Zhu H (2018) Spatially controllable image synthesis with internal representation collaging. arXiv preprint arXiv:1811.10153"},{"key":"10987_CR136","doi-asserted-by":"crossref","unstructured":"Tan Z, Chai M, Chen D, Liao J, Chu Q, Yuan L, Tulyakov S, Yu N (2020) Michigan: multi-input-conditioned hair image generation for portrait editing. arXiv preprint arXiv:2010.16417","DOI":"10.1145\/3386569.3392488"},{"key":"10987_CR137","doi-asserted-by":"publisher","first-page":"7903","DOI":"10.1109\/TIP.2021.3109531","volume":"30","author":"H Tang","year":"2021","unstructured":"Tang H, Sebe N (2021) Layout-to-image translation with double pooling generative adversarial networks. IEEE Trans Image Process 30:7903\u20137913","journal-title":"IEEE Trans Image Process"},{"key":"10987_CR138","first-page":"16083","volume":"37","author":"Z Tang","year":"2024","unstructured":"Tang Z, Yang Z, Zhu C, Zeng M, Bansal M (2024) Any-to-any generation via composable diffusion. Adv Neural Inf Process Syst 37:16083\u201316099","journal-title":"Adv Neural Inf Process Syst"},{"issue":"6","key":"10987_CR139","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3414685.3417803","volume":"39","author":"A Tewari","year":"2020","unstructured":"Tewari A, Elgharib M, Bernard F, Seidel H-P, P\u00e9rez P, Zollh\u00f6fer M, Theobalt C (2020) Pie: portrait image embedding for semantic control. ACM Trans Graph (TOG) 39(6):1\u201314","journal-title":"ACM Trans Graph (TOG)"},{"key":"10987_CR140","doi-asserted-by":"crossref","unstructured":"Tewari A, Elgharib M, Bharaj G, Bernard F, Seidel H-P, P\u00e9rez P, Zollhofer M, Theobalt C (2020) Stylerig: Rigging stylegan for 3d control over portrait images. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6142\u20136151","DOI":"10.1109\/CVPR42600.2020.00618"},{"issue":"4","key":"10987_CR141","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3450626.3459838","volume":"40","author":"O Tov","year":"2021","unstructured":"Tov O, Alaluf Y, Nitzan Y, Patashnik O, Cohen-Or D (2021) Designing an encoder for stylegan image manipulation. ACM Trans Graph (TOG) 40(4):1\u201314","journal-title":"ACM Trans Graph (TOG)"},{"key":"10987_CR142","first-page":"2685","volume":"1\u201321:","author":"S Tyagi","year":"2021","unstructured":"Tyagi S, Yadav D (2021) A comprehensive review on image synthesis with adversarial networks: theory, literature, and applications. Arch Comput Methods Eng 1\u201321:2685\u20132705","journal-title":"Arch Comput Methods Eng"},{"key":"10987_CR143","first-page":"6309","volume":"30","author":"A Van Den Oord","year":"2017","unstructured":"Van Den Oord A, Vinyals O et al (2017) Neural discrete representation learning. Adv Neural Inf Process Syst 30:6309\u20136318","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR144","unstructured":"Van Oord A, Kalchbrenner N, Kavukcuoglu K (2016) Pixel recurrent neural networks. In: International conference on machine learning. PMLR, pp 1747\u20131756"},{"key":"10987_CR145","unstructured":"Voynov A, Babenko A (2020) Unsupervised discovery of interpretable directions in the gan latent space. In: International conference on machine learning. PMLR, pp 9786\u20139796"},{"issue":"4","key":"10987_CR146","doi-asserted-by":"publisher","first-page":"69-1","DOI":"10.1145\/3386569.3392456","volume":"39","author":"Y Wang","year":"2020","unstructured":"Wang Y, Gao Y, Lian Z (2020) Attribute2font: creating fonts you want from attributes. ACM Trans Graph (TOG) 39(4):69\u20131","journal-title":"ACM Trans Graph (TOG)"},{"key":"10987_CR147","doi-asserted-by":"crossref","unstructured":"Wang Y, Lin C, Luo D, Tai Y, Zhang Z, Xie Y (2023) High-resolution gan inversion for degraded images in large diverse datasets. In: Proceedings of the AAAI conference on artificial intelligence, vol 37, pp 2716\u20132723","DOI":"10.1609\/aaai.v37i3.25371"},{"key":"10987_CR148","doi-asserted-by":"crossref","unstructured":"Wang P, Li Y, Singh KK, Lu J, Vasconcelos N (2021) Imagine: Image synthesis by image-guided model inversion. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3681\u20133690","DOI":"10.1109\/CVPR46437.2021.00368"},{"key":"10987_CR149","doi-asserted-by":"publisher","first-page":"4375","DOI":"10.1109\/TMM.2023.3322326","volume":"26","author":"J Wang","year":"2024","unstructured":"Wang J, Liu P, Liu J, Xu W (2024) Text-guided eyeglasses manipulation with spatial constraints. IEEE Trans Multimed 26:4375\u20134388","journal-title":"IEEE Transactions on Multimedia"},{"key":"10987_CR150","doi-asserted-by":"crossref","unstructured":"Wang T, Zhang Y, Fan Y, Wang J, Chen Q (2022) High-fidelity gan inversion for image attribute editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 11379\u201311388","DOI":"10.1109\/CVPR52688.2022.01109"},{"key":"10987_CR151","doi-asserted-by":"publisher","first-page":"8658","DOI":"10.1109\/TIP.2021.3112059","volume":"30","author":"X Wu","year":"2021","unstructured":"Wu X, Zhang Q, Wu Y, Wang H, Li S, Sun L, Li X (2021) $$\\text{ F}^3$$a-gan: facial flow for face animation with generative adversarial networks. IEEE Trans Image Process 30:8658\u20138670","journal-title":"IEEE Trans Image Process"},{"key":"10987_CR152","doi-asserted-by":"crossref","unstructured":"Wu Z, Lischinski D, Shechtman E (2021) Stylespace analysis: disentangled controls for stylegan image generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12863\u201312872","DOI":"10.1109\/CVPR46437.2021.01267"},{"key":"10987_CR153","doi-asserted-by":"crossref","unstructured":"Wu R, Zhang G, Lu S, Chen T (2020) Cascade ef-gan: progressive facial expression editing with local focuses. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5021\u20135030","DOI":"10.1109\/CVPR42600.2020.00507"},{"key":"10987_CR154","unstructured":"Xiao Z, Kreis K, Vahdat A (2021) Tackling the generative learning trilemma with denoising diffusion gans. arXiv preprint arXiv:2112.07804"},{"key":"10987_CR155","doi-asserted-by":"crossref","unstructured":"Xia W, Zhang Y, Yang Y, Xue J-H, Zhou B, Yang M-H (2022) Gan inversion: a survey. IEEE Trans Pattern Anal Mach Intell 3121\u20133138","DOI":"10.1109\/TPAMI.2022.3181070"},{"key":"10987_CR156","doi-asserted-by":"crossref","unstructured":"Xie S, Zhang Z, Lin Z, Hinz T, Zhang K (2023) Smartbrush: text and shape guided object inpainting with diffusion model. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 22428\u201322437","DOI":"10.1109\/CVPR52729.2023.02148"},{"key":"10987_CR157","doi-asserted-by":"crossref","unstructured":"Xin Y, et al (2024) Vmt-adapter: parameter-efficient transfer learning for multi-task dense. In: Proceedings of the AAAI conference on artificial intelligence (AAAI), pp 16085\u201316093","DOI":"10.1609\/aaai.v38i14.29541"},{"key":"10987_CR158","unstructured":"Xin Y, Luo S, Zhou H, Du J, Liu X, Fan Y, Li Q, Du Y (2024) Parameter-efficient fine-tuning for pre-trained vision models: a survey. arXiv preprint arXiv:2402.02242"},{"key":"10987_CR159","first-page":"10359","volume":"37","author":"S Xu","year":"2024","unstructured":"Xu S, Ma Z, Huang Y, Lee H, Chai J (2024) Cyclenet: rethinking cycle consistency in text-guided diffusion for image manipulation. Adv Neural Inf Process Syst 37:10359\u201310384","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR160","doi-asserted-by":"crossref","unstructured":"Xu Y, Yin Y, Jiang L, Wu Q, Zheng C, Loy CC, Dai B, Wu W (2022) Transeditor: transformer-based dual-space gan for highly controllable facial editing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7683\u20137692","DOI":"10.1109\/CVPR52688.2022.00753"},{"issue":"5","key":"10987_CR161","doi-asserted-by":"publisher","first-page":"1451","DOI":"10.1007\/s11263-020-01429-5","volume":"129","author":"C Yang","year":"2021","unstructured":"Yang C, Shen Y, Zhou B (2021) Semantic hierarchy emerges in deep generative representations for scene synthesis. Int J Comput Vis 129(5):1451\u20131466","journal-title":"Int J Comput Vis"},{"key":"10987_CR162","doi-asserted-by":"publisher","first-page":"8797","DOI":"10.1109\/TIP.2021.3120669","volume":"30","author":"S Yang","year":"2021","unstructured":"Yang S, Wang Z, Liu J, Guo Z (2021) Controllable sketch-to-image translation for robust face synthesis. IEEE Trans Image Process 30:8797\u20138810","journal-title":"IEEE Trans Image Process"},{"key":"10987_CR163","doi-asserted-by":"publisher","first-page":"698","DOI":"10.1016\/j.ins.2023.03.042","volume":"632","author":"M Yang","year":"2023","unstructured":"Yang M, Wang Z, Chi Z, Du W (2023) Protogan: towards high diversity and fidelity image synthesis under limited data. Inf Sci 632:698\u2013714. https:\/\/doi.org\/10.1016\/j.ins.2023.03.042","journal-title":"Inf Sci"},{"key":"10987_CR164","doi-asserted-by":"crossref","unstructured":"Yang M, Liu F, Chen Z, Shen X, Hao J, Wang J (2021) Causalvae: disentangled representation learning via neural structural causal models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9593\u20139602","DOI":"10.1109\/CVPR46437.2021.00947"},{"key":"10987_CR165","doi-asserted-by":"crossref","unstructured":"Yariv G, Gat I, Wolf L, Adi Y, Schwartz I (2023) Audiotoken: adaptation of text-conditioned diffusion models for audio-to-image generation. arXiv preprint arXiv:2305.13050","DOI":"10.21437\/Interspeech.2023-852"},{"key":"10987_CR166","doi-asserted-by":"crossref","unstructured":"Ye T, Chen S, Bai J, Shi J, Xue C, Jiang J, Yin J, Chen E, Liu Y (2023) Adverse weather removal with codebook priors. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 12653\u201312664","DOI":"10.1109\/ICCV51070.2023.01163"},{"key":"10987_CR167","doi-asserted-by":"crossref","unstructured":"Ye T, Chen S, Chai W, Xing Z, Qin J, Lin G, Zhu L (2024) Learning diffusion texture priors for image restoration. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 2524\u20132534","DOI":"10.1109\/CVPR52733.2024.00244"},{"key":"10987_CR168","doi-asserted-by":"crossref","unstructured":"Yi Z, Zhang H, Tan P, Gong M (2017) Dualgan: unsupervised dual learning for image-to-image translation. In: Proceedings of the IEEE international conference on computer vision, pp 2849\u20132857","DOI":"10.1109\/ICCV.2017.310"},{"key":"10987_CR169","doi-asserted-by":"crossref","unstructured":"Y\u00fcksel OK, Simsar E, Er EG, Yanardag P (2021) Latentclr: a contrastive learning approach for unsupervised discovery of interpretable directions. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 14263\u201314272","DOI":"10.1109\/ICCV48922.2021.01400"},{"key":"10987_CR170","doi-asserted-by":"crossref","unstructured":"Yun J, Lee S, Park M, Choo J (2023) icolorit: towards propagating local hints to the right region in interactive colorization by leveraging vision transformer. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision (WACV), pp 1787\u20131796","DOI":"10.1109\/WACV56688.2023.00183"},{"key":"10987_CR171","first-page":"21125","volume":"34","author":"Y Zeng","year":"2021","unstructured":"Zeng Y, Yang H, Chao H, Wang J, Fu J (2021) Improving visual quality of image synthesis by a token-based generator with transformers. Adv Neural Inf Process Syst 34:21125\u201321137","journal-title":"Adv Neural Inf Process Syst"},{"issue":"8","key":"10987_CR172","doi-asserted-by":"publisher","first-page":"1947","DOI":"10.1109\/TPAMI.2018.2856256","volume":"41","author":"H Zhang","year":"2018","unstructured":"Zhang H, Xu T, Li H, Zhang S, Wang X, Huang X, Metaxas DN (2018) Stackgan++: realistic image synthesis with stacked generative adversarial networks. IEEE Trans Pattern Anal Mach Intell 41(8):1947\u20131962","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"10987_CR173","doi-asserted-by":"crossref","unstructured":"Zhang Z, Han L, Ghosh A, Metaxas DN, Ren J (2023) Sine: single image editing with text-to-image diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp 6027\u20136037","DOI":"10.1109\/CVPR52729.2023.00584"},{"key":"10987_CR174","doi-asserted-by":"crossref","unstructured":"Zhang G, Kan M, Shan S, Chen X (2018) Generative adversarial network with spatial attention for face attribute editing. In: Proceedings of the European conference on computer vision (ECCV), pp 417\u2013432","DOI":"10.1007\/978-3-030-01231-1_26"},{"key":"10987_CR175","doi-asserted-by":"crossref","unstructured":"Zhang W, Liao J, Zhang Y, Liu L (2022) Cmgan: a generative adversarial network embedded with causal matrix. Appl Intell 16233\u201316245","DOI":"10.1007\/s10489-021-03094-8"},{"key":"10987_CR176","doi-asserted-by":"crossref","unstructured":"Zhang J, Li K, Lai Y-K, Yang J (2021) Pise: person image synthesis and editing with decoupled gan. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7982\u20137990","DOI":"10.1109\/CVPR46437.2021.00789"},{"key":"10987_CR177","first-page":"27196","volume":"34","author":"Z Zhang","year":"2021","unstructured":"Zhang Z, Ma J, Zhou C, Men R, Li Z, Ding M, Tang J, Zhou J, Yang H (2021) Ufc-bert: unifying multi-modal controls for conditional image synthesis. Adv Neural Inf Process Syst 34:27196\u201327208","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR178","doi-asserted-by":"crossref","unstructured":"Zhang L, Rao A, Agrawala M (2023) Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3836\u20133847","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"10987_CR179","doi-asserted-by":"crossref","unstructured":"Zhang H, Xu T, Li H, Zhang S, Wang X, Huang X, Metaxas DN (2017) Stackgan: text to photo-realistic image synthesis with stacked generative adversarial networks. In: Proceedings of the IEEE international conference on computer vision, pp 5907\u20135915","DOI":"10.1109\/ICCV.2017.629"},{"key":"10987_CR180","unstructured":"Zhang C, Zhang C, Zheng S, Qiao Y, Li C, Zhang M, Dam SK, Thwal CM, Tun YL, Huy LL, et al (2023) A complete survey on generative ai (aigc): Is chatgpt from gpt-4 to gpt-5 all you need? arXiv preprint arXiv:2303.11717"},{"key":"10987_CR181","first-page":"11127","volume":"36","author":"S Zhao","year":"2024","unstructured":"Zhao S, Chen D, Chen Y-C, Bao J, Hao S, Yuan L, Wong K-YK (2024) Uni-controlnet: all-in-one control to text-to-image diffusion models. Adv Neural Inf Process Syst 36:11127\u201311150","journal-title":"Adv Neural Inf Process Syst"},{"key":"10987_CR182","doi-asserted-by":"crossref","unstructured":"Zhao B, Meng L, Yin W, Sigal L (2019) Image generation from layout. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8584\u20138593","DOI":"10.1109\/CVPR.2019.00878"},{"key":"10987_CR183","doi-asserted-by":"crossref","unstructured":"Zheng Y, Huang Y-K, Tao R, Shen Z, Savvides M (2021) Unsupervised disentanglement of linear-encoded facial semantics. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3917\u20133926","DOI":"10.1109\/CVPR46437.2021.00391"},{"key":"10987_CR184","doi-asserted-by":"crossref","unstructured":"Zheng G, Zhou X, Li X, Qi Z, Shan Y, Li X (2023) Layoutdiffusion: controllable diffusion model for layout-to-image generation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 22490\u201322499","DOI":"10.1109\/CVPR52729.2023.02154"},{"key":"10987_CR185","doi-asserted-by":"crossref","unstructured":"Zhou X, Yin M, Chen X, Sun L, Gao C, Li Q (2022) Cross attention based style distribution for controllable person image synthesis. In: Computer Vision\u2013ECCV 2022: 17th European conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XV. Springer, pp 161\u2013178","DOI":"10.1007\/978-3-031-19784-0_10"},{"key":"10987_CR186","doi-asserted-by":"crossref","unstructured":"Zhu J-Y, Kr\u00e4henb\u00fchl P, Shechtman E, Efros AA (2016) Generative visual manipulation on the natural image manifold. In: European conference on computer vision. Springer, pp 597\u2013613","DOI":"10.1007\/978-3-319-46454-1_36"},{"key":"10987_CR187","doi-asserted-by":"crossref","unstructured":"Zhu J-Y, Park T, Isola P, Efros AA (2017) Unpaired image-to-image translation using cycle-consistent adversarial networks. In: Proceedings of the IEEE international conference on computer vision, pp 2223\u20132232","DOI":"10.1109\/ICCV.2017.244"},{"key":"10987_CR188","doi-asserted-by":"crossref","unstructured":"Zhu P, Abdal R, Qin Y, Wonka P (2020) Sean: image synthesis with semantic region-adaptive normalization. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5104\u20135113","DOI":"10.1109\/CVPR42600.2020.00515"},{"key":"10987_CR189","doi-asserted-by":"crossref","unstructured":"Zhu J, et al (2023) Visual prompt multi-modal tracking. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR). pp 9516\u20139526","DOI":"10.1109\/CVPR52729.2023.00918"},{"key":"10987_CR190","doi-asserted-by":"crossref","unstructured":"Zhu J, Shen Y, Zhao D, Zhou B (2020) In-domain gan inversion for real image editing. In: European conference on computer vision. Springer, pp 592\u2013608","DOI":"10.1007\/978-3-030-58520-4_35"},{"key":"10987_CR191","doi-asserted-by":"crossref","unstructured":"Zhu J, Yang C, Shen Y, Shi Z, Zhao D, Chen Q (2023) Linkgan: linking gan latents to pixels for controllable image synthesis. arXiv preprint arXiv:2301.04604","DOI":"10.1109\/ICCV51070.2023.00704"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-024-10987-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-024-10987-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-024-10987-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,13]],"date-time":"2024-11-13T15:14:28Z","timestamp":1731510868000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-024-10987-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,18]]},"references-count":190,"journal-issue":{"issue":"12","published-online":{"date-parts":[[2024,12]]}},"alternative-id":["10987"],"URL":"https:\/\/doi.org\/10.1007\/s10462-024-10987-w","relation":{},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,18]]},"assertion":[{"value":"26 September 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 October 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors certify that they have no affiliations with or involvement in any organization or entity with any financial interest or non-financial interest in the subject matter or materials discussed in this manuscript.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"336"}}