{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T14:13:55Z","timestamp":1783520035808,"version":"3.55.0"},"reference-count":87,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2024,12,12]],"date-time":"2024-12-12T00:00:00Z","timestamp":1733961600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,12]],"date-time":"2024-12-12T00:00:00Z","timestamp":1733961600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s11263-024-02294-2","type":"journal-article","created":{"date-parts":[[2024,12,12]],"date-time":"2024-12-12T06:16:28Z","timestamp":1733984188000},"page":"2805-2824","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["MaskDiffusion: Boosting Text-to-Image Consistency with Conditional Mask"],"prefix":"10.1007","volume":"133","author":[{"given":"Yupeng","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daquan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yaxing","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiashi","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8388-8708","authenticated-orcid":false,"given":"Qibin","family":"Hou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,12]]},"reference":[{"key":"2294_CR1","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Lischinski, D., & Fried, O. (2022). Blended diffusion for text-driven editing of natural images. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 18208-18218).","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"2294_CR2","unstructured":"Balaji, Y., Nah, S., Huang, X., Vahdat, A., Song, J., Zhang, Q., Kreis, K., Aittala, M., Aila, T., Laine, S., Catanzaro, B., Karras, T., & Liu, M.Y. (2023). ediff-i: Text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324"},{"key":"2294_CR3","unstructured":"Bao, F., Li, C., Zhu, J., & Zhang, B. (2022). Analytic-dpm: An analytic estimate of the optimal reverse variance in diffusion probabilistic models. In International conference on learning representations"},{"key":"2294_CR4","unstructured":"Baranchuk, D., Rubachev, I., Voynov, A., Khrulkov, V., & Babenko, A. (2022). Label-efficient semantic segmentation with diffusion models. In International conference on learning representations"},{"key":"2294_CR5","unstructured":"Brack, M., Schramowski, P., Friedrich, F., Hintersdorf, D., & Kersting, K. (2022). The stable artist: Steering semantics in diffusion latent space. arXiv preprint arXiv:2212.06013"},{"key":"2294_CR6","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., & Askell, A., et al. (2020). Language models are few-shot learners. arXiv preprint arXiv:2005.14165"},{"key":"2294_CR7","unstructured":"Chang, H., Zhang, H., Barber, J., Maschinot, A., Lezama, J., Jiang, L., Yang, M.H., Murphy, K., Freeman, W.T., & Rubinstein, M., et al. (2023). Muse: Text-to-image generation via masked generative transformers. In International conference on machine learning (ICML). (pp. 4055\u20134075)."},{"issue":"4","key":"2294_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592116","volume":"42","author":"H Chefer","year":"2023","unstructured":"Chefer, H., Alaluf, Y., Vinker, Y., Wolf, L., & Cohen-Or, D. (2023). Attend-and-excite: Attention-based semantic guidance for text-to-image diffusion models. ACM Transactions on Graphics, 42(4), 1\u201310.","journal-title":"ACM Transactions on Graphics"},{"key":"2294_CR9","doi-asserted-by":"crossref","unstructured":"Chen, M., Laina, I., & Vedaldi, A. (2024). Training-free layout control with cross-attention guidance. In Winter conference on applications of computer vision. (pp. 5343\u20135353).","DOI":"10.1109\/WACV57701.2024.00526"},{"key":"2294_CR10","unstructured":"Chen, R.T.Q., Behrmann, J., Duvenaud, D., & Jacobsen, J. (2019). Residual flows for invertible generative modeling. In Advances in neural information processing systems (vol. 32)."},{"key":"2294_CR11","doi-asserted-by":"crossref","unstructured":"Chen, R., Chen, Y., Jiao, N., & Jia, K. (2023). Fantasia3d: Disentangling geometry and appearance for high-quality text-to-3d content creation. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 22246-22256).","DOI":"10.1109\/ICCV51070.2023.02033"},{"key":"2294_CR12","unstructured":"Cheng, J., Liang, X., Shi, X., He, T., Xiao, T., & Li, M. (2023). Layoutdiffuse: Adapting foundational diffusion models for layout-to-image generation. arXiv preprint arXiv:2302.08908"},{"key":"2294_CR13","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal, P., & Nichol, A. (2021). Diffusion models beat gans on image synthesis. Advances in Neural Information Processing Systems, 34, 8780\u20138794.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR14","first-page":"19822","volume":"34","author":"M Ding","year":"2021","unstructured":"Ding, M., Yang, Z., Hong, W., Zheng, W., Zhou, C., Yin, D., Lin, J., Zou, X., Shao, Z., Yang, H., & Tang, J. (2021). Cogview: Mastering text-to-image generation via transformers. Advances in Neural Information Processing Systems, 34, 19822\u201319835.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR15","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., & Ommer, B. (2021). Taming transformers for high-resolution image synthesis. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 12873-12883).","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"2294_CR16","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., & Ommer, B. (2021). Taming transformers for high-resolution image synthesis. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 12873-12883).","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"2294_CR17","unstructured":"Feng, W., He, X., Fu, T.J., Jampani, V., Akula, A., Narayana, P., Basu, S., Wang, X.E., & Wang, W.Y. (2023). Training-free structured diffusion guidance for compositional text-to-image synthesis. In International conference on learning representations."},{"issue":"4","key":"2294_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3528223.3530164","volume":"41","author":"R Gal","year":"2022","unstructured":"Gal, R., Patashnik, O., Maron, H., Bermano, A. H., Chechik, G., & Cohen-Or, D. (2022). Stylegan-nada: Clip-guided domain adaptation of image generators. ACM Transactions on Graphics, 41(4), 1\u201313.","journal-title":"ACM Transactions on Graphics"},{"key":"2294_CR19","doi-asserted-by":"crossref","unstructured":"Galatolo, F., Cimino, M., & Vaglini, G. (2021). Generating images from caption and vice versa via clip-guided generative latent space search. In Proceedings of the international conference on image processing and vision engineering","DOI":"10.5220\/0010503701660174"},{"key":"2294_CR20","doi-asserted-by":"crossref","unstructured":"Gao, S., Liu, X., Zeng, B., Xu, S., Li, Y., Luo, X., Liu, J., Zhen, X., & Zhang, B. (2023). Implicit diffusion models for continuous super-resolution. In Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 8998\u20139008)","DOI":"10.1109\/CVPR52729.2023.00966"},{"key":"2294_CR21","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., & Bengio, Y. (2014). Generative adversarial networks. In Advances in neural information processing systems (vol. 27)."},{"key":"2294_CR22","first-page":"14715","volume":"35","author":"A Graikos","year":"2022","unstructured":"Graikos, A., Malkin, N., Jojic, N., & Samaras, D. (2022). Diffusion models as plug-and-play priors. Advances in Neural Information Processing Systems, 35, 14715\u201314728.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR23","first-page":"23968","volume":"34","author":"M Grci\u0107","year":"2021","unstructured":"Grci\u0107, M., Grubi\u0161i\u0107, I., & \u0160egvi\u0107, S. (2021). Densely connected normalizing flows. Advances in Neural Information Processing Systems, 34, 23968\u201323982.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR24","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., & Cohen-Or, D. (2022). Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:2208.01626"},{"key":"2294_CR25","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. Advances in Neural Information Processing Systems, 33, 6840\u20136851.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"47","key":"2294_CR26","first-page":"1","volume":"23","author":"J Ho","year":"2021","unstructured":"Ho, J., Saharia, C., Chan, W., Fleet, D. J., Norouzi, M., & Salimans, T. (2021). Cascaded diffusion models for high fidelity image generation. Journal of Machine Learning Research (JMLR), 23(47), 1\u201333.","journal-title":"Journal of Machine Learning Research (JMLR)"},{"key":"2294_CR27","unstructured":"Ho, J., & Salimans, T. (2022). Classifier-free diffusion guidance. arXiv preprint arXiv:2207.12598"},{"issue":"1","key":"2294_CR28","first-page":"411","volume":"7","author":"M Honnibal","year":"2017","unstructured":"Honnibal, M., & Montani, I. (2017). Spacy 2: Natural language understanding with bloom embeddings, convolutional neural networks and incremental parsing. To Appear, 7(1), 411\u2013420.","journal-title":"To Appear"},{"key":"2294_CR29","first-page":"22863","volume":"34","author":"CW Huang","year":"2021","unstructured":"Huang, C. W., Lim, J., & Courville, A. (2021). A variational perspective on diffusion-based generative models and score matching. Advances in Neural Information Processing Systems, 34, 22863\u201322876.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR30","doi-asserted-by":"crossref","unstructured":"Jiang, Y., Zhou, Y., Liang, Y., Liu, W., Jiao, J., Quan, Y., & He, S. (2023). Diffuse3d: Wide-angle 3d photography via bilateral diffusion. In International conference on computer vision","DOI":"10.1109\/ICCV51070.2023.00826"},{"key":"2294_CR31","doi-asserted-by":"crossref","unstructured":"Kang, M., Zhu, J.Y., Zhang, R., Park, J., Shechtman, E., Paris, S., & Park, T. (2023). Scaling up gans for text-to-image synthesis. In  IEEE conference on computer vision and pattern recognition (pp. 10124\u201310134).","DOI":"10.1109\/CVPR52729.2023.00976"},{"key":"2294_CR32","doi-asserted-by":"crossref","unstructured":"Kawar, B., Zada, S., Lang, O., Tov, O., Chang, H., Dekel, T., Mosseri, I., & Irani, M. (2023). Imagic: Text-based real image editing with diffusion models. In  IEEE conference on computer vision and pattern recognition (pp. 6007\u20136017).","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"2294_CR33","doi-asserted-by":"crossref","unstructured":"Kumari, N., Zhang, B., Zhang, R., Shechtman, E., & Zhu, J.Y. (2023). Multi-concept customization of text-to-image diffusion. In  IEEE conference on computer vision and pattern recognition (pp. 1931\u20131941).","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"2294_CR34","doi-asserted-by":"crossref","unstructured":"Li, Z., Zhou, Q., Zhang, X., Zhang, Y., Wang, Y., & Xie, W. (2023). Open-vocabulary object segmentation with diffusion models. In International conference on computer vision (pp. 7667\u20137676).","DOI":"10.1109\/ICCV51070.2023.00705"},{"key":"2294_CR35","doi-asserted-by":"crossref","unstructured":"Liu, N., Li, S., Du, Y., Torralba, A., & Tenenbaum, J.B. (2022). Compositional visual generation with composable diffusion models. In European conference on computer vision (pp. 423\u2013439).","DOI":"10.1007\/978-3-031-19790-1_26"},{"key":"2294_CR36","first-page":"5775","volume":"35","author":"C Lu","year":"2022","unstructured":"Lu, C., Zhou, Y., Bao, F., Chen, J., Li, C., & Zhu, J. (2022). Dpm-solver: A fast ode solver for diffusion probabilistic model sampling in around 10 steps. Advances in Neural Information Processing Systems, 35, 5775\u20135787.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR37","doi-asserted-by":"crossref","unstructured":"Lugmayr, A., Danelljan, M., Romero, A., Yu, F., Timofte, R., & Van Gool, L. (2022). Repaint: Inpainting using denoising diffusion probabilistic models. In IEEE conference on computer vision and pattern recognition (pp. 11461\u201311471).","DOI":"10.1109\/CVPR52688.2022.01117"},{"key":"2294_CR38","doi-asserted-by":"publisher","first-page":"4098","DOI":"10.1609\/aaai.v38i5.28204","volume":"38","author":"WDK Ma","year":"2024","unstructured":"Ma, W. D. K., Lewis, J. P., Lahiri, A., Leung, T., & Kleijn, W. B. (2024). Directed diffusion: Direct control of object placement through attention guidance. AAAI, 38, 4098\u20134106.","journal-title":"AAAI"},{"key":"2294_CR39","unstructured":"Mansimov, E., Parisotto, E., Ba, J.L., & Salakhutdinov, R. (2016). Generating images from captions with attention. In International conference on learning representations"},{"key":"2294_CR40","doi-asserted-by":"crossref","unstructured":"Mao, J., & Wang, X. (2023). Training-free location-aware text-to-image synthesis. In IEEE international conference on image processing (pp. 995\u2013999).","DOI":"10.1109\/ICIP49359.2023.10222616"},{"key":"2294_CR41","doi-asserted-by":"crossref","unstructured":"Mou, C., Wang, X., Xie, L., Wu, Y., Zhang, J., Qi, Z., Shan, Y., & Qie, X. (2023). T2i-adapter: Learning adapters to dig out more controllable ability for text-to-image diffusion models. arXiv preprint arXiv:2302.08908","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"2294_CR42","unstructured":"Nichol, A., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., McGrew, B., Sutskever, I., & Chen, M. (2022). Glide: Towards photorealistic image generation and editing with text-guided diffusion models. In Proceedings of machine learning research (PMLR). (pp. 16784\u201316804)"},{"key":"2294_CR43","unstructured":"van den Oord, A., Vinyals, O., & kavukcuoglu, K. (2017). Neural discrete representation learning. In Advances in neural information processing systems (vol. 30)."},{"key":"2294_CR44","doi-asserted-by":"crossref","unstructured":"Parmar, G., Li, D., Lee, K., & Tu, Z. (2021). Dual contradistinctive generative autoencoder. In IEEE conference on computer vision and pattern recognition (pp. 823\u2013832).","DOI":"10.1109\/CVPR46437.2021.00088"},{"key":"2294_CR45","doi-asserted-by":"crossref","unstructured":"Peebles, W., & Xie, S. (2023). Scalable diffusion models with transformers. In Proceedings of the IEEE\/CVF international conference on computer vision. (pp. 4195\u20134205)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"2294_CR46","doi-asserted-by":"crossref","unstructured":"Phung, Q., Ge, S., & Huang, J.B. (2024). Grounded text-to-image synthesis with attention refocusing. In IEEE conference on computer vision and pattern recognition (pp. 7932\u20137942).","DOI":"10.1109\/CVPR52733.2024.00758"},{"key":"2294_CR47","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., & Clark, J., et al. (2021). Learning transferable visual models from natural language supervision. In International conference on machine learning (ICML) (pp. 8748\u20138763)."},{"key":"2294_CR48","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., Shazeer, N., Roberts, A., Lee, K., Narang, S., Matena, M., Zhou, Y., Li, W., & Liu, P. J. (2020). Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of Machine Learning Research (JMLR), 21, 1\u201367.","journal-title":"Journal of Machine Learning Research (JMLR)"},{"key":"2294_CR49","doi-asserted-by":"crossref","unstructured":"Raj, A., Kaza, S., Poole, B., Niemeyer, M., Ruiz, N., Mildenhall, B., Zada, S., Aberman, K., Rubinstein, M., & Barron, J., et\u00a0al. (2023). Dreambooth3d: Subject-driven text-to-3d generation. In International conference on computer vision (pp. 2349\u20132359).","DOI":"10.1109\/ICCV51070.2023.00223"},{"key":"2294_CR50","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 1(2), 3"},{"key":"2294_CR51","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., & Sutskever, I. (2021). Zero-shot text-to-image generation. In International conference on machine learning (ICML) (pp. 8821\u20138831)."},{"key":"2294_CR52","unstructured":"Razavi, A., van\u00a0den Oord, A., & Vinyals, O. (2019). Generating diverse high-fidelity images with vq-vae-2. In Advances in neural information processing systems (vol. 32)."},{"key":"2294_CR53","unstructured":"Reed, S., Akata, Z., Yan, X., Logeswaran, L., Schiele, B., & Lee, H. (2016). Generative adversarial text to image synthesis. In: International Conference on Machine Learning (ICML). pp. 1060\u20131069. PMLR"},{"key":"2294_CR54","unstructured":"Ren, T., Liu, S., Zeng, A., Lin, J., Li, K., Cao, H., Chen, J., Huang, X., Chen, Y., & Yan, F., et\u00a0al. (2024). Grounded sam: Assembling open-world models for diverse visual tasks. arXiv preprint arXiv:2401.14159"},{"key":"2294_CR55","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In  IEEE conference on computer vision and pattern recognition (pp. 10684\u201310695).","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2294_CR56","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. In textit Medical image computing and computer assisted intervention (pp. 234\u2013241).","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2294_CR57","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al. (2022). Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems, 35, 36479\u201336494.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"4","key":"2294_CR58","first-page":"4713","volume":"45","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Ho, J., Chan, W., Salimans, T., Fleet, D. J., & Norouzi, M. (2022). Image super-resolution via iterative refinement. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(4), 4713\u20134726.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"4","key":"2294_CR59","first-page":"4713","volume":"45","author":"C Saharia","year":"2023","unstructured":"Saharia, C., Ho, J., Chan, W., Salimans, T., Fleet, D. J., & Norouzi, M. (2023). Image super-resolution via iterative refinement. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(4), 4713\u20134726.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2294_CR60","first-page":"25278","volume":"35","author":"C Schuhmann","year":"2022","unstructured":"Schuhmann, C., Beaumont, R., Vencu, R., Gordon, C., Wightman, R., Cherti, M., Coombes, T., Katta, A., Mullis, C., Wortsman, M., et al. (2022). Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems, 35, 25278\u201325294.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR61","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., & Ganguli, S. (2015). Deep unsupervised learning using nonequilibrium thermodynamics. In International conference on machine learning (ICML) (pp. 2256\u20132265)."},{"key":"2294_CR62","unstructured":"Song, J., Meng, C., & Ermon, S. (2020). Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502"},{"key":"2294_CR63","unstructured":"Song, Y., & Ermon, S. (2019). Generative modeling by estimating gradients of the data distribution. In Advances in neural information processing systems (vol. 32)."},{"key":"2294_CR64","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D., Kumar, A., Ermon, S., & Poole, B. (2021). Score-based generative modeling through stochastic differential equations. In International conference on learning representations"},{"key":"2294_CR65","doi-asserted-by":"crossref","unstructured":"Tao, M., Bao, B.K., Tang, H., & Xu, C. (2023). Galip: Generative adversarial clips for text-to-image synthesis. In IEEE conference on computer vision and pattern recognition (pp. 14214\u201314223).","DOI":"10.1109\/CVPR52729.2023.01366"},{"key":"2294_CR66","doi-asserted-by":"crossref","unstructured":"Thrush, T., Jiang, R., Bartolo, M., Singh, A., Williams, A., Kiela, D., & Ross, C. (2022). Winoground: Probing vision and language models for visio-linguistic compositionality. In IEEE conference on computer vision and pattern recognition (pp. 5238\u20135248).","DOI":"10.1109\/CVPR52688.2022.00517"},{"key":"2294_CR67","unstructured":"Van Den\u00a0Oord, A., & Vinyals, O., et al. (2017). Neural discrete representation learning. In Advances in neural information processing systems (vol. 30)."},{"key":"2294_CR68","doi-asserted-by":"publisher","first-page":"5544","DOI":"10.1609\/aaai.v38i6.28364","volume":"38","author":"R Wang","year":"2024","unstructured":"Wang, R., Chen, Z., Chen, C., Ma, J., Lu, H., & Lin, X. (2024). Compositional text-to-image synthesis with attention map control of diffusion models. AAAI, 38, 5544\u20135552.","journal-title":"AAAI"},{"key":"2294_CR69","unstructured":"Wang, Z., Zheng, H., He, P., Chen, W., & Zhou, M. (2022). Diffusion-gan: Training gans with diffusion. arXiv preprint arXiv:2206.02262"},{"key":"2294_CR70","doi-asserted-by":"crossref","unstructured":"Wu, Q., Liu, Y., Zhao, H., Bui, T., Lin, Z., Zhang, Y., & Chang, S. (2023). Harnessing the spatial-temporal attention of diffusion models for high-fidelity text-to-image synthesis. In International conference on computer vision (pp. 7766\u20137776).","DOI":"10.1109\/ICCV51070.2023.00714"},{"key":"2294_CR71","doi-asserted-by":"crossref","unstructured":"Wyatt, J., Leach, A., Schmon, S., & Willcocks, C. (2022). Anoddpm: Anomaly detection with denoising diffusion probabilistic models using simplex noise. In IEEE conference on computer vision and pattern recognition Workshops (pp. 650\u2013656).","DOI":"10.1109\/CVPRW56347.2022.00080"},{"key":"2294_CR72","doi-asserted-by":"crossref","unstructured":"Xie, S., Zhang, Z., Lin, Z., Hinz, T., & Zhang, K. (2023). Smartbrush: Text and shape guided object inpainting with diffusion model. In IEEE conference on computer vision and pattern recognition (pp. 22428\u201322437).","DOI":"10.1109\/CVPR52729.2023.02148"},{"key":"2294_CR73","doi-asserted-by":"crossref","unstructured":"Xu, T., Zhang, P., Huang, Q., Zhang, H., Gan, Z., Huang, X., & He, X. (2018). Attngan: Fine-grained text to image generation with attentional generative adversarial networks. In IEEE conference on computer vision and pattern recognition (pp. 1316-1324).","DOI":"10.1109\/CVPR.2018.00143"},{"key":"2294_CR74","doi-asserted-by":"crossref","unstructured":"Xu, X., Wang, Z., Zhang, G., Wang, K., & Shi, H. (2023). Versatile diffusion: Text, images and variations all in one diffusion model. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 7754\u20137765).","DOI":"10.1109\/ICCV51070.2023.00713"},{"key":"2294_CR75","doi-asserted-by":"crossref","unstructured":"Yang, B., Gu, S., Zhang, B., Zhang, T., Chen, X., Sun, X., Chen, D., & Wen, F. (2022). Paint by example: Exemplar-based image editing with diffusion models. In IEEE conference on computer vision and pattern recognition (pp. 18381\u201318391).","DOI":"10.1109\/CVPR52729.2023.01763"},{"key":"2294_CR76","doi-asserted-by":"crossref","unstructured":"Yang, X., Zhou, D., Feng, J., & Wang, X. (2023). Diffusion probabilistic model made slim. In IEEE conference on computer vision and pattern recognition (pp. 22552\u201322562).","DOI":"10.1109\/CVPR52729.2023.02160"},{"issue":"3","key":"2294_CR77","first-page":"5","volume":"2","author":"J Yu","year":"2022","unstructured":"Yu, J., Xu, Y., Koh, J. Y., Luong, T., Baid, G., Wang, Z., Vasudevan, V., Ku, A., Yang, Y., Ayan, B. K., et al. (2022). Scaling autoregressive models for content-rich text-to-image generation. Transactions on Machine Learning Research, 2(3), 5.","journal-title":"Transactions on Machine Learning Research"},{"key":"2294_CR78","first-page":"10021","volume":"35","author":"X Zeng","year":"2022","unstructured":"Zeng, X., Vahdat, A., Williams, F., Gojcic, Z., Litany, O., Fidler, S., & Kreis, K. (2022). Lion: Latent point diffusion models for 3d shape generation. Advances in Neural Information Processing Systems, 35, 10021\u201310039.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2294_CR79","doi-asserted-by":"crossref","unstructured":"Zhang, H., Xu, T., Li, H., Zhang, S., Wang, X., Huang, X., & Metaxas, D. (2017). Stackgan: Text to photo-realistic image synthesis with stacked generative adversarial networks. In International conference on computer vision (pp. 5907\u20135915).","DOI":"10.1109\/ICCV.2017.629"},{"issue":"8","key":"2294_CR80","doi-asserted-by":"publisher","first-page":"1947","DOI":"10.1109\/TPAMI.2018.2856256","volume":"41","author":"H Zhang","year":"2019","unstructured":"Zhang, H., Xu, T., Li, H., Zhang, S., Wang, X., Huang, X., & Metaxas, D. N. (2019). Stackgan++: Realistic image synthesis with stacked generative adversarial networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(8), 1947\u20131962.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2294_CR81","doi-asserted-by":"crossref","unstructured":"Zhang, L., & Agrawala, M. (2023). Adding conditional control to text-to-image diffusion models. In International conference on computer vision (pp. 3836\u20133847).","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"2294_CR82","unstructured":"Zhang, Q., & Chen, Y. (2023). Fast sampling of diffusion models with exponential integrator. In International conference on learning representations"},{"key":"2294_CR83","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Huang, X., Ma, J., Li, Z., Luo, Z., Xie, Y., Qin, Y., Luo, T., Li, Y., & Liu, S., et al. (2024). Recognize anything: A strong image tagging model. In IEEE conference on computer vision and pattern recognition (pp. 1724\u20131732).","DOI":"10.1109\/CVPRW63382.2024.00179"},{"key":"2294_CR84","doi-asserted-by":"crossref","unstructured":"Zheng, G., Zhou, X., Li, X., Qi, Z., Shan, Y., & Li, X. (2023). Layoutdiffusion: Controllable diffusion model for layout-to-image generation. In IEEE conference on computer vision and pattern recognition (pp. 22490\u201322499).","DOI":"10.1109\/CVPR52729.2023.02154"},{"key":"2294_CR85","unstructured":"Zhou, D., Wang, W., Yan, H., Lv, W., Zhu, Y., & Feng, J. (2023). Magicvideo: Efficient video generation with latent diffusion models. arXiv preprint arXiv:2211.11018"},{"key":"2294_CR86","doi-asserted-by":"crossref","unstructured":"Zhou, L., Du, Y., & Wu, J. (2021). 3d shape generation and completion through point-voxel diffusion. In International conference on computer vision (pp. 5826\u20135835).","DOI":"10.1109\/ICCV48922.2021.00577"},{"key":"2294_CR87","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Zhang, R., Chen, C., Li, C., Tensmeyer, C., Yu, T., Gu, J., Xu, J., & Sun, T. (2022). Towards language-free training for text-to-image generation. In IEEE conference on computer vision and pattern recognition (pp. 17907\u201317917).","DOI":"10.1109\/CVPR52688.2022.01738"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02294-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02294-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02294-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,17]],"date-time":"2025-04-17T06:03:58Z","timestamp":1744869838000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02294-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,12]]},"references-count":87,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["2294"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02294-2","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,12]]},"assertion":[{"value":"29 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 October 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}