{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T11:42:39Z","timestamp":1783597359860,"version":"3.55.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:00:00Z","timestamp":1729814400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:00:00Z","timestamp":1729814400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National key R&D Program","award":["No.2022ZD0161000"],"award-info":[{"award-number":["No.2022ZD0161000"]}]},{"name":"General Research Fund of Hong Kong","award":["No.17200622 and 17209324"],"award-info":[{"award-number":["No.17200622 and 17209324"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s11263-024-02253-x","type":"journal-article","created":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T10:02:20Z","timestamp":1729850540000},"page":"1894-1911","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":17,"title":["StyleAdapter: A Unified Stylized Image Generation Model"],"prefix":"10.1007","volume":"133","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4677-5760","authenticated-orcid":false,"given":"Zhouxia","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xintao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liangbin","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongang","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Shan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenping","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ping","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,25]]},"reference":[{"key":"2253_CR1","unstructured":"https:\/\/civitai.com, https:\/\/wall.alphacoders.com, and https:\/\/foreverclassicgames.com"},{"key":"2253_CR2","unstructured":"https:\/\/huggingface.co\/openai\/clip-vit-large-patch14"},{"key":"2253_CR3","unstructured":"Brock, A., Donahue, J., & Simonyan, K. (2018). Large scale gan training for high fidelity natural image synthesis. arXiv preprint arXiv:1809.11096"},{"key":"2253_CR4","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02127-2","author":"T Chen","year":"2024","unstructured":"Chen, T., Pu, T., Liu, L., Shi, Y., Yang, Z., & Lin, L. (2024). Heterogeneous semantic transfer for multi-label recognition with partial labels. International Journal of Computer Vision. https:\/\/doi.org\/10.1007\/s11263-024-02127-2","journal-title":"International Journal of Computer Vision"},{"key":"2253_CR5","first-page":"10663362","volume":"33","author":"T Chen","year":"2024","unstructured":"Chen, T., Wang, W., Pu, T., Qin, J., Yang, Z., Liu, J., & Lin, L. (2024). Dynamic correlation learning and regularization for multi-label confidence calibration. IEEE Transactions on Image Processing, 33, 10663362.","journal-title":"IEEE Transactions on Image Processing"},{"key":"2253_CR6","doi-asserted-by":"crossref","unstructured":"Deng, Y., Tang, F., Dong, W., Ma, C., Pan, X., Wang, L., & Xu, C. (2022). Stytr2: Image style transfer with transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 11,326\u201311,336","DOI":"10.1109\/CVPR52688.2022.01104"},{"key":"2253_CR7","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal, P., & Nichol, A. (2021). Diffusion models beat Gans on image synthesis. Advances in neural information processing systems, 34, 8780\u20138794.","journal-title":"Advances in neural information processing systems"},{"key":"2253_CR8","first-page":"19822","volume":"34","author":"M Ding","year":"2021","unstructured":"Ding, M., Yang, Z., Hong, W., Zheng, W., Zhou, C., Yin, D., Lin, J., Zou, X., Shao, Z., Yang, H., et al. (2021). Cogview: Mastering text-to-image generation via transformers. Advances in neural information processing systems, 34, 19822\u201319835.","journal-title":"Advances in neural information processing systems"},{"key":"2253_CR9","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et al. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations."},{"key":"2253_CR10","doi-asserted-by":"crossref","unstructured":"Gafni, O., Polyak, A., Ashual, O., Sheynin, S., Parikh, D., & Taigman, Y. (2022). Make-a-scene: Scene-based text-to-image generation with human priors. In: European Conference on Computer Vision, pp. 89\u2013106. Springer: Switzerland.","DOI":"10.1007\/978-3-031-19784-0_6"},{"key":"2253_CR11","unstructured":"Gal, R., Alaluf, Y., Atzmon, Y., Patashnik, O., Bermano, A.H., Chechik, G., & Cohen-Or, D. (2022). An image is worth one word: Personalizing text-to-image generation using textual inversion. arXiv preprint arXiv:2208.01618"},{"key":"2253_CR12","doi-asserted-by":"crossref","unstructured":"Gatys, L.A., Ecker, A.S., & Bethge, M. (2016). Image style transfer using convolutional neural networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2414\u20132423","DOI":"10.1109\/CVPR.2016.265"},{"key":"2253_CR13","doi-asserted-by":"crossref","unstructured":"Gatys, L.A., Ecker, A.S., Bethge, M., Hertzmann, A., & Shechtman, E. (2017). Controlling perceptual factors in neural style transfer. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3985\u20133993","DOI":"10.1109\/CVPR.2017.397"},{"key":"2253_CR14","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising diffusion probabilistic models. Advances in neural information processing systems, 33, 6840\u20136851.","journal-title":"Advances in neural information processing systems"},{"issue":"47","key":"2253_CR15","first-page":"1","volume":"23","author":"J Ho","year":"2022","unstructured":"Ho, J., Saharia, C., Chan, W., Fleet, D. J., Norouzi, M., & Salimans, T. (2022). Cascaded diffusion models for high fidelity image generation. Journal of Machine Learning Research, 23(47), 1\u201333.","journal-title":"Journal of Machine Learning Research"},{"key":"2253_CR16","unstructured":"Hu, E.J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685"},{"key":"2253_CR17","unstructured":"Kingma, D.P., Ba, J. (2014). Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"key":"2253_CR18","doi-asserted-by":"crossref","unstructured":"Kolkin, N., Salavon, J., & Shakhnarovich, G. (2019). Style transfer by relaxed optimal transport and self-similarity. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10,051\u201310,060","DOI":"10.1109\/CVPR.2019.01029"},{"key":"2253_CR19","unstructured":"Li, B., Qi, X., Lukasiewicz, T., & Torr, P. (2019). Controllable text-to-image generation. Advances in neural information processing systems 32"},{"key":"2253_CR20","unstructured":"Li, J., Li, D., Savarese, S., & Hoi, S. (2023). Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International conference on machine learning 19: 730\u2013742"},{"issue":"5","key":"2253_CR21","doi-asserted-by":"crossref","first-page":"4296","DOI":"10.1609\/aaai.v38i5.28226","volume":"38","author":"C Mou","year":"2024","unstructured":"Mou, C., Wang, X., Xie, L., Wu, Y., Zhang, J., Qi, Z., & Shan, Y. (2024). T2i-adapter: Learning adapters to dig out more controllable ability for text-to-image diffusion models. Proceedings of the AAAI Conference on Artificial Intelligence, 38(5), 4296\u20134304.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2253_CR22","unstructured":"Nichol, A.Q., Dhariwal, P. (2021). Improved denoising diffusion probabilistic models. In: International conference on machine learning, pp. 8162\u20138171. PMLR"},{"key":"2253_CR23","unstructured":"Nichol, A.Q., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., Mcgrew, B., Sutskever, I., & Chen, M. (2022). Glide: Towards photorealistic image generation and editing with text-guided diffusion models. In: International Conference on Machine Learning."},{"key":"2253_CR24","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J., & Rombach, R. (2024). Sdxl: Improving latent diffusion models for high-resolution image synthesis. In: International Conference on Learning Representations."},{"key":"2253_CR25","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al. (2021). Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp. 8748\u20138763. PMLR"},{"key":"2253_CR26","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.061251(2), 3"},{"key":"2253_CR27","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., & Sutskever, I. (2021). Zero-shot text-to-image generation. In: International conference on machine learning, pp. 8821\u20138831. PMLR"},{"key":"2253_CR28","unstructured":"Reed, S., Akata, Z., Yan, X., Logeswaran, L., Schiele, B., & Lee, H. (2016). Generative adversarial text to image synthesis. In: International conference on machine learning, pp. 1060\u20131069. PMLR"},{"key":"2253_CR29","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10,684\u201310,695","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2253_CR30","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. In: MICCAI.","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2253_CR31","first-page":"500","volume":"22","author":"N Ruiz","year":"2023","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., & Aberman, K. (2023). Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 22, 500\u2013510.","journal-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition"},{"key":"2253_CR32","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E. L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al. (2022). Photorealistic text-to-image diffusion models with deep language understanding. Advances in neural information processing systems, 35, 36479\u201336494.","journal-title":"Advances in neural information processing systems"},{"issue":"4","key":"2253_CR33","first-page":"4713","volume":"45","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Ho, J., Chan, W., Salimans, T., Fleet, D. J., & Norouzi, M. (2022). Image super-resolution via iterative refinement. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(4), 4713\u20134726.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2253_CR34","first-page":"25278","volume":"35","author":"C Schuhmann","year":"2022","unstructured":"Schuhmann, C., Beaumont, R., Vencu, R., Gordon, C., Wightman, R., Cherti, M., Coombes, T., Katta, A., Mullis, C., Wortsman, M., et al. (2022). Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems, 35, 25278\u201325294.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2253_CR35","unstructured":"Seitzer, M. (2020). pytorch-fid: FID Score for PyTorch. https:\/\/github.com\/mseitzer\/pytorch-fid"},{"key":"2253_CR36","unstructured":"Sohn, K., Ruiz, N., Lee, K., Chin, D.C., Blok, I., Chang, H., Barber, J., Jiang, L., Entis, G., Li, Y., et al. (2023). Styledrop: Text-to-image generation in any style. Advances in Neural Information Processing Systems."},{"key":"2253_CR37","unstructured":"Song, J., Meng, C., & Ermon, S. (2020). Denoising diffusion implicit models. arXiv preprint arXiv:2010.02502"},{"key":"2253_CR38","unstructured":"Voynov, A., Chu, Q., Cohen-Or, D., & Aberman, K. (2023). $$ p+ $$: Extended textual conditioning in text-to-image generation. arXiv preprint arXiv:2303.09522"},{"issue":"3","key":"2253_CR39","doi-asserted-by":"crossref","first-page":"266","DOI":"10.1109\/TVCG.2004.1272726","volume":"10","author":"B Wang","year":"2004","unstructured":"Wang, B., Wang, W., Yang, H., & Sun, J. (2004). Efficient example-based painting and synthesis of 2d directional texture. IEEE Transactions on Visualization and Computer Graphics, 10(3), 266\u2013277.","journal-title":"IEEE Transactions on Visualization and Computer Graphics"},{"key":"2253_CR40","unstructured":"Wang, H., Wang, Q., Bai, X., Qin, Z., & Chen, A. (2024). Instantstyle: Free lunch towards style-preserving in text-to-image generation. arXiv preprint arXiv:2404.02733"},{"key":"2253_CR41","doi-asserted-by":"crossref","unstructured":"Wu, X., Hu, Z., Sheng, L., & Xu, D. (2021). Styleformer: Real-time arbitrary style transfer via parametric style composition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14,618\u201314,627","DOI":"10.1109\/ICCV48922.2021.01435"},{"key":"2253_CR42","doi-asserted-by":"crossref","unstructured":"Xu, T., Zhang, P., Huang, Q., Zhang, H., Gan, Z., Huang, X., & He, X. (2018). Attngan: Fine-grained text to image generation with attentional generative adversarial networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 1316\u20131324","DOI":"10.1109\/CVPR.2018.00143"},{"key":"2253_CR43","unstructured":"Ye, H., Zhang, J., Liu, S., Han, X., & Yang, W. (2023). Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:2308.06721"},{"issue":"8","key":"2253_CR44","doi-asserted-by":"crossref","first-page":"1947","DOI":"10.1109\/TPAMI.2018.2856256","volume":"41","author":"H Zhang","year":"2018","unstructured":"Zhang, H., Xu, T., Li, H., Zhang, S., Wang, X., Huang, X., & Metaxas, D. N. (2018). Stackgan++: Realistic image synthesis with stacked generative adversarial networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(8), 1947\u20131962.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2253_CR45","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., & Agrawala, M. (2023). Adding conditional control to text-to-image diffusion models. pp. 3836\u20133847","DOI":"10.1109\/ICCV51070.2023.00355"},{"issue":"7","key":"2253_CR46","doi-asserted-by":"crossref","first-page":"1594","DOI":"10.1109\/TMM.2013.2265675","volume":"15","author":"W Zhang","year":"2013","unstructured":"Zhang, W., Cao, C., Chen, S., Liu, J., & Tang, X. (2013). Style transfer via image component analysis. IEEE Transactions on Multimedia, 15(7), 1594\u20131601.","journal-title":"IEEE Transactions on Multimedia"},{"issue":"6","key":"2253_CR47","first-page":"1","volume":"42","author":"Y Zhang","year":"2023","unstructured":"Zhang, Y., Dong, W., Tang, F., Huang, N., Huang, H., Ma, C., Lee, T. Y., Deussen, O., & Xu, C. (2023). Prospect: Prompt spectrum for attribute-aware personalization of diffusion models. ACM Transactions on Graphics (TOG), 42(6), 1\u201314.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"2253_CR48","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Huang, N., Tang, F., Huang, H., Ma, C., Dong, W., & Xu, C. (2022). Inversion-based creativity transfer with diffusion models. arXiv preprint arXiv:2211.13203","DOI":"10.1109\/CVPR52729.2023.00978"},{"key":"2253_CR49","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Huang, N., Tang, F., Huang, H., Ma, C., Dong, W., & Xu, C. (2023). Inversion-based style transfer with diffusion models pp. 10,146\u201310,156","DOI":"10.1109\/CVPR52729.2023.00978"},{"key":"2253_CR50","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Tang, F., Dong, W., Huang, H., Ma, C., Lee, T.Y., & Xu, C. (2022). Domain enhanced arbitrary image style transfer via contrastive learning. In: ACM SIGGRAPH 2022 conference proceedings, pp. 1\u20138","DOI":"10.1145\/3528233.3530736"},{"key":"2253_CR51","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Zhang, R., Chen, C., Li, C., Tensmeyer, C., Yu, T., Gu, J., Xu, J., & Sun, T. (2021). Lafite: Towards language-free training for text-to-image generation. arXiv preprint arXiv:2111.13792","DOI":"10.1109\/CVPR52688.2022.01738"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02253-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02253-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02253-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T22:05:41Z","timestamp":1743372341000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02253-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,25]]},"references-count":51,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["2253"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02253-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,25]]},"assertion":[{"value":"6 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}