{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T14:13:51Z","timestamp":1783520031360,"version":"3.55.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T00:00:00Z","timestamp":1752192000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T00:00:00Z","timestamp":1752192000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s11263-025-02526-z","type":"journal-article","created":{"date-parts":[[2025,7,11]],"date-time":"2025-07-11T17:51:23Z","timestamp":1752256283000},"page":"7037-7053","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["DreamArtist: Controllable One-Shot Text-to-Image Generation via Positive-Negative Adapter"],"prefix":"10.1007","volume":"133","author":[{"given":"Ziyi","family":"Dong","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pengxu","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2248-3755","authenticated-orcid":false,"given":"Liang","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,7,11]]},"reference":[{"key":"2526_CR1","unstructured":"Anonymous, D.\u00a0community, G.\u00a0Branwen. (2022). Danbooru2021: A large-scale crowdsourced and tagged anime illustration dataset. URL https:\/\/www.gwern.net\/Danbooru2021"},{"key":"2526_CR2","unstructured":"Brock, A., Donahue, J., & Simonyan, K. (2019). Large Scale GAN Training for High Fidelity Natural Image Synthesis, In International Conference on Learning Representations."},{"key":"2526_CR3","unstructured":"Brown, T. B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., Agarwal, S., Herbert-Voss, A., Krueger, G., Henighan, T., Child, R., Ramesh, A., Ziegler, D. M., Wu, J., Winter, C., ... & D.\u00a0Amodei. (2020). Language Models are Few-Shot Learners, In Advances in Neural Information Processing Systems."},{"key":"2526_CR4","doi-asserted-by":"crossref","unstructured":"Cheng, B., Liu, Z., Peng, Y., & Lin, Y. (2023). General Image-to-Image Translation with One-Shot Image Guidance, In IEEE\/CVF International Conference on Computer Vision, ICCV 2023, Paris, France, October 1-6, 2023 , (pp. 22679\u201322689)","DOI":"10.1109\/ICCV51070.2023.02078"},{"key":"2526_CR5","doi-asserted-by":"crossref","unstructured":"Cheng, J., Wu, F., Tian, Y., Wang, L., & Tao, D. (2020). RiFeGAN: Rich Feature Generation for Text-to-Image Synthesis From Prior Knowledge, In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 10908\u201310917)","DOI":"10.1109\/CVPR42600.2020.01092"},{"key":"2526_CR6","doi-asserted-by":"crossref","unstructured":"Crowson, K., Biderman, S., Kornis, D., Stander, D., Hallahan, E., Castricato, L., & Raff, E. (2022). VQGAN-CLIP: Open Domain Image Generation and Editing with Natural Language Guidance, In ECCV, (pp. 88\u2013105)","DOI":"10.1007\/978-3-031-19836-6_6"},{"key":"2526_CR7","unstructured":"Dhariwal, P., & Nichol, A. Q. (2021). Diffusion Models Beat GANs on Image Synthesis, In Advances in Neural Information Processing Systems, (pp. 8780\u20138794)"},{"key":"2526_CR8","unstructured":"Ding, K., Wang, Y., Liu, P., Yu, Q., Zhang, H., Xiang, S., & Pan, C. (2022). Prompt tuning with soft context sharing for vision-language models. CoRR arXiv:2208.13474."},{"key":"2526_CR9","unstructured":"Ding, M., Yang, Z., Hong, W., Zheng, W., Zhou, C., Yin, D., Lin, J., Zou, X., Shao, Z., Yang, H., & Tang, J. (2021). CogView: Mastering Text-to-Image Generation via Transformers, In Advances in Neural Information Processing Systems, pp. (19822\u201319835)"},{"key":"2526_CR10","unstructured":"Gal, R., Alaluf, Y., Atzmon, Y., Patashnik, O., Bermano, A. H., Chechik, G., & Cohen-Or, D. (2023). An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion, In International Conference on Learning Representations."},{"key":"2526_CR11","doi-asserted-by":"crossref","unstructured":"Gatys, L. A., Ecker, A. S. & Bethge, M. (2016). Image Style Transfer Using Convolutional Neural Networks, In IEEE Conference on Computer Vision and Pattern Recognition, (pp. 2414\u20132423)","DOI":"10.1109\/CVPR.2016.265"},{"key":"2526_CR12","unstructured":"Ho, J., & Salimans, T. (2021). Classifier-Free Diffusion Guidance, In NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications."},{"key":"2526_CR13","unstructured":"Ho, J., Jain, A., & Abbeel, P. (2020). Denoising Diffusion Probabilistic Models, In Advances in Neural Information Processing Systems."},{"key":"2526_CR14","unstructured":"Hu, E. J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2022). LoRA: Low-Rank Adaptation of Large Language Models, In ICLR The Tenth International Conference on Learning Representations."},{"key":"2526_CR15","doi-asserted-by":"crossref","unstructured":"Jain, A., Mildenhall, B., Barron, J. T., Abbeel, P. & Poole, B. (2022). Zero-Shot Text-Guided Object Generation with Dream Fields, In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 857\u2013866)","DOI":"10.1109\/CVPR52688.2022.00094"},{"key":"2526_CR16","unstructured":"Karras, T., Aittala, M., Laine, S., H\u00e4rk\u00f6nen, E., Hellsten, J., Lehtinen, J., Aila, T. (2021). Alias-Free Generative Adversarial Networks, In Advances in Neural Information Processing Systems, (pp. 852\u2013863)"},{"key":"2526_CR17","unstructured":"Kingma, D. P., & Dhariwal, P. (2018). Glow: Generative Flow with Invertible 1x1 Convolutions, In Advances in Neural Information Processing Systems, (pp. 10236\u201310245)"},{"key":"2526_CR18","doi-asserted-by":"crossref","unstructured":"Kumari, N., Zhang, B., Zhang, R., Shechtman, E., & Zhu, J. Y. (2023). Multi-Concept Customization of Text-to-Image Diffusion, In IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"2526_CR19","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., & Constant, N. (2021). The Power of Scale for Parameter-Efficient Prompt Tuning, In Proceedings of the Conference on Empirical Methods in Natural Language Processing, EMNLP, (pp. 3045\u20133059)","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"2526_CR20","doi-asserted-by":"crossref","unstructured":"Li, X. L. & Liang, P. (2021). Prefix-Tuning: Optimizing Continuous Prompts for Generation, In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, ACL\/IJCNLP, pp. 4582\u20134597","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"2526_CR21","unstructured":"Li, B., Qi, X., Lukasiewicz, T., & Torr, P. H. S. (2019). Controllable Text-to-Image Generation, In Advances in Neural Information Processing Systems, (pp. 2063\u20132073)"},{"key":"2526_CR22","doi-asserted-by":"crossref","unstructured":"Liu, X., Ji, K., Fu, Y., Du, Z., Yang, Z., & Tang, J. (2021). P-tuning v2: Prompt tuning can be comparable to fine-tuning universally across scales and tasks. CoRR arXiv:2110.07602.","DOI":"10.18653\/v1\/2022.acl-short.8"},{"key":"2526_CR23","unstructured":"Liu, X., Park, D. H., Azadi, S., Zhang, G., Chopikyan, A., Hu, Y., Shi, H., Rohrbach, A., & Darrell, T. (2021). More control for free! image synthesis with semantic diffusion guidance. CoRR arXiv:2112.05744."},{"key":"2526_CR24","doi-asserted-by":"crossref","unstructured":"Liu, Y., Peng, J., Yu, J. J. Q., & Wu, Y. (2019). PPGAN: Privacy-Preserving Generative Adversarial Network, In IEEE International Conference on Parallel and Distributed Systems, (pp. 985\u2013989)","DOI":"10.1109\/ICPADS47876.2019.00150"},{"key":"2526_CR25","doi-asserted-by":"crossref","unstructured":"Mou, C., Wang, X., Xie, L., Wu, Y., Zhang, J., Qi, Z., & Shan, Y. (2024). T2I-Adapter: Learning Adapters to Dig Out More Controllable Ability for Text-to-Image Diffusion Models, In Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI 2024, (pp. 4296\u20134304)","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"2526_CR26","unstructured":"Nichol, A. Q., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., McGrew, B., Sutskever, I., & Chen, M. (2022). GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models, In International Conference on Machine Learning, (pp. 16784\u201316804)"},{"key":"2526_CR27","doi-asserted-by":"crossref","unstructured":"Qiao, T., Zhang, J., Xu, D., & Tao, D. (2019). MirrorGAN: Learning Text-To-Image Generation by Redescription, In IEEE Conference on Computer Vision and Pattern Recognition, (pp. 1505\u20131514)","DOI":"10.1109\/CVPR.2019.00160"},{"key":"2526_CR28","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with CLIP latents. CoRR arXiv:2204.06125."},{"key":"2526_CR29","unstructured":"Reed, S. E., Akata, Z., Yan, X., Logeswaran, L., Schiele, B. & Lee, H. (2016). Generative Adversarial Text to Image Synthesis, In Proceedings of the International Conference on Machine Learning, vol.\u00a048, (pp. 1060\u20131069)"},{"key":"2526_CR30","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., & Ommer, B. (2022). High-Resolution Image Synthesis with Latent Diffusion Models, In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 10674\u201310685)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2526_CR31","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-Net: Convolutional Networks for Biomedical Image Segmentation, In Medical Image Computing and Computer-Assisted Intervention, MICCAI, vol. 9351, (pp. 234\u2013241)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2526_CR32","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., & Aberman, K. (2022) Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. CoRR arXiv:2208.12242.","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"2526_CR33","doi-asserted-by":"crossref","unstructured":"Saharia, C., Chan, W., Chang, H., Lee, C. A., Ho, J., Salimans, T., Fleet, D. J., & Norouzi, M. (2022). Palette: Image-to-Image Diffusion Models, In SIGGRAPH \u201922: Special Interest Group on Computer Graphics and Interactive Techniques Conference, (pp. 15:1\u201315:10)","DOI":"10.1145\/3528233.3530757"},{"key":"2526_CR34","doi-asserted-by":"crossref","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E., Ghasemipour, S. K. S., Ayan, B. K., Mahdavi, S. S., Lopes, R. G., Salimans, T., Ho, J., Fleet, D. J., & Norouzi, M. (2022). Photorealistic text-to-image diffusion models with deep language understanding. CoRR arXiv:2205.11487.","DOI":"10.1145\/3528233.3530757"},{"key":"2526_CR35","doi-asserted-by":"crossref","unstructured":"Schick, T., & Sch\u00fctze, H. (2021). Exploiting Cloze-Questions for Few-Shot Text Classification and Natural Language Inference, In Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics, EACL, (pp. 255\u2013269)","DOI":"10.18653\/v1\/2021.eacl-main.20"},{"key":"2526_CR36","unstructured":"Schuhmann, C., Beaumont, R., Vencu, R., Gordon, C., Wightman, R., Cherti, M., Coombes, T., Katta, A., Mullis, C., Wortsman, M., Schramowski, P., Kundurthy, S., Crowson, K., Schmidt, L., Kaczmarczyk, R. & Jitsev, J. (2022). LAION-5B: an open large-scale dataset for training next generation image-text models. CoRR arXiv:2210.08402."},{"key":"2526_CR37","unstructured":"Song, J., Meng, C., & Ermon, S. (2021). Denoising Diffusion Implicit Models, In International Conference on Learning Representations."},{"key":"2526_CR38","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D. P., Kumar, A., Ermon, S., & Poole, B. (2021). Score-Based Generative Modeling through Stochastic Differential Equations, In International Conference on Learning Representations."},{"key":"2526_CR39","doi-asserted-by":"crossref","unstructured":"Tang, R., Pandey, A., Jiang, Z., Yang, G., Kumar, K., Lin, J., & Ture, F. (2022). What the DAAM: interpreting stable diffusion using cross attention. CoRR arXiv:2210.04885.","DOI":"10.18653\/v1\/2023.acl-long.310"},{"key":"2526_CR40","unstructured":"Tao, M., Tang, H., Wu, S., Sebe, N., Wu, F., & Jing, X. (2020). DF-GAN: deep fusion generative adversarial networks for text-to-image synthesis. CoRR arXiv:2008.05865."},{"key":"2526_CR41","unstructured":"Ye, H., Zhang, J., Liu, S., Han, X., & Yang, W. (2023). Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. CoRR arXiv:2308.06721."},{"key":"2526_CR42","unstructured":"Yu, J., Xu, Y., Koh, J. Y., Luong, T., Baid, G., Wang, Z., Vasudevan, V., Ku, A., Yang, Y., Ayan, B. K., Hutchinson, B., Han, W., Parekh, Z., Li, X., Zhang, H., Baldridge, J., & Wu, Y. (2022). Scaling autoregressive models for content-rich text-to-image generation. CoRR arXiv:2206.10789."},{"key":"2526_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A. A., Shechtman, E. & Wang, O. (2018). The Unreasonable Effectiveness of Deep Features as a Perceptual Metric, In IEEE Conference on Computer Vision and Pattern Recognition, (pp. 586\u2013595)","DOI":"10.1109\/CVPR.2018.00068"},{"key":"2526_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., & Agrawala, M. (2023). Adding Conditional Control to Text-to-Image Diffusion Models, In IEEE\/CVF International Conference on Computer Vision, (pp. 3813\u20133824)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"2526_CR45","unstructured":"Zhang, H., Yin, W., Fang, Y., Li, L., Duan, B., Wu, Z., Sun, Y., Tian, H., Wu, H., & Wang, H. (2021). Ernie-vilg: Unified generative pre-training for bidirectional vision-language generation. CoRR arXiv:2112.15283."},{"key":"2526_CR46","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C. C., & Liu, Z. (2022). Conditional Prompt Learning for Vision-Language Models, In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 16795\u201316804)","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"9","key":"2526_CR47","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C. C., & Liu, Z. (2022). Learning to prompt for vision-language models. Int. J. Comput. Vis., 130(9), 2337\u20132348.","journal-title":"Int. J. Comput. Vis."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02526-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02526-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02526-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,10]],"date-time":"2025-10-10T08:51:35Z","timestamp":1760086295000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02526-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,11]]},"references-count":47,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["2526"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02526-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,7,11]]},"assertion":[{"value":"14 June 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 July 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}