{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:00:56Z","timestamp":1777568456605,"version":"3.51.4"},"publisher-location":"Cham","reference-count":62,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729720","type":"print"},{"value":"9783031729737","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72973-7_24","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:03:04Z","timestamp":1730383384000},"page":"410-426","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Customized Generation Reimagined: Fidelity and\u00a0Editability Harmonized"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-0466-3121","authenticated-orcid":false,"given":"Jian","family":"Jin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6344-9951","authenticated-orcid":false,"given":"Yang","family":"Shen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenyong","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"24_CR1","unstructured":"Brown, T., et\u00a0al.: Language models are few-shot learners. In: NeurIPS, pp. 1877\u20131901 (2020)"},{"key":"24_CR2","unstructured":"Chen, W., et al.: Subject-driven text-to-image generation via apprenticeship learning. Adv. Neural Inf. Proce. Syst. 36 (2024)"},{"key":"24_CR3","unstructured":"Couairon, G., Verbeek, J., Schwenk, H., Cord, M.: Diffedit: diffusion-based semantic image editing with mask guidance. In: ICLR (2023)"},{"key":"24_CR4","unstructured":"Ding, M., et\u00a0al.: Cogview: Mastering text-to-image generation via transformers. In: NeurIPS, pp. 19822\u201319835 (2021)"},{"key":"24_CR5","doi-asserted-by":"crossref","unstructured":"Du, Y., Wei, F., Zhang, Z., Shi, M., Gao, Y., Li, G.: Learning to prompt for open-vocabulary object detection with vision-language model. In: CVPR, pp. 14084\u201314093 (2022)","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"24_CR6","unstructured":"Feng, W., et al.: Layoutgpt: compositional visual planning and generation with large language models. In: NeurIPS, pp. 18225\u201318250 (2024)"},{"key":"24_CR7","unstructured":"Gal, R., et al.: An image is worth one word: personalizing text-to-image generation using textual inversion. In: ICLR (2023)"},{"key":"24_CR8","doi-asserted-by":"crossref","unstructured":"Gal, R., Arar, M., Atzmon, Y., Bermano, A.H., Chechik, G., Cohen-Or, D.: Designing an encoder for fast personalization of text-to-image models. In: Siggraph (2023)","DOI":"10.1145\/3610548.3618173"},{"key":"24_CR9","doi-asserted-by":"crossref","unstructured":"Gao, T., Fisch, A., Chen, D.: Making pre-trained language models better few-shot learners. In: ACL, pp. 3816\u20133830 (2021)","DOI":"10.18653\/v1\/2021.acl-long.295"},{"key":"24_CR10","unstructured":"Goodfellow, I., et al.: Generative adversarial nets. In: NeurIPS, pp. 2672\u20132680 (2014)"},{"key":"24_CR11","unstructured":"Gu, Y., et\u00a0al.: Mix-of-show: decentralized low-rank adaptation for multi-concept customization of diffusion models. In: NeurIPS, pp. 15890\u201315902 (2024)"},{"key":"24_CR12","doi-asserted-by":"crossref","unstructured":"Guo, J., et al.: Zero-shot generative model adaptation via image-specific prompt learning. In: CVPR, pp. 11494\u201311503 (2023)","DOI":"10.1109\/CVPR52729.2023.01106"},{"key":"24_CR13","doi-asserted-by":"crossref","unstructured":"Han, L., Li, Y., Zhang, H., Milanfar, P., Metaxas, D., Yang, F.: Svdiff: compact parameter space for diffusion fine-tuning. In: ICCV, pp. 7323\u20137334 (2023)","DOI":"10.1109\/ICCV51070.2023.00673"},{"key":"24_CR14","unstructured":"Hao, S., Han, K., Zhao, S., Wong, K.Y.K.: Vico: Plug-and-play visual condition for personalized text-to-image generation. arXiv preprint arXiv:2306.00971 (2023)"},{"key":"24_CR15","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., Cohen-Or, D.: Prompt-to-prompt image editing with cross attention control. In: ICLR (2023)"},{"key":"24_CR16","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: NeurIPS, pp. 6840\u20136851 (2020)"},{"key":"24_CR17","doi-asserted-by":"crossref","unstructured":"Huang, Z., Chan, K.C., Jiang, Y., Liu, Z.: Collaborative diffusion for multi-modal face generation and editing. In: CVPR, pp. 6080\u20136090 (2023)","DOI":"10.1109\/CVPR52729.2023.00589"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., Xie, W.: Prompting visual-language models for efficient video understanding. In: ECCV, pp. 105\u2013124 (2022)","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"24_CR19","doi-asserted-by":"crossref","unstructured":"Kumari, N., Zhang, B., Zhang, R., Shechtman, E., Zhu, J.Y.: Multi-concept customization of text-to-image diffusion. In: CVPR, pp. 1931\u20131941 (2023)","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., Constant, N.: The power of scale for parameter-efficient prompt tuning. In: EMNLP, pp. 3045\u20133059 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"24_CR21","unstructured":"Li, B., Qi, X., Lukasiewicz, T., Torr, P.: Controllable text-to-image generation. In: NeurIPS, pp. 2065\u20132075 (2019)"},{"key":"24_CR22","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: ICML, pp. 12888\u201312900 (2022)"},{"key":"24_CR23","doi-asserted-by":"crossref","unstructured":"Li, L.H., et\u00a0al.: Grounded language-image pre-training. In: CVPR, pp. 10965\u201310975 (2022)","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"24_CR24","unstructured":"Li, X.L., Liang, P.: Prefix-tuning: optimizing continuous prompts for generation. In: ACL, pp. 4582\u20134597 (2021)"},{"key":"24_CR25","doi-asserted-by":"crossref","unstructured":"Liu, P., Yuan, W., Fu, J., Jiang, Z., Hayashi, H., Neubig, G.: Pre-train, prompt, and predict: a systematic survey of prompting methods in natural language processing. ACM Comput. Surv. 55(9), 1\u201335 (2023)","DOI":"10.1145\/3560815"},{"key":"24_CR26","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: Hierarchical prompt learning for multi-task learning. In: CVPR, pp. 10888\u201310898 (2023)","DOI":"10.1109\/CVPR52729.2023.01048"},{"key":"24_CR27","unstructured":"Liu, Z., et al.: Cones: Concept neurons in diffusion models for customized generation. arXiv preprint arXiv:2303.05125 (2023)"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"Lu, Y., Liu, J., Zhang, Y., Liu, Y., Tian, X.: Prompt distribution learning. In: CVPR, pp. 5206\u20135215 (2022)","DOI":"10.1109\/CVPR52688.2022.00514"},{"key":"24_CR29","unstructured":"Nichol, A., et al.: Glide: towards photorealistic image generation and editing with text-guided diffusion models. In: ICML, pp. 16784\u201316804 (2022)"},{"key":"24_CR30","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML, pp. 8748\u20138763 (2021)"},{"key":"24_CR31","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)"},{"key":"24_CR32","unstructured":"Ramesh, A., et al.: Zero-shot text-to-image generation. In: ICML, pp. 8821\u20138831 (2021)"},{"key":"24_CR33","doi-asserted-by":"crossref","unstructured":"Rao, Y., et al.: DenseCLIP: language-guided dense prediction with context-aware prompting. In: CVPR, pp. 18082\u201318091 (2022)","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"24_CR34","unstructured":"Reed, S., Akata, Z., Yan, X., Logeswaran, L., Schiele, B., Lee, H.: Generative adversarial text to image synthesis. In: ICML, pp. 1060\u20131069 (2016)"},{"key":"24_CR35","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"24_CR36","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: MICCAI, pp. 234\u2013241 (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"24_CR37","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: Dreambooth: fine tuning text-to-image diffusion models for subject-driven generation. In: CVPR, pp. 22500\u201322510 (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"24_CR38","unstructured":"Saharia, C., et\u00a0al.: Photorealistic text-to-image diffusion models with deep language understanding. In: NeurIPS, pp. 36479\u201336494 (2022)"},{"key":"24_CR39","doi-asserted-by":"crossref","unstructured":"Shin, T., Razeghi, Y., Logan\u00a0IV, R.L., Wallace, E., Singh, S.: Autoprompt: eliciting knowledge from language models with automatically generated prompts. In: EMNLP, pp. 4222\u20134235 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.346"},{"key":"24_CR40","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. In: ICML, pp. 2256\u20132265 (2015)"},{"key":"24_CR41","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: ICLR (2021)"},{"key":"24_CR42","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS, pp. 5998\u20136008 (2017)"},{"key":"24_CR43","doi-asserted-by":"crossref","unstructured":"Wang, Z., et al.: Learning to prompt for continual learning. In: CVPR, pp. 139\u2013149 (2022)","DOI":"10.1109\/CVPR52688.2022.00024"},{"key":"24_CR44","doi-asserted-by":"crossref","unstructured":"Wei, Y., Zhang, Y., Ji, Z., Bai, J., Zhang, L., Zuo, W.: Elite: encoding visual concepts into textual embeddings for customized text-to-image generation. In: ICCV, pp. 15943\u201315953 (2023)","DOI":"10.1109\/ICCV51070.2023.01461"},{"key":"24_CR45","doi-asserted-by":"crossref","unstructured":"Wu, C., et al.: N\u00fcwa: visual synthesis pre-training for neural visual world creation. In: ECCV, pp. 720\u2013736 (2022)","DOI":"10.1007\/978-3-031-19787-1_41"},{"key":"24_CR46","doi-asserted-by":"crossref","unstructured":"Xiao, G., Yin, T., Freeman, W.T., Durand, F., Han, S.: Fastcomposer: Tuning-free multi-subject image generation with localized attention. arXiv preprint arXiv:2305.10431 (2023)","DOI":"10.1007\/s11263-024-02227-z"},{"key":"24_CR47","doi-asserted-by":"crossref","unstructured":"Xu, T., et al.: Attngan: Fine-grained text to image generation with attentional generative adversarial networks. In: CVPR, pp. 1316\u20131324 (2018)","DOI":"10.1109\/CVPR.2018.00143"},{"key":"24_CR48","doi-asserted-by":"crossref","unstructured":"Xue, H., Huang, Z., Sun, Q., Song, L., Zhang, W.: Freestyle layout-to-image synthesis. In: CVPR, pp. 14256\u201314266 (2023)","DOI":"10.1109\/CVPR52729.2023.01370"},{"key":"24_CR49","doi-asserted-by":"publisher","unstructured":"Yan, Z., Li, X., Wang, K., Zhang, Z., Li, J., Yang, J.: Multi-modal masked pre-training for monocular panoramic depth completion. In: ECCV, pp. 378\u2013395. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19769-7_22","DOI":"10.1007\/978-3-031-19769-7_22"},{"key":"24_CR50","doi-asserted-by":"crossref","unstructured":"Yan, Z., .: Tri-perspective view decomposition for geometry-aware depth completion. In: CVPR, pp. 4874\u20134884 (2024)","DOI":"10.1109\/CVPR52733.2024.00466"},{"key":"24_CR51","doi-asserted-by":"publisher","unstructured":"Yan, Z., Wang, K., Li, X., Zhang, Z., Li, J., Yang, J.: Rignet: repetitive image guided network for depth completion. In: ECCV, pp. 214\u2013230. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19812-0_13","DOI":"10.1007\/978-3-031-19812-0_13"},{"key":"24_CR52","doi-asserted-by":"crossref","unstructured":"Yang, Z., et\u00a0al.: Reco: region-controlled text-to-image generation. In: CVPR, pp. 14246\u201314255 (2023)","DOI":"10.1109\/CVPR52729.2023.01369"},{"key":"24_CR53","doi-asserted-by":"crossref","unstructured":"Yao, H., Zhang, R., Xu, C.: Visual-language prompt tuning with knowledge-guided context optimization. In: CVPR, pp. 6757\u20136767 (2023)","DOI":"10.1109\/CVPR52729.2023.00653"},{"key":"24_CR54","unstructured":"Yu, J., et\u00a0al.: Scaling autoregressive models for content-rich text-to-image generation. TMLR (2022)"},{"key":"24_CR55","doi-asserted-by":"crossref","unstructured":"Zhang, H., et al.: Stackgan: text to photo-realistic image synthesis with stacked generative adversarial networks. In: ICCV, pp. 5907\u20135915 (2017)","DOI":"10.1109\/ICCV.2017.629"},{"key":"24_CR56","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Han, L., Ghosh, A., Metaxas, D.N., Ren, J.: Sine: single image editing with text-to-image diffusion models. In: CVPR, pp. 6027\u20136037 (2023)","DOI":"10.1109\/CVPR52729.2023.00584"},{"key":"24_CR57","doi-asserted-by":"crossref","unstructured":"Zhao, J., Zheng, H., Wang, C., Lan, L., Yang, W.: Magicfusion: boosting text-to-image generation performance by fusing diffusion models. In: ICCV, pp. 22592\u201322602 (2023)","DOI":"10.1109\/ICCV51070.2023.02065"},{"key":"24_CR58","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Friedman, D., Chen, D.: Factual probing is [mask]: Learning vs. learning to recall. In: NAACL, pp. 5017\u20135033 (2021)","DOI":"10.18653\/v1\/2021.naacl-main.398"},{"key":"24_CR59","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Conditional prompt learning for vision-language models. In: CVPR, pp. 16816\u201316825 (2022)","DOI":"10.1109\/CVPR52688.2022.01631"},{"issue":"9","key":"24_CR60","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. IJCV 130(9), 2337\u20132348 (2022)","journal-title":"IJCV"},{"key":"24_CR61","unstructured":"Zhou, Y., Zhang, R., Sun, T., Xu, J.: Enhancing detail preservation for customized text-to-image generation: A regularization-free approach. arXiv preprint arXiv:2305.13579 (2023)"},{"key":"24_CR62","doi-asserted-by":"crossref","unstructured":"Zhu, B., Niu, Y., Han, Y., Wu, Y., Zhang, H.: Prompt-aligned gradient for prompt tuning. arXiv preprint arXiv:2205.14865 (2022)","DOI":"10.1109\/ICCV51070.2023.01435"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72973-7_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,15]],"date-time":"2025-02-15T14:59:26Z","timestamp":1739631566000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72973-7_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031729720","9783031729737"],"references-count":62,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72973-7_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}