{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:03:59Z","timestamp":1784358239145,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":29,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819235124","type":"print"},{"value":"9789819235131","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3513-1_32","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T06:41:28Z","timestamp":1784356888000},"page":"383-394","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["VLGDiff: A Vision-Language Guided Diffusion Model for Image Inpainting"],"prefix":"10.1007","author":[{"given":"Shuling","family":"Zheng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3167-2718","authenticated-orcid":false,"given":"Zhan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhanglu","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinglue","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiheng","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"32_CR1","doi-asserted-by":"crossref","unstructured":"Zhuang, J., Zeng, Y., Liu, W., Yuan, C., Chen, K.: A task is worth one word: Learning with task prompts for high-quality versatile image inpainting. In: ECCV, pp. 195\u2013211 (2024)","DOI":"10.1007\/978-3-031-73636-0_12"},{"key":"32_CR2","doi-asserted-by":"crossref","unstructured":"Wan, Z., et al.: Bringing old photos back to life. In: CVPR, pp. 2747\u20132757 (2020)","DOI":"10.1109\/CVPR42600.2020.00282"},{"key":"32_CR3","doi-asserted-by":"publisher","DOI":"10.1145\/3422622","volume-title":"Generative adversarial networks","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow, I., et al.: Generative adversarial networks. ACM, Commun (2020)"},{"key":"32_CR4","unstructured":"Dhariwal, P., Nichol, A.Q.: Diffusion models beat GANs on image synthesis. In: NeurIPS, pp. 8780\u20138794 (2021)"},{"issue":"4","key":"32_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073659","volume":"36","author":"S Iizuka","year":"2017","unstructured":"Iizuka, S., Simo-Serra, E., Ishikawa, H.: Globally and locally consistent image completion. ACM Trans. Graph. 36(4), 1\u201314 (2017)","journal-title":"ACM Trans. Graph."},{"key":"32_CR6","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"32_CR7","unstructured":"Lykon: Dreamshaper-8 inpainting (2023). https:\/\/civitai.com\/models\/4384?modelVersionId=131004"},{"key":"32_CR8","unstructured":"Manukyan, H., Sargsyan, A., Atanyan, B., Zhangyang, W., Navasardyan, S., Shi, H.: HD-Painter: high-resolution and prompt-faithful text-guided image inpainting with diffusion models. In: ICLR (2025)"},{"key":"32_CR9","doi-asserted-by":"crossref","unstructured":"Ju, X., Liu, X., Wang, X., Bian, Y., Shan, Y., Xu, Q.: Brushnet: a plug-and-play image inpainting model with decomposed dual-branch diffusion. In: ECCV, pp. 150\u2013168 (2024)","DOI":"10.1007\/978-3-031-72661-3_9"},{"key":"32_CR10","doi-asserted-by":"crossref","unstructured":"Gong, C., Li, D., Pan, Y., Chen, J., Yao, T., Mei, T.: FreeInpaint: tuning-free prompt alignment and visual rationality enhancement in image inpainting. In: AAAI, pp. 4239\u20134247 (2026)","DOI":"10.1609\/aaai.v40i6.42420"},{"key":"32_CR11","doi-asserted-by":"crossref","unstructured":"Kim, S., Suh, S., Lee, M.: RAD: region-aware diffusion models for image inpainting. In: CVPR, pp. 2439\u20132448 (2025)","DOI":"10.1109\/CVPR52734.2025.00233"},{"key":"32_CR12","doi-asserted-by":"crossref","unstructured":"Liu, H., Wang, Y., Qian, B., Wang, M., Rui, Y.: Structure matters: tackling the semantic discrepancy in diffusion models for image inpainting. In: CVPR, pp. 8038\u20138047 (2024)","DOI":"10.1109\/CVPR52733.2024.00768"},{"key":"32_CR13","doi-asserted-by":"crossref","unstructured":"Liu, K.-H., Yang, C.-K., Chen, M.-H., Liu, Y.-L., Lin, Y.-Y.: CorrFill: enhancing faithfulness in reference-based inpainting with correspondence guidance in diffusion models. In: WACV, pp. 1618\u20131627 (2025)","DOI":"10.1109\/WACV61041.2025.00165"},{"key":"32_CR14","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: ICML, pp. 8748\u20138763. PMLR (2021)"},{"key":"32_CR15","doi-asserted-by":"crossref","unstructured":"Zhang, T., et al.: CAS-ViT: convolutional additive self-attention vision transformers for efficient mobile applications. IEEE Trans. Image Process. (2026)","DOI":"10.1109\/TIP.2026.3655121"},{"key":"32_CR16","unstructured":"Ho, J., Salimans, T.: Classifier-free diffusion guidance. In: NeurIPS Workshop (2021)"},{"key":"32_CR17","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS, pp. 6000\u20136010 (2017)"},{"issue":"1\u20133","key":"32_CR18","first-page":"3","volume":"125","author":"S Xie","year":"2015","unstructured":"Xie, S., Tu, Z.: Holistically-nested edge detection. Int. J. Comput. Vision 125(1\u20133), 3\u201318 (2015)","journal-title":"Int. J. Comput. Vision"},{"key":"32_CR19","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. IEEE Trans. Pattern Anal. Mach. Intell. PP(99), 2999\u20133007 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"32_CR20","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., et al.: Microsoft coco: common objects in context. In: ECCV, pp. 740\u2013755. Springer (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"32_CR21","doi-asserted-by":"crossref","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. In: ICCV, pp. 2641\u20132649 (2015)","DOI":"10.1109\/ICCV.2015.303"},{"key":"32_CR22","doi-asserted-by":"crossref","unstructured":"Wang, S., et al.: Imagen editor and editbench: advancing and evaluating text-guided image inpainting. In: CVPR, pp. 18359\u201318369 (2023)","DOI":"10.1109\/CVPR52729.2023.01761"},{"key":"32_CR23","doi-asserted-by":"crossref","unstructured":"Lee, C.-H., Liu, Z., Wu, L., Luo, P.: MaskGAN: towards diverse and interactive facial image manipulation. In: CVPR, pp. 5549\u20135558 (2020)","DOI":"10.1109\/CVPR42600.2020.00559"},{"key":"32_CR24","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., Wang, O.: The unreasonable effectiveness of deep features as a perceptual metric. In: CVPR, pp. 586\u2013595 (2018)","DOI":"10.1109\/CVPR.2018.00068"},{"key":"32_CR25","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: GANs trained by a two time-scale update rule converge to a local Nash equilibrium. In: NeurIPS, pp. 6626\u20136637 (2017)"},{"key":"32_CR26","doi-asserted-by":"crossref","unstructured":"Kirstain, Y., Polyak, A., Singer, U., Matiana, S., Penna, J., Levy, O.: Pick-a-pic: an open dataset of user preferences for text-to-image generation. In: NeurIPS, pp. 36652\u201336663 (2023)","DOI":"10.52202\/075280-1594"},{"key":"32_CR27","doi-asserted-by":"crossref","unstructured":"Hessel, J., Holtzman, A., Forbes, M., Le Bras, R., Choi, Y.: CLIPScore: a reference-free evaluation metric for image captioning. In: EMNLP, pp. 7514\u20137528 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"32_CR28","unstructured":"Chen, K., Wang, J., Pang, J., et al.: MMDetection: open MMLab detection toolbox and benchmark (2019). arXiv preprint arXiv:1906.07155"},{"issue":"3","key":"32_CR29","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1006\/gmip.1995.1022","volume":"57","author":"H Li","year":"1995","unstructured":"Li, H., Manjunath, B.S., Mitra, S.K.: Multisensor image fusion using the wavelet transform. Graph. Mod. Image Process. 57(3), 235\u2013245 (1995)","journal-title":"Graph. Mod. Image Process."}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3513-1_32","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T06:41:32Z","timestamp":1784356892000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3513-1_32"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819235124","9789819235131"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3513-1_32","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}