{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T22:51:46Z","timestamp":1776293506083,"version":"3.50.1"},"reference-count":25,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T00:00:00Z","timestamp":1776211200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T00:00:00Z","timestamp":1776211200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100020595","name":"National Science and Technology Council","doi-asserted-by":"publisher","award":["NSTC 113-2221-E-197-023"],"award-info":[{"award-number":["NSTC 113-2221-E-197-023"]}],"id":[{"id":"10.13039\/100020595","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-026-21623-w","type":"journal-article","created":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T21:57:05Z","timestamp":1776290225000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A latent space-based image inpainting approach for the stable diffusion pipeline: enhancing global style consistency"],"prefix":"10.1007","volume":"85","author":[{"given":"Wei-Cheng","family":"Lai","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0782-0891","authenticated-orcid":false,"given":"Fay","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,15]]},"reference":[{"key":"21623_CR1","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y: Generative adversarial nets. In: Advances in Neural Information Processing Systems, Montreal, QC, Canada, pp. 2672\u20132680 (2014). arXiv:1406.2661"},{"key":"21623_CR2","unstructured":"Kingma D.P, Welling M: Auto-encoding variational bayes. In: International Conference on Learning Representations (ICLR) (2014). arXiv:1312.6114"},{"key":"21623_CR3","unstructured":"Rezende D, Mohamed S: Variational inference with normalizing flows. In: Proceedings of the 32nd International Conference on Machine Learning, Lille, France, pp. 1530\u20131538 (2015). arXiv:1505.05770"},{"key":"21623_CR4","unstructured":"Ho J, Jain A, Abbeel P: Denoising diffusion probabilistic models. In: Advances in Neural Information Processing Systems, vol. 33, pp. 6840\u20136851 (2020)"},{"key":"21623_CR5","doi-asserted-by":"crossref","unstructured":"Rombach R, Blattmann A, Lorenz D, Esser P, Ommer B: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"21623_CR6","unstructured":"Song J, Meng C, Ermon S: Denoising diffusion implicit models. In: International Conference on Learning Representations (ICLR) (2021). arXiv:2010.02502"},{"key":"21623_CR7","doi-asserted-by":"publisher","unstructured":"Ronneberger O, Fischer P, Brox T: U-net: Convolutional networks for biomedical image segmentation. In: Navab N, Hornegger J, Wells W.M, Frangi A.F. (eds.) Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015, pp. 234\u2013241. Springer, (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"21623_CR8","unstructured":"Podell D, English Z, Lacey K, Blattmann A, Dockhorn T, M\u00fcller J, Penna J, Rombach R: Sdxl: Improving latent diffusion models for high-resolution image synthesis. In: International Conference on Learning Representations (ICLR) (2024). arXiv:2307.01952"},{"key":"21623_CR9","doi-asserted-by":"crossref","unstructured":"He K, Chen X, Xie S, Li Y, Doll\u00e1r P, Girshick R: Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"21623_CR10","unstructured":"Couairon G, Verbeek J, Schwenk H, Cord M: Diffedit: Diffusion-based semantic image editing with mask guidance. In: International Conference on Learning Representations (ICLR) (2023). arXiv:2210.11427"},{"key":"21623_CR11","doi-asserted-by":"crossref","unstructured":"Zhang L, Rao A, Agrawala M: Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2023). arXiv:2302.05543","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"21623_CR12","unstructured":"Radford A, Kim J.W, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J, Krueger G, Sutskever I: Learning transferable visual models from natural language supervision. In: Proceedings of the 38th International Conference on Machine Learning (ICML), pp. 8748\u20138763 (2021)"},{"issue":"1","key":"21623_CR13","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1080\/10867651.2004.10487596","volume":"9","author":"A Telea","year":"2004","unstructured":"Telea A (2004) An image inpainting technique based on the fast marching method. Journal of Graphics Tools 9(1):23\u201334","journal-title":"Journal of Graphics Tools"},{"issue":"9","key":"21623_CR14","doi-asserted-by":"publisher","first-page":"1200","DOI":"10.1109\/TIP.2004.833105","volume":"13","author":"A Criminisi","year":"2004","unstructured":"Criminisi A, Perez P, Toyama K (2004) Region filling and object removal by exemplar-based image inpainting. IEEE Trans Image Process 13(9):1200\u20131212. https:\/\/doi.org\/10.1109\/TIP.2004.833105","journal-title":"IEEE Trans Image Process"},{"issue":"4","key":"21623_CR15","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1145\/3072959.3073659","volume":"36","author":"S Iizuka","year":"2017","unstructured":"Iizuka S, Simo-Serra E, Ishikawa H (2017) Globally and locally consistent image completion. ACM Transactions on Graphics 36(4):107\u2013110714. https:\/\/doi.org\/10.1145\/3072959.3073659","journal-title":"ACM Transactions on Graphics"},{"key":"21623_CR16","doi-asserted-by":"crossref","unstructured":"Yu J, Lin Z, Yang J, Shen X, Lu X, Huang T.S: Generative image inpainting with contextual attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5505\u20135514 (2018)","DOI":"10.1109\/CVPR.2018.00577"},{"key":"21623_CR17","doi-asserted-by":"crossref","unstructured":"Lugmayr A, Danelljan M, Van Gool L, Timofte R: Repaint: Inpainting using denoising diffusion probabilistic models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 11461\u201311471 (2022)","DOI":"10.1109\/CVPR52688.2022.01117"},{"key":"21623_CR18","unstructured":"Avrahami O, Lischinski D, Fried O: Paint by example: Exemplar-based image editing with diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4238\u20134247 (2023)"},{"key":"21623_CR19","unstructured":"Saharia C, Chan W, Saxena S, Li L, Whang J, Denton E, Ghasemipour S.K, Ayan B.K, Mahdavi A, Norouzi M: Palette: Image-to-image diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 16861\u201316872. IEEE, (2022). arXiv:2111.05826"},{"key":"21623_CR20","doi-asserted-by":"crossref","unstructured":"Feng K, Ma Y, Wang B, Qi C, Chen H, Chen Q, Wang Z: Dit4edit: Diffusion transformer for image editing. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), pp. 2969\u20132977 (2025). arXiv:2411.03286","DOI":"10.1609\/aaai.v39i3.32304"},{"key":"21623_CR21","doi-asserted-by":"crossref","unstructured":"Nitzan Y, Wu Z, Zhang R, Shechtman E, Cohen-Or D, Park T, Gharbi M: Lazy diffusion transformer for interactive image editing. In: European Conference on Computer Vision (ECCV), pp. 55\u201372 (2024). arXiv:2404.12382","DOI":"10.1007\/978-3-031-72691-0_4"},{"key":"21623_CR22","doi-asserted-by":"crossref","unstructured":"Liu H, Wang Y, Qian B, Wang M, Rui Y: Structure matters: Tackling the semantic discrepancy in diffusion models for image inpainting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8038\u20138047 (2024). arXiv:2403.19898","DOI":"10.1109\/CVPR52733.2024.00768"},{"key":"21623_CR23","doi-asserted-by":"crossref","unstructured":"Huang J, Liu T, Wu Y, Qu X, Liu L, Hu X: Mtadiffusion: Mask text alignment diffusion model for object inpainting. arXiv preprint arXiv:2506.23482 (2025)","DOI":"10.1109\/CVPR52734.2025.01708"},{"key":"21623_CR24","doi-asserted-by":"crossref","unstructured":"Ju X, Liu X, Wang X, Bian Y, Shan Y, Xu Q: Brushnet: A plug-and-play image inpainting model with decomposed dual-branch diffusion. In: European Conference on Computer Vision (ECCV), pp. 150\u2013168 (2024). arXiv:2403.06976","DOI":"10.1007\/978-3-031-72661-3_9"},{"key":"21623_CR25","doi-asserted-by":"crossref","unstructured":"Zhuang J, Zeng Y, Liu W, Yuan C, Chen K: A task is worth one word: Learning with task prompts for high-quality versatile image inpainting. In: European Conference on Computer Vision (ECCV), pp. 195\u2013211 (2024). arXiv:2312.03594","DOI":"10.1007\/978-3-031-73636-0_12"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21623-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-026-21623-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-026-21623-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T21:57:09Z","timestamp":1776290229000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-026-21623-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,15]]},"references-count":25,"journal-issue":{"issue":"4","published-online":{"date-parts":[[2026,4]]}},"alternative-id":["21623"],"URL":"https:\/\/doi.org\/10.1007\/s11042-026-21623-w","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,15]]},"assertion":[{"value":"9 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 February 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 April 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 April 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"394"}}