{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T08:05:36Z","timestamp":1782893136722,"version":"3.54.5"},"reference-count":70,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.cviu.2026.104807","type":"journal-article","created":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T11:46:08Z","timestamp":1778759168000},"page":"104807","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["DiffuseFit: Shape-guided warping and limb-aware diffusion synthesis for occlusion-resilient and semantically consistent virtual try-on"],"prefix":"10.1016","volume":"269","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9352-2921","authenticated-orcid":false,"given":"Tareq Mahmod","family":"AlZubi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Umar Raza","family":"Mukhtar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9986-1123","authenticated-orcid":false,"given":"Omar A.","family":"Alzubi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0750-3711","authenticated-orcid":false,"given":"Hamza","family":"Mukhtar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104807_b1","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","article-title":"Blended diffusion for text-driven editing of natural images","author":"Avrahami","year":"2022"},{"key":"10.1016\/j.cviu.2026.104807_b2","series-title":"European Conference on Computer Vision","first-page":"409","article-title":"Single stage virtual try-on via deformable attention flows","author":"Bai","year":"2022"},{"key":"10.1016\/j.cviu.2026.104807_b3","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","article-title":"Multimodal garment designer: Human-centric latent diffusion models for fashion image editing","author":"Baldrati","year":"2023"},{"issue":"3","key":"10.1016\/j.cviu.2026.104807_b4","first-page":"8","article-title":"Improving image generation with better captions","volume":"2","author":"Betker","year":"2023","journal-title":"Comput. Sci."},{"key":"10.1016\/j.cviu.2026.104807_b5","series-title":"An image equals 16x16 words: Scaling image recognition with transformers","author":"Beyer","year":"2025"},{"issue":"1","key":"10.1016\/j.cviu.2026.104807_b6","doi-asserted-by":"crossref","first-page":"563","DOI":"10.1007\/s00371-024-03347-w","article-title":"CS-VITON: A realistic virtual try-on network based on clothing region alignment and SPM","volume":"41","author":"Chen","year":"2025","journal-title":"Vis. Comput."},{"issue":"1","key":"10.1016\/j.cviu.2026.104807_b7","doi-asserted-by":"crossref","first-page":"302","DOI":"10.1109\/TCSVT.2021.3059706","article-title":"PMAN: Progressive multi-attention network for human pose transfer","volume":"32","author":"Chen","year":"2021","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104807_b8","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"14131","article-title":"VITON-HD: High-resolution virtual try-on via misalignment-aware normalization","author":"Choi","year":"2021"},{"key":"10.1016\/j.cviu.2026.104807_b9","series-title":"Catvton: Concatenation is all you need for virtual try-on with diffusion models","author":"Chong","year":"2024"},{"key":"10.1016\/j.cviu.2026.104807_b10","series-title":"Enhancing early diabetic retinopathy detection through synthetic DR1 image generation: A StyleGAN3 approach","author":"Das","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b11","series-title":"Averaged adam accelerates stochastic optimization in the training of deep neural network approximations for partial differential equation and optimal control problems","author":"Dereich","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b12","doi-asserted-by":"crossref","DOI":"10.1109\/TMM.2024.3354622","article-title":"PG-VTON: A novel image-based virtual try-on method via progressive inference paradigm","author":"Fang","year":"2024","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.cviu.2026.104807_b13","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"7599","article-title":"Taming the power of diffusion models for high-quality virtual try-on with appearance flow","author":"Gou","year":"2023"},{"key":"10.1016\/j.cviu.2026.104807_b14","doi-asserted-by":"crossref","DOI":"10.1016\/j.image.2025.117283","article-title":"DA-net: Deep attention network for biomedical image segmentation","volume":"135","author":"Gu","year":"2025","journal-title":"Signal Process., Image Commun."},{"key":"10.1016\/j.cviu.2026.104807_b15","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7297","article-title":"DensePose: Dense human pose estimation in the wild","author":"G\u00fcler","year":"2018"},{"key":"10.1016\/j.cviu.2026.104807_b16","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"7543","article-title":"VITON: An image-based virtual try-on network","author":"Han","year":"2018"},{"key":"10.1016\/j.cviu.2026.104807_b17","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia","first-page":"2593","article-title":"Shape-guided clothing warping for virtual try-on","author":"Han","year":"2024"},{"key":"10.1016\/j.cviu.2026.104807_b18","series-title":"Proceedings of the 30th ACM International Conference on Multimedia","first-page":"2420","article-title":"Progressive limb-aware virtual try-on","author":"Han","year":"2022"},{"key":"10.1016\/j.cviu.2026.104807_b19","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"16000","article-title":"Masked autoencoders are scalable vision learners","author":"He","year":"2022"},{"key":"10.1016\/j.cviu.2026.104807_b20","series-title":"RETI-diff: Illumination degradation image restoration with retinex-based latent diffusion model","author":"He","year":"2023"},{"key":"10.1016\/j.cviu.2026.104807_b21","series-title":"Diffusion models in low-level vision: A survey","author":"He","year":"2024"},{"key":"10.1016\/j.cviu.2026.104807_b22","series-title":"Advances in Neural Information Processing Systems","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"vol. 33","author":"Ho","year":"2020"},{"key":"10.1016\/j.cviu.2026.104807_b23","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2024.112233","article-title":"A Markov chain approach for video-based virtual try-on with denoising diffusion GAN","volume":"300","author":"Hou","year":"2024","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.cviu.2026.104807_b24","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"1501","article-title":"Arbitrary style transfer in real-time with adaptive instance normalization","author":"Huang","year":"2017"},{"key":"10.1016\/j.cviu.2026.104807_b25","doi-asserted-by":"crossref","first-page":"23593","DOI":"10.52202\/068431-1714","article-title":"Denoising diffusion restoration models","volume":"35","author":"Kawar","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104807_b26","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"8176","article-title":"StableVITON: Learning semantic correspondence with latent diffusion model for virtual try-on","author":"Kim","year":"2024"},{"key":"10.1016\/j.cviu.2026.104807_b27","unstructured":"Kingma,\u00a0Diederik\u00a0P., Welling,\u00a0Max, 2014. Auto-encoding variational Bayes. In: Proceedings of the International Conference on Learning Representations."},{"key":"10.1016\/j.cviu.2026.104807_b28","series-title":"2022 8th International Conference on Optimization and Applications","first-page":"1","article-title":"Mode collapse in generative adversarial networks: An overview","author":"Kossale","year":"2022"},{"key":"10.1016\/j.cviu.2026.104807_b29","series-title":"Deep learning of thermodynamic laws from microscopic dynamics","author":"Kuroyanagi","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b30","series-title":"European Conference on Computer Vision","first-page":"204","article-title":"High-resolution virtual try-on with misalignment and occlusion-handled conditions","author":"Lee","year":"2022"},{"key":"10.1016\/j.cviu.2026.104807_b31","doi-asserted-by":"crossref","first-page":"47","DOI":"10.1016\/j.neucom.2022.01.029","article-title":"SRDiff: Single image super-resolution with diffusion probabilistic models","volume":"479","author":"Li","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cviu.2026.104807_b32","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15546","article-title":"Toward accurate and realistic outfits visualization with attention to details","author":"Li","year":"2021"},{"key":"10.1016\/j.cviu.2026.104807_b33","series-title":"Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence","article-title":"RMGN: A regional mask guided network for parser-free virtual try-on","author":"Lin","year":"2022"},{"key":"10.1016\/j.cviu.2026.104807_b34","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"2118","article-title":"Toward realistic virtual try-on through landmark guided shape matching","volume":"vol. 35","author":"Liu","year":"2021"},{"key":"10.1016\/j.cviu.2026.104807_b35","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.123213","article-title":"A progressive distillation network for practical image-based virtual try-on","volume":"246","author":"Luo","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.cviu.2026.104807_b36","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"8580","article-title":"LaDI-VTON: Latent diffusion textual-inversion enhanced virtual try-on","author":"Morelli","year":"2023"},{"key":"10.1016\/j.cviu.2026.104807_b37","doi-asserted-by":"crossref","unstructured":"Morelli,\u00a0Davide, Fincato,\u00a0Matteo, Cornia,\u00a0Marcella, Landi,\u00a0Federico, Cesari,\u00a0Fabio, Cucchiara,\u00a0Rita, 2022. Dress code: High-resolution multi-category virtual try-on. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2231\u20132235.","DOI":"10.1007\/978-3-031-20074-8_20"},{"key":"10.1016\/j.cviu.2026.104807_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2025.107546","article-title":"Wp-VTON: A wrinkle-preserving virtual try-on network via clothing texture book","volume":"189","author":"Mu","year":"2025","journal-title":"Neural Netw."},{"key":"10.1016\/j.cviu.2026.104807_b39","doi-asserted-by":"crossref","first-page":"363","DOI":"10.1016\/j.neunet.2023.09.047","article-title":"STMMOT: Advancing multi-object tracking through spatiotemporal memory networks and multi-scale attention pyramids","volume":"168","author":"Mukhtar","year":"2023","journal-title":"Neural Netw."},{"key":"10.1016\/j.cviu.2026.104807_b40","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8138","article-title":"Aipparel: A multimodal foundation model for digital garments","author":"Nakayama","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b41","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2022.103741","article-title":"VICTOR: Visual incompatibility detection with transformers and fashion-specific contrastive pretraining","volume":"90","author":"Papadopoulos","year":"2023","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.cviu.2026.104807_b42","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"808","article-title":"Diffusion-based image translation with label guidance for domain adaptive semantic segmentation","author":"Peng","year":"2023"},{"issue":"5","key":"10.1016\/j.cviu.2026.104807_b43","doi-asserted-by":"crossref","first-page":"476","DOI":"10.3390\/bioengineering12050476","article-title":"Modelling the Ki67 index in synthetic HE-stained images using conditional StyleGAN model","volume":"12","author":"Piatrikov\u00e1","year":"2025","journal-title":"Bioengineering"},{"key":"10.1016\/j.cviu.2026.104807_b44","series-title":"SDXL: Improving latent diffusion models for high-resolution image synthesis","author":"Podell","year":"2023"},{"key":"10.1016\/j.cviu.2026.104807_b45","series-title":"Proceedings of the International Conference on Machine Learning","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.cviu.2026.104807_b46","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"181","article-title":"StructureFlow: Image inpainting via structure-aware appearance flow","author":"Ren","year":"2019"},{"key":"10.1016\/j.cviu.2026.104807_b47","doi-asserted-by":"crossref","first-page":"8622","DOI":"10.1109\/TIP.2020.3018224","article-title":"Deep spatial transformation for pose-guided person image generation and animation","volume":"29","author":"Ren","year":"2020","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104807_b48","series-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention","first-page":"234","article-title":"U-net: Convolutional networks for biomedical image segmentation","author":"Ronneberger","year":"2015"},{"key":"10.1016\/j.cviu.2026.104807_b49","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","article-title":"Dreambooth: Fine-tuning text-to-image diffusion models for subject-driven generation","author":"Ruiz","year":"2023"},{"issue":"4","key":"10.1016\/j.cviu.2026.104807_b50","first-page":"4713","article-title":"Image super-resolution via iterative refinement","volume":"45","author":"Saharia","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104807_b51","doi-asserted-by":"crossref","first-page":"36479","DOI":"10.52202\/068431-2643","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104807_b52","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"7707","article-title":"Imagdressing-v1: Customizable virtual dressing","volume":"vol. 39","author":"Shen","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b53","series-title":"Long-term talkingface generation via motion-prior conditional diffusion model","author":"Shen","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b54","series-title":"Imaggarment-1: Fine-grained garment generation for controllable fashion design","author":"Shen","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b55","unstructured":"Simonyan,\u00a0Karen, Zisserman,\u00a0Andrew, 2015. Very deep convolutional networks for large-scale image recognition. In: International Conference on Learning Representations."},{"issue":"6","key":"10.1016\/j.cviu.2026.104807_b56","doi-asserted-by":"crossref","first-page":"4434","DOI":"10.1109\/TCSVT.2023.3338459","article-title":"Fashion customization: Image generation based on editing clue","volume":"34","author":"Song","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104807_b57","article-title":"Better fit: Accommodate variations in clothing types for virtual try-on","author":"Song","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104807_b58","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"15287","article-title":"Investigating the role of weight decay in enhancing nonconvex SGD","author":"Sun","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b59","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2023.103778","article-title":"TsrNet: A two-stage unsupervised approach for clothing region-specific textures style transfer","volume":"91","author":"Sun","year":"2023","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.cviu.2026.104807_b60","series-title":"European Conference on Computer Vision","first-page":"184","article-title":"Improving virtual try-on with garment-focused diffusion models","author":"Wan","year":"2024"},{"key":"10.1016\/j.cviu.2026.104807_b61","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"550","article-title":"4D-dress: A 4D dataset of real-world human clothing with semantic annotations","author":"Wang","year":"2024"},{"key":"10.1016\/j.cviu.2026.104807_b62","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"23550","article-title":"GP-VTON: Towards general purpose virtual try-on via collaborative local-flow global-parsing learning","author":"Xie","year":"2023"},{"key":"10.1016\/j.cviu.2026.104807_b63","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3723","article-title":"Deep flow-guided video inpainting","author":"Xu","year":"2019"},{"key":"10.1016\/j.cviu.2026.104807_b64","article-title":"OccluMix: Towards de-occlusion virtual try-on by semantically-guided mixup","author":"Yang","year":"2023","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.cviu.2026.104807_b65","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18381","article-title":"Paint by example: Exemplar-based image editing with diffusion models","author":"Yang","year":"2023"},{"key":"10.1016\/j.cviu.2026.104807_b66","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","article-title":"Towards photo-realistic virtual try-on by adaptively generating-preserving image content","author":"Yang","year":"2020"},{"key":"10.1016\/j.cviu.2026.104807_b67","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8372","article-title":"Cat-DM: Controllable accelerated virtual try-on with diffusion model","author":"Zeng","year":"2024"},{"key":"10.1016\/j.cviu.2026.104807_b68","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"26399","article-title":"BooW-VTON: Boosting in-the-wild virtual try-on via mask-free pseudo data training","author":"Zhang","year":"2025"},{"key":"10.1016\/j.cviu.2026.104807_b69","series-title":"European Conference on Computer Vision","first-page":"286","article-title":"View synthesis by appearance flow","author":"Zhou","year":"2016"},{"key":"10.1016\/j.cviu.2026.104807_b70","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","article-title":"TryOnDiffusion: A tale of two unets","author":"Zhu","year":"2023"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001748?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001748?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T06:11:01Z","timestamp":1782886261000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001748"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":70,"alternative-id":["S1077314226001748"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104807","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"DiffuseFit: Shape-guided warping and limb-aware diffusion synthesis for occlusion-resilient and semantically consistent virtual try-on","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104807","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104807"}}