{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T15:16:49Z","timestamp":1782314209014,"version":"3.54.5"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2025,9,5]],"date-time":"2025-09-05T00:00:00Z","timestamp":1757030400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,5]],"date-time":"2025-09-05T00:00:00Z","timestamp":1757030400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1007\/s11760-025-04388-x","type":"journal-article","created":{"date-parts":[[2025,9,5]],"date-time":"2025-09-05T15:35:00Z","timestamp":1757086500000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["CAF-VTON: cross-attention layered fusion based latent diffusion virtual try-On"],"prefix":"10.1007","volume":"19","author":[{"given":"Xiangyan","family":"Fu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingquan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weihua","family":"Pu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shengling","family":"Geng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin","family":"Gan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,5]]},"reference":[{"issue":"2","key":"4388_CR1","doi-asserted-by":"publisher","first-page":"1161","DOI":"10.1007\/s11760-023-02830-6","volume":"18","author":"A Aghamohammadi","year":"2024","unstructured":"Aghamohammadi, A.: A deep learning model for ergonomics risk assessment and sports and health monitoring in self-occluded images. SIViP 18(2), 1161\u20131173 (2024)","journal-title":"SIViP"},{"key":"4388_CR2","unstructured":"M.\u00a0Binkowski, D.\u00a0J. Sutherland, M.\u00a0Arbel, and A.\u00a0Gretton. Demystifying mmd gans. arXiv preprint arXiv:1801.01401, 2018"},{"key":"4388_CR3","doi-asserted-by":"crossref","unstructured":"C.\u00a0Y. Chen, Y.\u00a0C. Chen, H.\u00a0H. Shuai, and W.\u00a0H. Cheng. Size does matter:size-aware virtual try-on via clothing-oriented transformation try-on network. Proceedings of the IEEE\/CVF international conference on computer vision, pages 7513\u20137522, 2023","DOI":"10.1109\/ICCV51070.2023.00691"},{"issue":"1","key":"4388_CR4","doi-asserted-by":"publisher","first-page":"563","DOI":"10.1007\/s00371-024-03347-w","volume":"41","author":"J Chen","year":"2024","unstructured":"Chen, J., Zhang, X., Ma, L., Yang, B., Zhang, K.: Cs-viton: a realistic virtual try-on network based on clothing region alignment and spm. Vis. Comput. 41(1), 563\u2013577 (2024)","journal-title":"Vis. Comput."},{"key":"4388_CR5","doi-asserted-by":"crossref","unstructured":"S.\u00a0Choi, S.\u00a0Park, M.\u00a0Lee, and J.\u00a0Choo. Viton-hd: High-resolution virtual try-on via misalignment-aware normalization. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 14131\u201314140, 2021","DOI":"10.1109\/CVPR46437.2021.01391"},{"key":"4388_CR6","unstructured":"Zh. Chong, X.\u00a0Dong, H.\u00a0Li, Sh. Zhang, W.\u00a0Zhang, X.\u00a0Zhang, H.\u00a0Zhao, D.\u00a0Jiang, and X.\u00a0Liang. Catvton: Concatenation is all you need for virtual try-on with diffusion models. arXiv preprint arXiv:2407.15886, 2024"},{"key":"4388_CR7","doi-asserted-by":"crossref","unstructured":"Y.\u00a0Ge, Y.\u00a0Song, R.\u00a0Zhang, C.\u00a0Ge, W.\u00a0Liu, and P.\u00a0Luo. Parser-free virtual try-on via distilling appearance flows. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 8485\u20138493, 2021","DOI":"10.1109\/CVPR46437.2021.00838"},{"key":"4388_CR8","doi-asserted-by":"crossref","unstructured":"J.\u00a0Gou, S.\u00a0Sun, J.\u00a0Zhang, J.\u00a0Si, C.\u00a0Qian, and L.\u00a0Zhang. Taming the power of diffusion models for high-quality virtual try-on with appearance flow. Proceedings of the 31st ACM International Conference on Multimedia, pages 7599\u20137607, 2023","DOI":"10.1145\/3581783.3612255"},{"key":"4388_CR9","doi-asserted-by":"crossref","unstructured":"X.\u00a0Han, Z.\u00a0Wu, Z.\u00a0Wu, R.\u00a0Yu, and L.S. Davis. Viton: An image-based virtual try-on network. Proceedings of the IEEE conference on computer vision and pattern recognition(CVPR), pages 7543\u20137552, 2018","DOI":"10.1109\/CVPR.2018.00787"},{"key":"4388_CR10","unstructured":"M.\u00a0Heusel, H.\u00a0Ramsauer, T.\u00a0Unterthiner, B.\u00a0Nessler, and S.\u00a0Hochreiter. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems, 30, 2017"},{"issue":"3","key":"4388_CR11","doi-asserted-by":"publisher","first-page":"e2233","DOI":"10.1002\/cav.2233","volume":"35","author":"L Jiang","year":"2024","unstructured":"Jiang, L., Xiong, Y., Wang, Q., Chen, T., Wu, W., Zhou, Z.: Sadnet generating immersive virtual reality avatars by realtime monocular pose estimation. Computer Animation and Virtual Worlds 35(3), e2233 (2024)","journal-title":"Computer Animation and Virtual Worlds"},{"key":"4388_CR12","doi-asserted-by":"crossref","unstructured":"T.\u00a0Karras, S.\u00a0Laine, M.\u00a0Aittala, J.\u00a0Hellsten, J.\u00a0Lehtinen, and T.\u00a0Aila. Analyzing and improving the image quality of stylegan. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 8110\u20138119, 2020","DOI":"10.1109\/CVPR42600.2020.00813"},{"key":"4388_CR13","doi-asserted-by":"crossref","unstructured":"J.\u00a0Kim, G.\u00a0Gu, M.\u00a0Park, S.\u00a0Park, and J.\u00a0Choo. Stableviton: Learning semantic correspondence with latent diffusion model for virtual try-on. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 8176\u20138185, 2024","DOI":"10.1109\/CVPR52733.2024.00781"},{"key":"4388_CR14","unstructured":"D.\u00a0P. Kingma. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114, 2013"},{"key":"4388_CR15","doi-asserted-by":"crossref","unstructured":"S.\u00a0Lee, G.\u00a0Gu, S.\u00a0Park, S.\u00a0Choi, and J.\u00a0Choo. High-resolution virtual try-on with misalignment and occlusion-handled conditions. European Conference on Computer Vision, pages 204\u2013219, 2022","DOI":"10.1007\/978-3-031-19790-1_13"},{"key":"4388_CR16","doi-asserted-by":"crossref","unstructured":"Z.\u00a0Li, P.\u00a0Wei, X.\u00a0Yin, Z.\u00a0Ma, and A.\u00a0C. Kot. Virtual try-on with pose-garment keypoints guided inpainting. Proceedings of the IEEE\/CVF international conference on computer vision, pages 22788\u201322797, 2023","DOI":"10.1109\/ICCV51070.2023.02083"},{"key":"4388_CR17","unstructured":"I.\u00a0Loshchilov and F.\u00a0Hutter. Fixing weight decay regularization in adam. arXiv preprint arXiv:1711.05101, 5(5):5, 2017"},{"key":"4388_CR18","first-page":"10","volume":"3","author":"MR Minar","year":"2020","unstructured":"Minar, M.R., Tuan, T.T., Ahn, H., Rosin, P., Lai, Y.K.: Cp-vton+:clothing shape and texture preserving image-based virtual try-on. CVPR workshops 3, 10\u201314 (2020)","journal-title":"CVPR workshops"},{"key":"4388_CR19","doi-asserted-by":"crossref","unstructured":"D.\u00a0Morelli, A.\u00a0Baldrati, G.\u00a0Cartella, M.\u00a0Cornia, M.\u00a0Bertini, and R.\u00a0Cucchiara. Ladi-vton: Latent diffusion textual-inversion enhanced virtual try-on. Proceedings of the 31st ACM international conference on multimedia, pages 8580\u20138589, 2023","DOI":"10.1145\/3581783.3612137"},{"key":"4388_CR20","doi-asserted-by":"crossref","unstructured":"D.\u00a0Morelli, M.\u00a0Fincato, M.\u00a0Cornia, F.\u00a0Landi, F.\u00a0Cesari, and R.\u00a0Cucchiara. Dress code: High-resolution multi-category virtual try-on. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 2231\u20132235, 2022","DOI":"10.1109\/CVPRW56347.2022.00243"},{"key":"4388_CR21","unstructured":"A.\u00a0Radford, J.\u00a0W. Kim, C.\u00a0Hallacy, A.\u00a0Ramesh, G.\u00a0Goh, S.\u00a0Agarwal, and I.\u00a0Sutskever. Learning transferable visual models from natural language supervision. International conference on machine learning, pages 8748\u20138763, 2021"},{"key":"4388_CR22","doi-asserted-by":"crossref","unstructured":"E.\u00a0Richardson, Y.\u00a0Alaluf, O.\u00a0Patashnik, Y.\u00a0Nitzan, Y.\u00a0Azar, S.\u00a0Shapiro, and D.\u00a0Cohen-Or. Encoding in style: a stylegan encoder for image-to-image translation. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 2287\u20132296, 2021","DOI":"10.1109\/CVPR46437.2021.00232"},{"key":"4388_CR23","doi-asserted-by":"crossref","unstructured":"R.\u00a0Rombach, A.\u00a0Blattmann, D.\u00a0Lorenz, P.\u00a0Esser, and B.\u00a0Ommer. High-resolution image synthesis with latent diffusion models. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 10684\u201310695, 2022","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"4388_CR24","doi-asserted-by":"crossref","unstructured":"O.\u00a0Ronneberger, P.\u00a0Fischer, and T.\u00a0Brox. U-net: Convolutional networks for biomedical image segmentation. International Conference on Medical image computing and computer-assisted intervention, pages 234\u2013241, 2015","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"4388_CR25","doi-asserted-by":"crossref","unstructured":"K.\u00a0Sun, B.\u00a0Xiao, D.\u00a0Liu, and J.\u00a0Wang. Deep high-resolution representation learning for human pose estimation. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 5693\u20135703, 2019","DOI":"10.1109\/CVPR.2019.00584"},{"issue":"3","key":"4388_CR26","doi-asserted-by":"publisher","first-page":"1435","DOI":"10.1007\/s00371-024-03432-0","volume":"41","author":"S Tong","year":"2025","unstructured":"Tong, S., Liu, H., Guo, R., Wang, W., Liu, D.: Context-aware enhanced virtual try-on network with fabric adaptive registration. Vis. Comput. 41(3), 1435\u20131451 (2025)","journal-title":"Vis. Comput."},{"key":"4388_CR27","doi-asserted-by":"crossref","unstructured":"B.\u00a0Wang, H.\u00a0Zheng, X.\u00a0Liang, Y.\u00a0Chen, L.\u00a0Lin, and M.\u00a0Yang. Toward characteristic-preserving image-based virtual try-on network. Proceedings of the European conference on computer vision (ECCV), pages 589\u2013604, 2018","DOI":"10.1007\/978-3-030-01261-8_36"},{"issue":"3","key":"4388_CR28","doi-asserted-by":"publisher","first-page":"e2278","DOI":"10.1002\/cav.2278","volume":"35","author":"L Wang","year":"2024","unstructured":"Wang, L., Wu, Y., Yang, Y.L., Liu, C., Jin, X.: Identity consistent transfer learning of portraits for digital apparel sample display. Computer Animation and Virtual Worlds 35(3), e2278 (2024)","journal-title":"Computer Animation and Virtual Worlds"},{"issue":"4","key":"4388_CR29","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A.C., Sheikh, H.R., Simoncelli, E.P.: Image quality assessment: from error visibility to structural similarity. IEEE Trans. Image Process. 13(4), 600\u2013612 (2004)","journal-title":"IEEE Trans. Image Process."},{"key":"4388_CR30","unstructured":"A.\u00a0Waswani, N.\u00a0Shazeer, N.\u00a0Parmar, J.\u00a0Uszkoreit, L.\u00a0Jones, A.\u00a0Gomez, and I.\u00a0Polosukhin. Attention is all you need. Advances in neural information processing systems, 30, 2017"},{"key":"4388_CR31","doi-asserted-by":"crossref","unstructured":"Z.\u00a0Xie, Z.\u00a0Huang, X.\u00a0Dong, F.\u00a0Zhao, H.\u00a0Dong, X.\u00a0Zhang, and X.\u00a0Liang. Gp-vton: Towards general purpose virtual try-on via collaborative local-flow global-parsing learning. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 23550\u201323559, 2023","DOI":"10.1109\/CVPR52729.2023.02255"},{"issue":"9","key":"4388_CR32","doi-asserted-by":"publisher","first-page":"8996","DOI":"10.1609\/aaai.v39i9.32973","volume":"39","author":"Y Xu","year":"2025","unstructured":"Xu, Y., Gu, T., Chen, W., Chen, C.: Ootdiffusion: Outfitting fusion based latent diffusion for controllable virtual try-on. Proceedings of the AAAI Conference on Artificial Intelligence 39(9), 8996\u20139004 (2025)","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"4388_CR33","doi-asserted-by":"crossref","unstructured":"K.\u00a0Yan, T.\u00a0Gao, H.\u00a0Zhang, and C.\u00a0Xie. Linking garment with person via semantically associated landmarks for virtual try-on. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 17194\u201317204, 2023","DOI":"10.1109\/CVPR52729.2023.01649"},{"key":"4388_CR34","doi-asserted-by":"crossref","unstructured":"X.\u00a0Yang, C.\u00a0Ding, Z.\u00a0Hong, J.\u00a0Huang, J.\u00a0Tao, and X.\u00a0Xu. Texture-preserving diffusionmodels for high-fidelity virtual try-on. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 7017\u20137026, 2024","DOI":"10.1109\/CVPR52733.2024.00670"},{"key":"4388_CR35","doi-asserted-by":"crossref","unstructured":"J.\u00a0Yao, J.\u00a0Chen, L.\u00a0Niu, and B.\u00a0Sheng. Scene-aware human pose generation using transformer. Proceedings of the 31st ACM international conference on multimedia, pages 2847\u20132855, 2023","DOI":"10.1145\/3581783.3612439"},{"key":"4388_CR36","doi-asserted-by":"crossref","unstructured":"Ye, J., Wang, Y., Xie, F., Wang, Q., Gu, X., Wu, Z.: Slot-vton:subject-driven diffusion-based virtual try-on with slot attention. Vis. Comput. 41(5), 3297\u20133308 (2025)","DOI":"10.1007\/s00371-024-03603-z"},{"key":"4388_CR37","doi-asserted-by":"crossref","unstructured":"J.\u00a0Zeng, D.\u00a0Song, W.\u00a0Nie, H.\u00a0Tian, T.\u00a0Wang, and A.\u00a0Liu. Cat-dm: Controllable accelerated virtual try-on with diffusion model. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, 2024","DOI":"10.1109\/CVPR52733.2024.00800"},{"key":"4388_CR38","doi-asserted-by":"crossref","unstructured":"R.\u00a0Zhang, P.\u00a0Isola, A.\u00a0A. Efros, E.\u00a0Shechtman, and O.\u00a0Wang. The unreasonable effectiveness of deep features as a perceptual metric. Proceedings of the IEEE conference on computer vision and pattern recognition, pages 586\u2013595, 2017","DOI":"10.1109\/CVPR.2018.00068"},{"issue":"9","key":"4388_CR39","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vision 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"4388_CR40","doi-asserted-by":"crossref","unstructured":"L.\u00a0Zhu, D.\u00a0Yang, T.\u00a0Zhu, F.\u00a0Reda, W.\u00a0Chan, C.\u00a0Saharia, and I.\u00a0Kemelmacher-Shlizerman. Tryondiffusion: A tale of two unets. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pages 4606\u20134615, 2023","DOI":"10.1109\/CVPR52729.2023.00447"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04388-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-025-04388-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04388-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,22]],"date-time":"2025-09-22T13:17:48Z","timestamp":1758547068000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-025-04388-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,5]]},"references-count":40,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2025,12]]}},"alternative-id":["4388"],"URL":"https:\/\/doi.org\/10.1007\/s11760-025-04388-x","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-6304101\/v1","asserted-by":"object"}]},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,5]]},"assertion":[{"value":"25 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 June 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 June 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 September 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"975"}}