{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T19:57:32Z","timestamp":1785095852658,"version":"3.55.0"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2024,8,23]],"date-time":"2024-08-23T00:00:00Z","timestamp":1724371200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,8,23]],"date-time":"2024-08-23T00:00:00Z","timestamp":1724371200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,3]]},"DOI":"10.1007\/s00371-024-03603-z","type":"journal-article","created":{"date-parts":[[2024,8,23]],"date-time":"2024-08-23T19:02:30Z","timestamp":1724439750000},"page":"3297-3308","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Slot-VTON: subject-driven diffusion-based virtual try-on with slot attention"],"prefix":"10.1007","volume":"41","author":[{"given":"Jianglei","family":"Ye","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yigang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fengmao","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoling","family":"Gu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2103-5037","authenticated-orcid":false,"given":"Zizhao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,8,23]]},"reference":[{"key":"3603_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592450","volume":"42","author":"O Avrahami","year":"2023","unstructured":"Avrahami, O., Fried, O., Lischinski, D.: Blended latent diffusion. ACM Trans. Graph. (TOG) 42, 1\u201311 (2023)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"3603_CR2","unstructured":"Bi\u0144kowski, M., Sutherland, D.J., Arbel, M., Gretton, A.: Demystifying mmd gans (2018). arXiv preprint arXiv:1801.01401"},{"key":"3603_CR3","doi-asserted-by":"publisher","first-page":"2583","DOI":"10.1007\/s00371-022-02480-8","volume":"39","author":"Y Chang","year":"2023","unstructured":"Chang, Y., Peng, T., Yu, F., He, R., Hu, X., Liu, J., Zhang, Z., Jiang, M.: Vtnct: an image-based virtual try-on network by combining feature with pixel transformation. Vis. Comput. 39, 2583\u20132596 (2023)","journal-title":"Vis. Comput."},{"key":"3603_CR4","doi-asserted-by":"crossref","unstructured":"Choi, S., Park, S., Lee, M., Choo, J.: Viton-hd: High-resolution virtual try-on via misalignment-aware normalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14131\u201314140 (2021)","DOI":"10.1109\/CVPR46437.2021.01391"},{"key":"3603_CR5","unstructured":"Cui, A., Mahajan, J., Shah, V., Gomathinayagam, P., Lazebnik, S.: Street tryon: learning in-the-wild virtual try-on from unpaired person images (2023). arXiv preprint arXiv:2311.16094"},{"key":"3603_CR6","doi-asserted-by":"crossref","unstructured":"Duchon, J.: Splines minimizing rotation-invariant semi-norms in sobolev spaces. In: Constructive Theory of Functions of Several Variables: Proceedings of a Conference Held at Oberwolfach April 25\u2013May 1, 1976, pp. 85\u2013100. Springer (1977)","DOI":"10.1007\/BFb0086566"},{"key":"3603_CR7","unstructured":"Gal, R., Alaluf, Y., Atzmon, Y., Patashnik, O., Bermano, A.H., Chechik, G., Cohen-Or, D.: An image is worth one word: personalizing text-to-image generation using textual inversion (2022). arXiv preprint arXiv:2208.01618"},{"key":"3603_CR8","doi-asserted-by":"crossref","unstructured":"Ge, Y., Song, Y., Zhang, R., Ge, C., Liu, W., Luo, P.: Parser-free virtual try-on via distilling appearance flows. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8485\u20138493 (2021)","DOI":"10.1109\/CVPR46437.2021.00838"},{"key":"3603_CR9","unstructured":"Goodfellow, I.J., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., Bengio, Y.: Generative adversarial networks (2014). arXiv:1406.2661"},{"key":"3603_CR10","doi-asserted-by":"crossref","unstructured":"Gou, J., Sun, S., Zhang, J., Si, J., Qian, C., Zhang, L.: Taming the power of diffusion models for high-quality virtual try-on with appearance flow. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 7599\u20137607 (2023)","DOI":"10.1145\/3581783.3612255"},{"key":"3603_CR11","doi-asserted-by":"publisher","first-page":"2735","DOI":"10.1109\/TCYB.2019.2934823","volume":"51","author":"H Guo","year":"2019","unstructured":"Guo, H., Sheng, B., Li, P., Chen, C.P.: Multiview high dynamic range image synthesis using fuzzy broad learning system. IEEE Trans. Cybern. 51, 2735\u20132747 (2019)","journal-title":"IEEE Trans. Cybern."},{"key":"3603_CR12","doi-asserted-by":"crossref","unstructured":"Han, X., Hu, X., Huang, W., Scott, M.R.: Clothflow: a flow-based model for clothed person generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10471\u201310480 (2019)","DOI":"10.1109\/ICCV.2019.01057"},{"key":"3603_CR13","doi-asserted-by":"crossref","unstructured":"Han, X., Wu, Z., Wu, Z., Yu, R., Davis, L.S.: Viton: an image-based virtual try-on network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7543\u20137552 (2018)","DOI":"10.1109\/CVPR.2018.00787"},{"key":"3603_CR14","doi-asserted-by":"crossref","unstructured":"He, S., Song, Y.Z., Xiang, T.: Style-based global appearance flow for virtual try-on. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3470\u20133479 (2022)","DOI":"10.1109\/CVPR52688.2022.00346"},{"key":"3603_CR15","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: Gans trained by a two time-scale update rule converge to a local nash equilibrium. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"3603_CR16","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3603_CR17","doi-asserted-by":"publisher","first-page":"3365","DOI":"10.1007\/s00371-022-02563-6","volume":"38","author":"X Hu","year":"2022","unstructured":"Hu, X., Zhang, J., Huang, J., Liang, J., Yu, F., Peng, T.: Virtual try-on based on attention u-net. Vis. Comput. 38, 3365\u20133376 (2022)","journal-title":"Vis. Comput."},{"key":"3603_CR18","doi-asserted-by":"crossref","unstructured":"Issenhuth, T., Mary, J., Calauzenes, C.: Do not mask what you do not need to mask: a parser-free virtual try-on. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XX 16, pp. 619\u2013635. Springer (2020)","DOI":"10.1007\/978-3-030-58565-5_37"},{"key":"3603_CR19","doi-asserted-by":"crossref","unstructured":"Johnson, J., Alahi, A., Fei-Fei, L.: Perceptual losses for real-time style transfer and super-resolution. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part II 14, pp. 694\u2013711. Springer (2016)","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"3603_CR20","doi-asserted-by":"crossref","unstructured":"Kawar, B., Zada, S., Lang, O., Tov, O., Chang, H., Dekel, T., Mosseri, I., Irani, M.: Imagic: text-based real image editing with diffusion models (2023). arXiv:2210.09276","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"3603_CR21","doi-asserted-by":"crossref","unstructured":"Kim, J., Gu, G., Park, M., Park, S., Choo, J.: Stableviton: learning semantic correspondence with latent diffusion model for virtual try-on (2023). arXiv preprint arXiv:2312.01725","DOI":"10.1109\/CVPR52733.2024.00781"},{"key":"3603_CR22","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes (2013). arXiv preprint arXiv:1312.6114"},{"key":"3603_CR23","unstructured":"Kipf, T., Elsayed, G.F., Mahendran, A., Stone, A., Sabour, S., Heigold, G., Jonschkowski, R., Dosovitskiy, A., Greff, K.: Conditional object-centric learning from video (2021). arXiv preprint arXiv:2111.12594"},{"key":"3603_CR24","doi-asserted-by":"crossref","unstructured":"Kumari, N., Zhang, B., Zhang, R., Shechtman, E., Zhu, J.Y.: Multi-concept customization of text-to-image diffusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1931\u20131941 (2023)","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"3603_CR25","doi-asserted-by":"crossref","unstructured":"Lee, S., Gu, G., Park, S., Choi, S., Choo, J.: High-resolution virtual try-on with misalignment and occlusion-handled conditions. In: European Conference on Computer Vision, pp. 204\u2013219. Springer (2022)","DOI":"10.1007\/978-3-031-19790-1_13"},{"key":"3603_CR26","doi-asserted-by":"publisher","first-page":"50","DOI":"10.1109\/TMM.2021.3120873","volume":"25","author":"X Lin","year":"2021","unstructured":"Lin, X., Sun, S., Huang, W., Sheng, B., Li, P., Feng, D.D.: Eapt: efficient attention pyramid transformer for image processing. IEEE Trans. Multimed. 25, 50\u201361 (2021)","journal-title":"IEEE Trans. Multimed."},{"key":"3603_CR27","unstructured":"Liu, L., Ren, Y., Lin, Z., Zhao, Z.: Pseudo numerical methods for diffusion models on manifolds (2022). arXiv preprint arXiv:2202.09778"},{"key":"3603_CR28","doi-asserted-by":"crossref","unstructured":"Liu, Y., Jiang, T., Si, P., Zhu, S., Yan, C., Wang, S., Yin, H.: Unpaired semantic neural person image synthesis. Vis. Comput. 1\u201315 (2024)","DOI":"10.1007\/s00371-024-03331-4"},{"key":"3603_CR29","first-page":"11525","volume":"33","author":"F Locatello","year":"2020","unstructured":"Locatello, F., Weissenborn, D., Unterthiner, T., Mahendran, A., Heigold, G., Uszkoreit, J., Dosovitskiy, A., Kipf, T.: Object-centric learning with slot attention. Adv. Neural Inf. Process. Syst. 33, 11525\u201311538 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3603_CR30","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization (2017). arXiv preprint arXiv:1711.05101"},{"key":"3603_CR31","doi-asserted-by":"crossref","unstructured":"Morelli, D., Baldrati, A., Cartella, G., Cornia, M., Bertini, M., Cucchiara, R.: Ladi-vton: latent diffusion textual-inversion enhanced virtual try-on (2023). arXiv preprint arXiv:2305.13501","DOI":"10.1145\/3581783.3612137"},{"key":"3603_CR32","doi-asserted-by":"crossref","unstructured":"Parmar, G., Kumar\u00a0Singh, K., Zhang, R., Li, Y., Lu, J., Zhu, J.Y.: Zero-shot image-to-image translation. In: ACM SIGGRAPH 2023 Conference Proceedings, pp 1\u201311 (2023)","DOI":"10.1145\/3588432.3591513"},{"key":"3603_CR33","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J., Rombach, R.: Sdxl: improving latent diffusion models for high-resolution image synthesis (2023). arXiv preprint arXiv:2307.01952"},{"key":"3603_CR34","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J. et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, PMLR. pp. 8748\u20138763 (2021)"},{"key":"3603_CR35","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"3603_CR36","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: Convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-assisted Intervention\u2013MICCAI 2015: 18th International Conference, Munich, Germany, October 5-9, 2015, proceedings, part III 18, pp. 234\u2013241. Springer (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"3603_CR37","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: Dreambooth: fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22500\u201322510 (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"3603_CR38","doi-asserted-by":"crossref","unstructured":"Saharia, C., Chan, W., Chang, H., Lee, C., Ho, J., Salimans, T., Fleet, D., Norouzi, M.: Palette: image-to-image diffusion models. In: ACM SIGGRAPH 2022 Conference Proceedings, pp. 1\u201310 (2022a)","DOI":"10.1145\/3528233.3530757"},{"key":"3603_CR39","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia, C., Chan, W., Saxena, S., Li, L., Whang, J., Denton, E.L., Ghasemipour, K., Gontijo Lopes, R., Karagol Ayan, B., Salimans, T., et al.: Photorealistic text-to-image diffusion models with deep language understanding. Adv. Neural Inf. Process. Syst. 35, 36479\u201336494 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3603_CR40","unstructured":"Singh, G., Deng, F., Ahn, S.: Illiterate dall-e learns to compose (2021). arXiv preprint arXiv:2110.11405"},{"key":"3603_CR41","first-page":"18181","volume":"35","author":"G Singh","year":"2022","unstructured":"Singh, G., Wu, Y.F., Ahn, S.: Simple unsupervised object-centric learning for complex and naturalistic videos. Adv. Neural Inf. Process. Syst. 35, 18181\u201318196 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"3603_CR42","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. In: International Conference on Machine Learning, PMLR. pp. 2256\u20132265 (2015)"},{"key":"3603_CR43","doi-asserted-by":"crossref","unstructured":"Song, D., Zhang, X., Zhou, J., Nie, W., Tong, R., Liu, A.A.: Image-based virtual try-on: a survey (2023). arXiv preprint arXiv:2311.04811","DOI":"10.1007\/s11263-024-02305-2"},{"key":"3603_CR44","doi-asserted-by":"crossref","unstructured":"Tumanyan, N., Geyer, M., Bagon, S., Dekel, T.: Plug-and-play diffusion features for text-driven image-to-image translation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1921\u20131930 (2023)","DOI":"10.1109\/CVPR52729.2023.00191"},{"key":"3603_CR45","doi-asserted-by":"crossref","unstructured":"Wang, B., Zheng, H., Liang, X., Chen, Y., Lin, L., Yang, M.: Toward characteristic-preserving image-based virtual try-on network. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 589\u2013604 (2018)","DOI":"10.1007\/978-3-030-01261-8_36"},{"key":"3603_CR46","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A.C., Sheikh, H.R., Simoncelli, E.P.: Image quality assessment: from error visibility to structural similarity. IEEE Trans. Image Process. 13, 600\u2013612 (2004)","journal-title":"IEEE Trans. Image Process."},{"key":"3603_CR47","unstructured":"Wu, Z., Hu, J., Lu, W., Gilitschenski, I., Garg, A.: Slotdiffusion: object-centric generative modeling with diffusion models. Adv. Neural Inf. Process. Syst. 36 (2024)"},{"key":"3603_CR48","doi-asserted-by":"crossref","unstructured":"Xie, Z., Huang, Z., Dong, X., Zhao, F., Dong, H., Zhang, X., Zhu, F., Liang, X.: Gp-vton: towards general purpose virtual try-on via collaborative local-flow global-parsing learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23550\u201323559 (2023)","DOI":"10.1109\/CVPR52729.2023.02255"},{"key":"3603_CR49","doi-asserted-by":"publisher","first-page":"4499","DOI":"10.1109\/TNNLS.2021.3116209","volume":"34","author":"Z Xie","year":"2021","unstructured":"Xie, Z., Zhang, W., Sheng, B., Li, P., Chen, C.P.: Bagfn: broad attentive graph fusion network for high-order feature interactions. IEEE Trans. Neural Netw. Learn. Syst. 34, 4499\u20134513 (2021)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"3603_CR50","doi-asserted-by":"crossref","unstructured":"Yan, K., Gao, T., Zhang, H., Xie, C.: Linking garment with person via semantically associated landmarks for virtual try-on. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17194\u201317204 (2023)","DOI":"10.1109\/CVPR52729.2023.01649"},{"key":"3603_CR51","doi-asserted-by":"crossref","unstructured":"Yang, B., Gu, S., Zhang, B., Zhang, T., Chen, X., Sun, X., Chen, D., Wen, F.: Paint by example: exemplar-based image editing with diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18381\u201318391 (2023)","DOI":"10.1109\/CVPR52729.2023.01763"},{"key":"3603_CR52","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., Wang, O.: The unreasonable effectiveness of deep features as a perceptual metric. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 586\u2013595 (2018)","DOI":"10.1109\/CVPR.2018.00068"},{"key":"3603_CR53","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Han, L., Ghosh, A., Metaxas, D., Ren, J.: Sine: single image editing with text-to-image diffusion models (2022). arXiv:2212.04489","DOI":"10.1109\/CVPR52729.2023.00584"},{"key":"3603_CR54","doi-asserted-by":"crossref","unstructured":"Zhu, L., Yang, D., Zhu, T., Reda, F., Chan, W., Saharia, C., Norouzi, M., Kemelmacher-Shlizerman, I.: Tryondiffusion: a tale of two unets. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4606\u20134615 (2023)","DOI":"10.1109\/CVPR52729.2023.00447"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03603-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-024-03603-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-024-03603-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,10]],"date-time":"2025-03-10T09:10:35Z","timestamp":1741597835000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-024-03603-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,23]]},"references-count":54,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,3]]}},"alternative-id":["3603"],"URL":"https:\/\/doi.org\/10.1007\/s00371-024-03603-z","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,23]]},"assertion":[{"value":"5 August 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 August 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}