{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:16:10Z","timestamp":1777655770311,"version":"3.51.4"},"reference-count":64,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,3,20]],"date-time":"2025-03-20T00:00:00Z","timestamp":1742428800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,20]],"date-time":"2025-03-20T00:00:00Z","timestamp":1742428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach. Intell. Res."],"published-print":{"date-parts":[[2025,8]]},"DOI":"10.1007\/s11633-024-1511-7","type":"journal-article","created":{"date-parts":[[2025,3,20]],"date-time":"2025-03-20T00:49:22Z","timestamp":1742431762000},"page":"752-768","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["CMSL: Cross-modal Style Learning for Few-shot Image Generation"],"prefix":"10.1007","volume":"22","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8026-9354","authenticated-orcid":false,"given":"Yue","family":"Jiang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4028-5250","authenticated-orcid":false,"given":"Yueming","family":"Lyu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Peng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2763-7832","authenticated-orcid":false,"given":"Jing","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,20]]},"reference":[{"key":"1511_CR1","doi-asserted-by":"publisher","first-page":"2672","DOI":"10.5555\/2969033.2969125","volume-title":"Proceedings of the 27th International Conference on Neural Information Processing Systems","author":"I J Goodfellow","year":"2014","unstructured":"I. J. Goodfellow, J. Pouget-Abadie, M. Mirza, B. Xu, D. Warde-Farley, S. Ozair, A. Courville, Y. Bengio. Generative adversarial nets. In Proceedings of the 27th International Conference on Neural Information Processing Systems, Montreal, Canada, pp. 2672\u20132680, 2014. DOI: https:\/\/doi.org\/10.5555\/2969033.2969125."},{"key":"1511_CR2","volume-title":"Unsupervised representation learning with deep convolutional generative adversarial networks","author":"A Radford","year":"2016","unstructured":"A. Radford, L. Metz, S. Chintala. Unsupervised representation learning with deep convolutional generative adversarial networks, [Online], Available: https:\/\/arxiv.org\/abs\/1511.06434, 2016."},{"key":"1511_CR3","doi-asserted-by":"publisher","first-page":"2242","DOI":"10.1109\/ICCV.2017.244","volume-title":"Proceedings of IEEE International Conference on Computer Vision","author":"J Y Zhu","year":"2017","unstructured":"J. Y. Zhu, T. Park, P. Isola, A. A. Efros. Unpaired image-to-image translation using cycle-consistent adversarial networks. In Proceedings of IEEE International Conference on Computer Vision, Venice, ttaly, pp. 2242\u20132251, 2017. DOI: https:\/\/doi.org\/10.1109\/ICCV.2017.244."},{"key":"1511_CR4","volume-title":"Conditional generative adversarial nets","author":"M Mirza","year":"2014","unstructured":"M. Mirza, S. Osindero. Conditional generative adversarial nets, [Online], Available: https:\/\/arxiv.org\/abs\/1411.1784, 2014."},{"key":"1511_CR5","doi-asserted-by":"publisher","first-page":"2642","DOI":"10.5555\/3305890.3305954","volume-title":"Proceedings of the 34th International Conference on Machine Learning","author":"A Odena","year":"2017","unstructured":"A. Odena, C. Olah, J. Shlens. Conditional image synthesis with auxiliary classifier GANs. In Proceedings of the 34th International Conference on Machine Learning, Sydney, Australia, pp. 2642\u20132651, 2017. DOI: https:\/\/doi.org\/10.5555\/3305890.3305954."},{"key":"1511_CR6","doi-asserted-by":"publisher","first-page":"2180","DOI":"10.5555\/3157096.3157340","volume-title":"Proceedings of the 30th International Conference on Neural Information Processing Systems","author":"X Chen","year":"2016","unstructured":"X. Chen, Y. Duan, R. Houthoftt, J. Schulman, I. Sutskever, P. Abbeel. InfoGAN: Interpretable representation learning by information maximizing generative adversarial nets. In Proceedings of the 30th International Conference on Neural Information Processing Systems, Barcelona, Spain, pp. 2180\u20132188, 2016. DOI: https:\/\/doi.org\/10.5555\/3157096.3157340."},{"key":"1511_CR7","doi-asserted-by":"publisher","first-page":"4396","DOI":"10.1109\/CVPR.2019.00453","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"T Karras","year":"2019","unstructured":"T. Karras, S. Laine, T. Aila. A style-based generator architecture for generative adversarial networks. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, USA, pp. 4396\u20134405, 2019. DOI: https:\/\/doi.org\/10.1109\/CVPR.2019.00453."},{"issue":"11","key":"1511_CR8","doi-asserted-by":"publisher","first-page":"13438","DOI":"10.1109\/TPAMI.2023.3290175","volume":"45","author":"Y M Lyu","year":"2023","unstructured":"Y. M Lyu, Y. Jiang, Z. W. He, B. Peng, Y. F. Liu, J. Dong. 3D-aware adversarial makeup generation for facial privacy protection. IEEE Transactions on Pattern Analysis and Machine Intelligence, vol. 45, no. 11, pp. 13438\u201313453, 2023. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2023.3290175.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"2","key":"1511_CR9","doi-asserted-by":"publisher","first-page":"369","DOI":"10.1007\/s11633-022-1408-2","volume":"21","author":"X L Zhu","year":"2024","unstructured":"X. L. Zhu, Z. Zhang, W. Wang, Z. L. Wang. Comprehensive relation modelling for image paragraph generation. Machine Intelligence Research, vol. 21, no. 2, pp. 369\u2013382, 2024. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1408-2.","journal-title":"Machine Intelligence Research"},{"key":"1511_CR10","volume-title":"Invertible conditional GANs for image editing","author":"G Perarnau","year":"2016","unstructured":"G. Perarnau, J. van de Weijer, B. Raducanu, \u00c1. J. M. lvarez. Invertible conditional GANs for image editing, [Online], Available: https:\/\/arxiv.org\/abs\/1611.06355, 2016."},{"key":"1511_CR11","doi-asserted-by":"publisher","first-page":"469","DOI":"10.5555\/3157096.3157149","volume-title":"Proceedings of the 30th International Conference on Neural Information Processing Systems","author":"M Y Liu","year":"2016","unstructured":"M. Y. Liu, O. Tuzel. Coupled generative adversarial networks. In Proceedings of the 30th International Conference on Neural Information Processing Systems, Barcelona, Spain, pp. 469\u2013477, 2016. DOI: https:\/\/doi.org\/10.5555\/3157096.3157149."},{"issue":"4","key":"1511_CR12","doi-asserted-by":"publisher","first-page":"257","DOI":"10.1007\/s11633-022-1349-9","volume":"19","author":"D P Fan","year":"2022","unstructured":"D. P. Fan, Z. L. Huang, P. Zheng, H. Liu, X. B. Qin, L. van Gool. Facial-sketch synthesis: a new challenge Machine Intelligence Research, vol. 19, no. 4, pp. 257\u2013287, 2022 DOI: https:\/\/doi.org\/10.1007\/s11633-022-1349-9","journal-title":"Machine Intelligence Research"},{"key":"1511_CR13","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1109\/CVPR.2017.19","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"C Ledig","year":"2017","unstructured":"C. Ledig, L. Theis, Huszar, J. Caballero, A. Cunningham, A. Acosta, A. Aitken, A. Teiani, J. Totz, Z. H. Wang, W. Z. Shi. Photo-realistic single image super-resolution using a generative adversarial network. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Honolulu, USA, pp. 105\u2013114, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.19."},{"key":"1511_CR14","doi-asserted-by":"publisher","first-page":"2536","DOI":"10.1109\/CVPR.2016.278","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"D Pathak","year":"2016","unstructured":"D. Pathak, P. Kr\u00e4henb\u00fchl, J. Donahue, T. Darrell, A. A. Efros. Context encoders: Feature learning by inpainting. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, USA, pp. 2536\u20132544, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.278."},{"key":"1511_CR15","doi-asserted-by":"publisher","first-page":"217","DOI":"10.5555\/3157096.3157121","volume-title":"Proceedings of the 30th International Conference on Neural Information Processing Systems","author":"S Reed","year":"2016","unstructured":"S. Reed, Z. Akata, S. Mohan, S. Tenka, B. Schiele, H. Lee. Learning what and where to draw. In Proceedings of the 30th International Conference on Neural Information Processing Systems, Barcelona, Spain, pp. 217\u2013225, 2016. DOI: https:\/\/doi.org\/10.5555\/3157096.3157121."},{"key":"1511_CR16","volume-title":"TAC-GAN-text conditioned auxiliary classifier generative adversarial network","author":"A Dash","year":"2017","unstructured":"A. Dash, J. C. B. Gamboa, S. Ahmed, M. Liwicki, M. Z. Afzal. TAC-GAN-text conditioned auxiliary classifier generative adversarial network, [Online], Available: https:\/\/arxiv.org\/abs\/1703.06412, 2017."},{"issue":"1","key":"1511_CR17","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1007\/s11633-023-1457-1","volume":"21","author":"N Bryan-Kinns","year":"2024","unstructured":"N. Bryan-Kinns, B. Y. Zhang, S. Y. Zhao, B. Banar. Exploring variational auto-encoder architectures, configurations, and datasets for generative music explainable AI. Machine Intelligence Research, vol. 21, no. 1, pp. 29\u201345, 2024. DOI: https:\/\/doi.org\/10.1007\/s11633-023-1457-1.","journal-title":"Machine Intelligence Research"},{"issue":"4","key":"1511_CR18","doi-asserted-by":"publisher","first-page":"569","DOI":"10.1007\/s11633-022-1386-4","volume":"20","author":"H Y Lu","year":"2023","unstructured":"H. Y. Lu, Y. Q. Huo, M. Y. Ding, M. Y. Fei, Z. W. Lu. Cross-modal contrastive learning for generalizable and efficient image-text retrival. Machine Intelligence Research, vol. 20, no. 4, pp. 569\u2013582, 2023. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1386-4.","journal-title":"Machine Intelligence Research"},{"key":"1511_CR19","doi-asserted-by":"publisher","first-page":"220","DOI":"10.1007\/978-3-030-01231-1_14","volume-title":"Proceedings of the 15th European Conference on Computer Vision","author":"Y X Wang","year":"2018","unstructured":"Y. X. Wang, C. S. Wu, L. Herranz, J. vein de Weijer, A. Gonzalez-Garcia, B. Raducanu. Transferring GANs: Generating images from limited data. In Proceedings of the 15th European Conference on Computer Vision, Munich, Germany, pp. 220\u2013236, 2018. DOI: https:\/\/doi.org\/10.1007\/978-3-030-01231-1_14."},{"key":"1511_CR20","volume-title":"Freeze the discriminator: a simple baseline for fine-tuning GANs","author":"S Mo","year":"2020","unstructured":"S. Mo, M. Cho, J. Shin. Freeze the discriminator: a simple baseline for fine-tuning GANs, [Online], Available: https:\/\/arxiv.org\/abs\/2002.10964, 2020."},{"key":"1511_CR21","doi-asserted-by":"publisher","first-page":"9329","DOI":"10.1109\/CVPR42600.2020.00935","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y X Wang","year":"2020","unstructured":"Y. X. Wang, A. Gonzalez-Garcia, D. Berga, L. Herranz, F. S. Khan, J. van de Weijer. MineGAN: Effective knowledge transfer from GANs to target domains with few images. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, USA, pp. 9329\u20139338, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00935."},{"key":"1511_CR22","volume-title":"Few-shot adaptation of generative adversarial networks","author":"E Robb","year":"2020","unstructured":"E. Robb, W. S. Chu, A. Kumar, J. B. Huang. Few-shot adaptation of generative adversarial networks, [Online], Available: https:\/\/arxiv.org\/abs\/2010.11943, 2020."},{"key":"1511_CR23","doi-asserted-by":"publisher","first-page":"10738","DOI":"10.1109\/CVPR46437.2021.01060","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"U Ojha","year":"2021","unstructured":"U. Ojha, Y. J. Li, J. W. Lu, A. A. Efros, Y. J. Lee, E. Shechtman, R Zhang. Few-shot image generation via cross-domain correspondence. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Nashville, USA, pp. 10738\u201310747, 2021. DOI: https:\/\/doi.org\/10.1109\/CVPR46437.2021.01060."},{"key":"1511_CR24","doi-asserted-by":"publisher","first-page":"11194","DOI":"10.1109\/CVPR52688.2022.01092","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J Y Xiao","year":"2022","unstructured":"J. Y. Xiao, L. Li, C. F. Wang, Z. J. Zha, Q. M. Huang. Few shot generative model adaption via relaxed spatial structural alignment. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, USA, pp. 11194\u201311203, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.01092."},{"key":"1511_CR25","doi-asserted-by":"publisher","unstructured":"J. Yaniv, Y. Newman, A. Shamir. The face of art: Landmark detection and geometric style in portraits. ACM Transactions on Graphics, vol. 38, no. 4, Article number 60, 2019. DOI: https:\/\/doi.org\/10.1145\/3306346.3322984.","DOI":"10.1145\/3306346.3322984"},{"key":"1511_CR26","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3495883","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"T B Brown","year":"2020","unstructured":"T. B. Brown, B. Mann, N. Ryder, M. Subbiah, J. Kaplan, P. Dhariwal, A. Neelakantan, P. Shyam, G. Sastry, A. Askell, S. Agarwal, A. Herbert-Voss, G. Krueger, T. Henighan, R. Child, A. Ramesh, D. M. Ziegler, J. Wu, C. Winter, C. Hesse, M. Chen, E. Sigler, M. Litwin, S. Gray, B. Chess, J. Clark, C. Berner, S. McCandlish, A. Radford, I. Sutskever, D. Amodei. Language models are few-shot learners. In Proceedings of the 34th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 159, 2020. DOI: https:\/\/doi.org\/10.5555\/3495724.3495883."},{"key":"1511_CR27","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3497056","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"Y J Li","year":"2020","unstructured":"Y. J. Li, R. Zhang, J. W. Lu, E. Shechtman. Few-shot image generation with elastic weight consolidation. In Proceedings of the 34th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 1332, 2020. DOI: https:\/\/doi.org\/10.5555\/3495724.3497056."},{"key":"1511_CR28","doi-asserted-by":"publisher","unstructured":"B. Cao, Q. H. Wang, P. F. Zhu, Q. H. Hu, D. W. Ren, W. M. Zuo, X. B. Gao. Multi-view knowledge ensemble with frequency consistency for cross-domain face translation. IEEE Transactions on Neural Networks and Learning Systems, to be published. DOI: https:\/\/doi.org\/10.1109\/TNNLS.2023.3236486.","DOI":"10.1109\/TNNLS.2023.3236486"},{"key":"1511_CR29","doi-asserted-by":"publisher","first-page":"9130","DOI":"10.1109\/CVPR52688.2022.00893","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Q Zhao","year":"2022","unstructured":"Y. Q. Zhao, H. H. Ding, H. J. Huang, N. M. Cheung. A closer look at few-shot image generation. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, USA, pp. 9130\u20139140, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.00893."},{"issue":"10","key":"1511_CR30","doi-asserted-by":"publisher","first-page":"12179","DOI":"10.1109\/TPAMI.2023.3283551","volume":"45","author":"G Kwon","year":"2023","unstructured":"G. Kwon, J. C. Ye. One-shot adaptation of GAN in just one clip. IEEE Transact ons on Pattern Analysis and Machine Intelligence, vol. 45, no. 10, pp. 12179\u201312191, 2023. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2023.3283551.","journal-title":"IEEE Transact ons on Pattern Analysis and Machine Intelligence"},{"key":"1511_CR31","doi-asserted-by":"publisher","DOI":"10.5555\/3600270.3602973","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Y Zhang","year":"2022","unstructured":"Y. Zhang, M. S. Yao, Y. X. Wei, Z. L. Ji, J. F. Bai, W. M. Zuo. Towards diverse and faithful one-shot adaption of generative advrsaarial network.. In Proceedings of the 36th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 2003, 2022. DOI: https:\/\/doi.org\/10.5555\/3600270.3602973."},{"key":"1511_CR32","doi-asserted-by":"publisher","first-page":"2256","DOI":"10.5555\/3045118.3045358","volume-title":"Proceedings of the 32nd International Conference on Machine Learning","author":"J Sohl-Dickstein","year":"2015","unstructured":"J. Sohl-Dickstein, E. A. Weiss, N. Maheswaranathan, S. Ganguli. Deep unsupervised learning using nonequilibrium thermodynamics. In Proceedings of the 32nd International Conference on Machine Learning, Lille, France, pp. 2256\u20132265, 2015. DOI: https:\/\/doi.org\/10.5555\/3045118.3045358."},{"key":"1511_CR33","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3496298","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"J Ho","year":"2020","unstructured":"J. Ho, A. Jain, P. Abbeel. Denoising diffusion probabilistic models. In Proceedings of the 34th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 504, 2020. DOI: https:\/\/doi.org\/10.5555\/3495724.3496298."},{"key":"1511_CR34","doi-asserted-by":"publisher","first-page":"10674","DOI":"10.1109\/CVPR52688.2022.01042","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"R Rombach","year":"2022","unstructured":"R. Rombach, A. Blattmann, D. Lorenz, P. Esser, B. Ommer. High-resolution image synthesis with latent diffusion models. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, USA, pp. 10674\u201310685, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.01042."},{"key":"1511_CR35","volume-title":"IP-adapter: Text compatible image prompt adapter for text-to-image diffusion model","author":"H Ye","year":"2023","unstructured":"H. Ye, J. Zhang, S. B. Liu, X. Han, W. Yang. IP-adapter: Text compatible image prompt adapter for text-to-image diffusion model, [Onlme], Available: https:\/\/arxiv.org\/abs\/2308.06721, 2023."},{"key":"1511_CR36","volume-title":"PhotoMaker: Customizing realistic human photos via stacked ID embedding","author":"Z Li","year":"2023","unstructured":"Z. Li, M. D. Cao, X. T. Wang, Z. A. Qi, M. M. Cheng, Y. Shan. PhotoMaker: Customizing realistic human photos via stacked ID embedding, [Online], Available: https:\/\/arxiv.org\/abs\/2312.04461, 2023."},{"key":"1511_CR37","doi-asserted-by":"publisher","first-page":"5967","DOI":"10.1109\/CVPR.2017.632","volume-title":"Proceedings of IEEE Conference on Computer Vision and Pattern Recognition","author":"P Isola","year":"2017","unstructured":"P. Isola, J. Y. Zhu, T. H. Zhou, A. A. Efros. Image-to-image translation with conditional adversarial networks. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Honolulu, USA, pp. 5967\u20135976, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.632."},{"issue":"4","key":"1511_CR38","doi-asserted-by":"publisher","first-page":"2000","DOI":"10.1109\/TCSVT.2023.3300906","volume":"34","author":"Y M Lyu","year":"2024","unstructured":"Y. M. Lyu, Y. Jiang, B. Peng, J. Dong. InfoStyler: Disentanglement information bottleneck for artistic style transfer. IEEE Transactions on Circuits and Systems for Video Technology, vol. 34, no. 4, pp. 2000\u20132082, 2024. DOI: https:\/\/doi.org\/10.1109\/TCSVT.2023.3300906.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"1511_CR39","doi-asserted-by":"publisher","first-page":"3474","DOI":"10.1109\/CVPR.2018.00366","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"H M Yang","year":"2018","unstructured":"H. M. Yang, X. Y. Zhang, F. Yin, C. L. Liu. Robust classification with convolutional prototype learning. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Salt Lake City, USA, pp. 3474\u20133482, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00366."},{"key":"1511_CR40","doi-asserted-by":"publisher","first-page":"3141","DOI":"10.5555\/3304889.3305097","volume-title":"Proceedings of the 20th International Joint Conference on Artificial Intelligence","author":"X X Zhang","year":"2018","unstructured":"X. X. Zhang, Z. F. Zhu, Y. Zhao, D. Q. Kong. Self-supervised deep low-rank assignment model for prototype selection. In Proceedings of the 20th International Joint Conference on Artificial Intelligence, Stockholm, Sweden, pp. 3141\u20133147, 2018. DOI: https:\/\/doi.org\/10.5555\/3304889.3305097."},{"issue":"7","key":"1511_CR41","doi-asserted-by":"publisher","first-page":"4513","DOI":"10.1109\/TCSVT.2021.3128054","volume":"32","author":"F T Zhou","year":"2022","unstructured":"F. T. Zhou, S. Huang, B. Liu, D. Yang. Multi-label image classification via category prototype compositional learning. IEEE Transactions on Circuits and Systems for Video Technology, vol. 32, no. 7, pp. 4513\u20134525, 2022. DOI: https:\/\/doi.org\/10.1109\/TCSVT.2021.3128054.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"1511_CR42","doi-asserted-by":"publisher","unstructured":"Y. Huang, Y. M. Wang, Y. Zeng, L. Wang. MACK: Multimodal aligned conceptual knowledge for unpaired image-text matchmg. In Proceedings of the 36th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 573, 2022. DOI: https:\/\/doi.org\/10.5555\/3600270.3600843.","DOI":"10.5555\/3600270.3600843"},{"key":"1511_CR43","doi-asserted-by":"publisher","first-page":"7723","DOI":"10.1109\/TCSVT.2023.3281151","volume-title":"IEEE Transactions on Circuits and Systems for Video Technology","author":"Y Y Su","year":"2023","unstructured":"Y. Y. Su, X. Xu, K. Jia. Weakly supervised 3D point cloud segmentation via multi-prototype learning. In IEEE Transactions on Circuits and Systems for Video Technology, vol. 33, no. 12, pp. 7723\u20137736, 2023. DOI: https:\/\/doi.org\/10.1109\/TCSVT.2023.3281151."},{"key":"1511_CR44","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"A Radford","year":"2021","unstructured":"A. Radford, J. W. Kim, C. Hallacy, A. Ramesh, G. Goh, S. Agarwal, G. Sastry, A. Askell, P. Mishkin, J. Clark, G. Krueger, I. Sutskever. Learning transferable visual models from natural language supervision. In Proceedings of the 38th International Conference on Machine Learning, pp. 8748\u20138763, 2021."},{"key":"1511_CR45","doi-asserted-by":"publisher","unstructured":"R. Gal, O. Patashnik, H. Maron, A. H. Bermano, G. Chechik, D. Cohen-Or. StyleGAN-NADA: CLIP-guided domain adaptation of image generators. ACM Transactions on Graphics, vol. 41, no. 4, Article number 141, 2022. DOI: https:\/\/doi.org\/10.1145\/3528223.3530164.","DOI":"10.1145\/3528223.3530164"},{"key":"1511_CR46","doi-asserted-by":"publisher","first-page":"18041","DOI":"10.1109\/CVPR52688.2022.01753","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"G Kwon","year":"2022","unstructured":"G. Kwon, J. C. Ye. CLIPstyler: Image style transfer with a single text condition. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, USA, pp. 18041\u201318050, 2022. DOI: https:\/\/doi.org\/10.1109\/CVPR52688.2022.01753."},{"key":"1511_CR47","doi-asserted-by":"publisher","first-page":"2065","DOI":"10.1109\/ICCV48922.2021.00209","volume-title":"Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"O Patashnik","year":"2021","unstructured":"O. Patashnik, Z. Z. Wu, E. Shechtman, D. Cohen-Or, D. Lischinski. StyleCLIP: Text-driven manipulation of styleGAN imagery. In Proceedings of IEEE\/CVF International Conference on Computer Vision, Montreal, Canada, pp. 2065\u20132074, 2021. DOI: https:\/\/doi.org\/10.1109\/ICCV48922.2021.00209."},{"key":"1511_CR48","doi-asserted-by":"publisher","first-page":"6894","DOI":"10.1109\/CVPR52729.2023.00666","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y M Lyu","year":"2023","unstructured":"Y. M. Lyu, T. W. Lin, F. Li, D. L. He, J. Dong, T. N. Tan. Notice of removal: DeltaEdit: Exploring text-free training for text-driven image manipulation. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Vancouver, Canada, pp. 6894\u20136903, 2023. DOI: https:\/\/doi.org\/10.1109\/CVPR52729.2023.00666."},{"key":"1511_CR49","volume-title":"Representation learning with contrastive predictive coding","author":"A van den Oord","year":"2019","unstructured":"A. van den Oord, Y. Z. Li, O. Vinyals. Representation learning with contrastive predictive coding, [Online], Available: https:\/\/arxiv.org\/abs\/1807.03748, 2019."},{"key":"1511_CR50","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3455679","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","author":"P Bachman","year":"2019","unstructured":"P. Bachman, R. D. Hielm, W. Buchwalter. Learning representations by maximizing mutual information across views. In Proceedings of the 33rd International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 1392, 2019. DOI: https:\/\/doi.org\/10.5555\/3454287.3455679."},{"key":"1511_CR51","doi-asserted-by":"publisher","first-page":"776","DOI":"10.1007\/978-3-030-58621-8_45","volume-title":"Proceedings of the 16th European Conference on Computer Vision","author":"Y L Tian","year":"2020","unstructured":"Y. L. Tian, D. Krishnan, P. Isola. Contrastive multiview coding. In Proceedings of the 16th European Conference on Computer Vision, Glasgow, UK, pp. 776\u2013794, 2020. DOI: https:\/\/doi.org\/10.1007\/978-3-030-58621-8_45."},{"key":"1511_CR52","first-page":"297","volume-title":"Proceedings of the 13th International Conference on Artificial Intelligence and Statistics","author":"M Gutmann","year":"2010","unstructured":"M. Gutmann, Hyv\u00e4rinen. Noise-contrastive estimation: A new estimation principle for unnormalized statistical models. In Proceedings of the 13th International Conference on Artificial Intelligence and Statistics, Sardinia, Italy, pp. 297\u2013304, 2010."},{"key":"1511_CR53","doi-asserted-by":"publisher","first-page":"5153","DOI":"10.1109\/CVPR42600.2020.00520","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Y Deng","year":"2020","unstructured":"Y. Deng, J. L. Yang, D. Chen, F. Wen, X. Tong. Disentangled and controllable face image generation via 3D imitative-contrastive learning. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, USA, pp. 5153\u20135162, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00520."},{"key":"1511_CR54","doi-asserted-by":"publisher","first-page":"3622","DOI":"10.1109\/TIP.2023.3286710","volume":"32","author":"C Liu","year":"2023","unstructured":"C. Liu, Y. Q. Zhang, H. S. Wang, W. H. Chen, F. Wang, Y. Huang, Y. D. Shen, L. Wang. Efficient token-guided image-text retrieval with consistent multimodal contrastive training. IEEE Transactions on Image Processing, vol. 32, pp. 3622\u20133633, 2023. DOI: https:\/\/doi.org\/10.1109\/TIP.2023.3286710.","journal-title":"IEEE Transactions on Image Processing"},{"key":"1511_CR55","doi-asserted-by":"publisher","unstructured":"T. Karras, M. Aittala, J. Hellsten, S. Laine, J. Lehtinen, T. Aila. Training generative adversarial networks with limited data. In Proceedings of the 34th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 1015, 2020. DOI: https:\/\/doi.org\/10.5555\/3495724.3496739.","DOI":"10.5555\/3495724.3496739"},{"key":"1511_CR56","doi-asserted-by":"publisher","unstructured":"M. Caron, I. Misra, J. Mairal, P. Goyal, P. Boianowski, A. Joulin. Unsupervised learning of visual features by contrasting cluster assignments. In Proceedings of the 34th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 831, 2020. DOI: https:\/\/doi.org\/10.5555\/3495724.3496555.","DOI":"10.5555\/3495724.3496555"},{"key":"1511_CR57","first-page":"2292","volume-title":"Proceedings of the 26th International Conference on Neural Information Processing Systems","author":"M Cuturi","year":"2013","unstructured":"M. Cuturi. Sinkhorn distances: Lightspeed computation of optimal transport. In Proceedings of the 26th International Conference on Neural Information Processing Systems, Lake Tahoe, Nevada, pp. 2292\u20132300, 2013."},{"key":"1511_CR58","doi-asserted-by":"publisher","first-page":"8107","DOI":"10.1109\/CVPR42600.2020.00813","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"T Karras","year":"2020","unstructured":"T. Karras, S. Laine, M. Aittala, J. Hellsten, J. Lehtinen, T. Aila. Analyzing and improving the image quality of stylegan. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Seattle, USA, pp. 8107\u20138116, 2020. DOI: https:\/\/doi.org\/10.1109\/CVPR42600.2020.00813."},{"key":"1511_CR59","volume-title":"LSUN: Construction of a large-scale image dataset using deep learning with humans in the loop","author":"F Yu","year":"2016","unstructured":"F. Yu, A. Seff, Y. D. Zhang, S. R. Song, T. Funkhouser, J. X. Xiao. LSUN: Construction of a large-scale image dataset using deep learning with humans in the loop, [Online], Available: https:\/\/arxiv.org\/abs\/1506.03365, 2016."},{"issue":"11","key":"1511_CR60","doi-asserted-by":"publisher","first-page":"1955","DOI":"10.1109\/TPAMI.2008.222","volume":"31","author":"X G Wang","year":"2009","unstructured":"X. G. Wang, X. O. Tang. Face photo-sketch synthesis and recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, vol. 31, no. 11, pp. 1955\u20131967, 2009. DOI: https:\/\/doi.org\/10.1109\/TPAMI.2008.222.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1511_CR61","doi-asserted-by":"publisher","first-page":"6629","DOI":"10.5555\/3295222.3295408","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","author":"M Heusel","year":"2017","unstructured":"M. Heusel, H. Ramsauer, T. Unterthiner, B. Nessler, S. Hochreiter. GANs trained by a two time-scale update rule converge to a local nash equilibrium. In Proceedings of the 31st International Conference on Neural Information Processing Systems, Long Beach, USA, pp. 6629\u20136640, 2017. DOI: https:\/\/doi.org\/10.5555\/3295222.3295408."},{"key":"1511_CR62","doi-asserted-by":"publisher","first-page":"2234","DOI":"10.5555\/3157096.3157346","volume-title":"Proceedings of the 30th International Conference on Neural Information Processing Systems","author":"T Salimans","year":"2016","unstructured":"T. Salimans, I. Goodfellow, W. Zaremba, V. Cheung, A. Radford, X. Chen. Improved techniques for training GANs. In Proceedings of the 30th International Conference on Neural Information Processing Systems, Barcelona, Spain, pp. 2234\u20132242, 2016. DOI: https:\/\/doi.org\/10.5555\/3157096.3157346."},{"key":"1511_CR63","volume-title":"Proceedings of the 3rd International Conference on Learning Representations","author":"D P Kingma","year":"2015","unstructured":"D. P. Kingma, J. Ba. Adam: A method for stochastic optimization. In Proceedings of the 3rd International Conference on Learning Representations, San Diego, USA, 2015."},{"key":"1511_CR64","doi-asserted-by":"publisher","first-page":"586","DOI":"10.1109\/CVPR.2018.00068","volume-title":"Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"R Zhang","year":"2018","unstructured":"R. Zhang, P. Isola, A. A. Efros, E. Shechtman, O. Wang. The unreasonable effectiveness of deep features as a perceptual metric. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Salt Lake City, USA, pp. 586\u2013595, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00068."}],"container-title":["Machine Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-024-1511-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11633-024-1511-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-024-1511-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T10:08:46Z","timestamp":1757153326000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11633-024-1511-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,20]]},"references-count":64,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,8]]}},"alternative-id":["1511"],"URL":"https:\/\/doi.org\/10.1007\/s11633-024-1511-7","relation":{},"ISSN":["2731-538X","2731-5398"],"issn-type":[{"value":"2731-538X","type":"print"},{"value":"2731-5398","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,20]]},"assertion":[{"value":"6 February 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 April 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declared that they have no conflicts of interest to this work.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations of conflict of interest"}}]}}