{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:33:37Z","timestamp":1777656817096,"version":"3.51.4"},"reference-count":49,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T00:00:00Z","timestamp":1736812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T00:00:00Z","timestamp":1736812800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100003392","name":"Natural Science Foundation of Fujian Province","doi-asserted-by":"publisher","award":["2023J01351"],"award-info":[{"award-number":["2023J01351"]}],"id":[{"id":"10.13039\/501100003392","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s10489-024-06215-1","type":"journal-article","created":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T00:36:43Z","timestamp":1736815003000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["FineDiffusion: scaling up diffusion models for fine-grained image generation with 10,000 classes"],"prefix":"10.1007","volume":"55","author":[{"given":"Ziying","family":"Pan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feihong","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongxuan","family":"Lai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,14]]},"reference":[{"key":"6215_CR1","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho J, Jain A, Abbeel P (2020) Denoising diffusion probabilistic models. Adv Neural Inf Process Syst 33:6840\u20136851","journal-title":"Adv Neural Inf Process Syst"},{"key":"6215_CR2","unstructured":"Song J, Meng C, Ermon S (2020) Denoising diffusion implicit models. arXiv:2010.02502"},{"key":"6215_CR3","unstructured":"Nichol A, Dhariwal P, Ramesh A, Shyam P, Mishkin P, McGrew B, Sutskever I, Chen M (2021) Glide: Towards photorealistic image generation and editing with text-guided diffusion models. arXiv:2112.10741"},{"key":"6215_CR4","doi-asserted-by":"crossref","unstructured":"Rombach R, Blattmann A, Lorenz D, Esser P, Ommer B (2022) High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10684\u201310695","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"6215_CR5","unstructured":"Ramesh A, Dhariwal P, Nichol A, Chu C, Chen M (2022) Hierarchical text-conditional image generation with clip latents. arXiv:2204.06125"},{"key":"6215_CR6","first-page":"36479","volume":"35","author":"C Saharia","year":"2022","unstructured":"Saharia C, Chan W, Saxena S, Li L, Whang J, Denton EL, Ghasemipour K, Gontijo Lopes R, Karagol Ayan B, Salimans T et al (2022) Photorealistic text-to-image diffusion models with deep language understanding. Adv Neural Inf Process Syst 35:36479\u201336494","journal-title":"Adv Neural Inf Process Syst"},{"key":"6215_CR7","doi-asserted-by":"crossref","unstructured":"Peebles W, Xie S (2023) Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 4195\u20134205","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"6215_CR8","doi-asserted-by":"crossref","unstructured":"Gao S, Zhou P, Cheng M-M, Yan S (2023) Masked diffusion transformer is a strong image synthesizer. arXiv:2303.14389","DOI":"10.1109\/ICCV51070.2023.02117"},{"key":"6215_CR9","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2014) Generative adversarial nets. In: Advances in Neural Information Processing Systems, pp 2672\u20132680"},{"key":"6215_CR10","unstructured":"Kim D, Kim Y, Kang W, Moon I-C (2022) Refining generative process with discriminator guidance in score-based diffusion models. arXiv:2211.17091"},{"key":"6215_CR11","doi-asserted-by":"crossref","unstructured":"Xie E, Yao L, Shi H, Liu Z, Zhou D, Liu Z, Li J, Li Z (2023) Difffit: Unlocking transferability of large diffusion models via simple parameter-efficient fine-tuning. arXiv:2304.06648","DOI":"10.1109\/ICCV51070.2023.00390"},{"key":"6215_CR12","doi-asserted-by":"crossref","unstructured":"Van\u00a0Horn G, Mac\u00a0Aodha O, Song Y, Cui Y, Sun C, Shepard A, Adam H, Perona P, Belongie S (2018) The inaturalist species classification and detection dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 8769\u20138778","DOI":"10.1109\/CVPR.2018.00914"},{"key":"6215_CR13","unstructured":"Hu EJ, Shen Y, Wallis P, Allen-Zhu Z, Li Y, Wang S, Wang L, Chen W (2021) Lora: Low-rank adaptation of large language models. arXiv:2106.09685"},{"key":"6215_CR14","doi-asserted-by":"crossref","unstructured":"Ruiz N, Li Y, Jampani V, Pritch Y, Rubinstein M, Aberman K (2023) Dreambooth: Fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 22500\u201322510","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"6215_CR15","unstructured":"Gal R, Alaluf Y, Atzmon Y, Patashnik O, Bermano AH, Chechik G, Cohen-Or D (2022) An image is worth one word: Personalizing text-to-image generation using textual inversion. arXiv:2208.01618"},{"key":"6215_CR16","unstructured":"Zaken EB, Ravfogel S, Goldberg Y (2021) Bitfit: Simple parameter-efficient fine-tuning for transformer-based masked language-models. arXiv:2106.10199"},{"key":"6215_CR17","unstructured":"Sohl-Dickstein J, Weiss E, Maheswaranathan N, Ganguli S (2015) Deep unsupervised learning using nonequilibrium thermodynamics. In: International Conference on Machine Learning, pp 2256\u20132265"},{"key":"6215_CR18","unstructured":"Li G, Zheng H, Wang C, Li C, Zheng C, Tao D (2022) 3ddesigner: Towards photorealistic 3d object generation and editing with text-guided diffusion models. arXiv:2211.14108"},{"key":"6215_CR19","unstructured":"He F, Li G, Zhang M, Yan L, Si L, Li F (2024) Freestyle: Free lunch for text-guided style transfer using diffusion models. arXiv:2401.15636"},{"key":"6215_CR20","doi-asserted-by":"crossref","unstructured":"Liu X, Zhao Y, Wang S, Wei J (2024) Transdiff: medical image segmentation method based on swin transformer with diffusion probabilistic model. Applied Intell, 1\u201315","DOI":"10.1007\/s10489-024-05496-w"},{"issue":"18","key":"6215_CR21","doi-asserted-by":"publisher","first-page":"20979","DOI":"10.1007\/s10489-023-04559-8","volume":"53","author":"W Li","year":"2023","unstructured":"Li W, Xu W, Wu X, Wang Q, Lu Q, Song T, Li H (2023) Ammgan: adaptive multi-scale modulation generative adversarial network for few-shot image generation. Applied Intell 53(18):20979\u201320997","journal-title":"Applied Intell"},{"issue":"4","key":"6215_CR22","doi-asserted-by":"publisher","first-page":"4703","DOI":"10.1007\/s10489-022-03660-8","volume":"53","author":"Y Ma","year":"2023","unstructured":"Ma Y, Liu L, Zhang H, Wang C, Wang Z (2023) Generative adversarial network based on semantic consistency for text-to-image generation. Applied Intell 53(4):4703\u20134716","journal-title":"Applied Intell"},{"key":"6215_CR23","unstructured":"Kingma DP, Welling M (2014) Auto-encoding variational bayes. In: International Conference on Learning Representations (ICLR)"},{"key":"6215_CR24","unstructured":"Dinh L, Krueger D, Bengio Y (2015) Nice: Non-linear independent components estimation. In: International Conference on Learning Representations (ICLR)"},{"key":"6215_CR25","doi-asserted-by":"crossref","unstructured":"Ronneberger O, Fischer P, Brox T (2015) U-net: Convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-Assisted Intervention\u2013MICCAI 2015: 18th International Conference, Munich, Germany, 5-9 October, 2015, Proceedings, Part III 18, pp 234\u2013241 . Springer","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"6215_CR26","unstructured":"Ho J, Salimans T (2022) Classifier-free diffusion guidance. arXiv:2207.12598"},{"key":"6215_CR27","doi-asserted-by":"crossref","unstructured":"Gafni O, Polyak A, Ashual O, Sheynin S, Parikh D, Taigman Y (2022) Make-a-scene: Scene-based text-to-image generation with human priors. In: European Conference on Computer Vision, pp 89\u2013106 . Springer","DOI":"10.1007\/978-3-031-19784-0_6"},{"key":"6215_CR28","doi-asserted-by":"crossref","unstructured":"Mokady R, Hertz A, Aberman K, Pritch Y, Cohen-Or D (2023) Null-text inversion for editing real images using guided diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 6038\u20136047","DOI":"10.1109\/CVPR52729.2023.00585"},{"key":"6215_CR29","doi-asserted-by":"crossref","unstructured":"Bao F, Nie S, Xue K, Cao Y, Li C, Su H, Zhu J (2023) All are worth words: A vit backbone for diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 22669\u201322679","DOI":"10.1109\/CVPR52729.2023.02171"},{"key":"6215_CR30","unstructured":"Mo S, Xie E, Chu R, Hong L, Niessner M, Li Z (2024) Dit-3d: Exploring plain diffusion transformers for 3d shape generation. Adv Neural Inf Process Syst 36"},{"key":"6215_CR31","doi-asserted-by":"crossref","unstructured":"He F, Li G, Si L, Yan L, Hou S, Dong H, Li F (2023) Cartoondiff: Training-free cartoon image generation with diffusion transformer models. arXiv:2309.08251","DOI":"10.1109\/ICASSP48485.2024.10447821"},{"key":"6215_CR32","unstructured":"Lu Z, Wang Z, Huang D, Wu C, Liu X, Ouyang W, Bai L (2024) Fit: Flexible vision transformer for diffusion model. arXiv:2402.12376"},{"key":"6215_CR33","doi-asserted-by":"crossref","unstructured":"Howard J, Ruder S (2018) Universal language model fine-tuning for text classification. arXiv:1801.06146","DOI":"10.18653\/v1\/P18-1031"},{"key":"6215_CR34","unstructured":"Dai AM, Le QV (2015) Semi-supervised sequence learning. Adv Neural Inf Process Syst 28"},{"key":"6215_CR35","unstructured":"Houlsby N, Giurgiu A, Jastrzebski S, Morrone B, De\u00a0Laroussilhe Q, Gesmundo A, Attariyan M, Gelly S (2019) Parameter-efficient transfer learning for nlp. In: International Conference on Machine Learning, pp 2790\u20132799. PMLR"},{"key":"6215_CR36","first-page":"16664","volume":"35","author":"S Chen","year":"2022","unstructured":"Chen S, Ge C, Tong Z, Wang J, Song Y, Wang J, Luo P (2022) Adaptformer: Adapting vision transformers for scalable visual recognition. Adv Neural Inf Process Syst 35:16664\u201316678","journal-title":"Adv Neural Inf Process Syst"},{"key":"6215_CR37","doi-asserted-by":"crossref","unstructured":"He S, Ding L, Dong D, Zhang M, Tao D (2022) Sparseadapter: An easy approach for improving the parameter-efficiency of adapters. arXiv:2210.04284","DOI":"10.18653\/v1\/2022.findings-emnlp.160"},{"issue":"9","key":"6215_CR38","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou K, Yang J, Loy CC, Liu Z (2022) Learning to prompt for vision-language models. Int J Comput Vis 130(9):2337\u20132348","journal-title":"Int J Comput Vis"},{"key":"6215_CR39","doi-asserted-by":"crossref","unstructured":"Yao Y, Zhang A, Zhang Z, Liu Z, Chua T-S, Sun M (2021) Cpt: Colorful prompt tuning for pre-trained vision-language models. arXiv:2109.11797","DOI":"10.18653\/v1\/2022.findings-acl.273"},{"key":"6215_CR40","doi-asserted-by":"crossref","unstructured":"Jia M, Tang L, Chen B-C, Cardie C, Belongie S, Hariharan B, Lim S-N (2022) Visual prompt tuning. In: European Conference on Computer Vision, pp 709\u2013727. Springer","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"6215_CR41","doi-asserted-by":"crossref","unstructured":"Rao Y, Zhao W, Chen G, Tang Y, Zhu Z, Huang G, Zhou J, Lu J (2022) Denseclip: Language-guided dense prediction with context-aware prompting. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18082\u201318091","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"6215_CR42","doi-asserted-by":"crossref","unstructured":"Zhao M, Lin T, Mi F, Jaggi M, Sch\u00fctze H (2020) Masking as an efficient alternative to finetuning for pretrained language models. arXiv:2004.12406","DOI":"10.18653\/v1\/2020.emnlp-main.174"},{"key":"6215_CR43","doi-asserted-by":"crossref","unstructured":"Xu R, Luo F, Zhang Z, Tan C, Chang B, Huang S, Huang F (2021) Raise a child in large language model: Towards effective and generalizable fine-tuning. arXiv:2109.05687","DOI":"10.18653\/v1\/2021.emnlp-main.749"},{"key":"6215_CR44","doi-asserted-by":"crossref","unstructured":"Hou S, Feng Y, Wang Z (2017) Vegfru: A domain-specific dataset for fine-grained visual categorization. In: Proceedings of the IEEE International Conference on Computer Vision, pp 541\u2013549","DOI":"10.1109\/ICCV.2017.66"},{"key":"6215_CR45","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv:1711.05101"},{"key":"6215_CR46","unstructured":"Heusel M, Ramsauer H, Unterthiner T, Nessler B, Hochreiter S (2017) Gans trained by a two time-scale update rule converge to a local nash equilibrium. Adv Neural Inf Process Syst 30"},{"key":"6215_CR47","doi-asserted-by":"crossref","unstructured":"Zhang R, Isola P, Efros AA, Shechtman E, Wang O (2018) The unreasonable effectiveness of deep features as a perceptual metric. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 586\u2013595","DOI":"10.1109\/CVPR.2018.00068"},{"key":"6215_CR48","unstructured":"Salimans T, Goodfellow I, Zaremba W, Cheung V, Radford A, Chen X (2016) Improved techniques for training gans. Adv Neural Inf Process Syst 29"},{"key":"6215_CR49","unstructured":"Maaten L, Hinton G (2008) Visualizing data using t-sne. J Mach Learn Res 9(11)"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-06215-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-06215-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-06215-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,22]],"date-time":"2025-02-22T17:20:29Z","timestamp":1740244829000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-06215-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,14]]},"references-count":49,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["6215"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-06215-1","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,14]]},"assertion":[{"value":"16 December 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 January 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics Approval and Consent to Participate"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}}],"article-number":"309"}}