{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T06:46:02Z","timestamp":1785653162780,"version":"3.56.0"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032316653","type":"print"},{"value":"9783032316660","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T00:00:00Z","timestamp":1785715200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-31666-0_29","type":"book-chapter","created":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:47:16Z","timestamp":1785649636000},"page":"437-451","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Unique Step Refinement for\u00a0Transformer-Based Generative Models"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-5116-9417","authenticated-orcid":false,"given":"Paul","family":"Grimal","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0520-8436","authenticated-orcid":false,"given":"Herv\u00e9","family":"Le Borgne","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0755-2361","authenticated-orcid":false,"given":"Olivier","family":"Ferret","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,3]]},"reference":[{"key":"29_CR1","doi-asserted-by":"crossref","unstructured":"Agarwal, A., Karanam, S., Joseph, K.J., Saxena, A., Goswami, K., Srinivasan, B.V.: A-STAR: test-time attention segregation and retention for text-to-image synthesis. In: IEEE\/CVF International Conference on Computer Vision (2023)","DOI":"10.1109\/ICCV51070.2023.00217"},{"key":"29_CR2","doi-asserted-by":"crossref","unstructured":"Chefer, H., Alaluf, Y., Vinker, Y., Wolf, L., Cohen-Or, D.: Attend-and-excite: attention-based semantic guidance for text-to-image diffusion models. ACM Trans. Graph 42(4) (2023)","DOI":"10.1145\/3592116"},{"key":"29_CR3","doi-asserted-by":"publisher","unstructured":"Chen, J., et al.: Pixart-$$\\sigma $$: weak-to-strong training of diffusion transformer for 4k text-to-image generation. In: European Conference on Computer Vision. Springer, Heidelberg (2024). https:\/\/doi.org\/10.1007\/978-3-031-73411-3_5","DOI":"10.1007\/978-3-031-73411-3_5"},{"key":"29_CR4","doi-asserted-by":"crossref","unstructured":"Choi, J., Lee, J., Shin, C., Kim, S., Kim, H., Yoon, S.: Perception prioritized training of diffusion models. In: CVPR, pp. 11472\u201311481 (2022)","DOI":"10.1109\/CVPR52688.2022.01118"},{"key":"29_CR5","unstructured":"Esser, P., et al.: Scaling rectified flow transformers for high-resolution image synthesis. In: Proceedings of the 41st International Conference on Machine Learning (2024)"},{"key":"29_CR6","doi-asserted-by":"crossref","unstructured":"Everaert, M.N., Fitsios, A., Bocchio, M., Arpa, S., S\u00fcsstrunk, S., Achanta, R.: Exploiting the signal-leak bias in diffusion models. In: WACV (2024)","DOI":"10.1109\/WACV57701.2024.00398"},{"key":"29_CR7","doi-asserted-by":"crossref","unstructured":"Ghosh, D., Hajishirzi, H., Schmidt, L.: Geneval: an object-focused framework for evaluating text-to-image alignment. In: Thirty-Seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track (2023)","DOI":"10.52202\/075280-2270"},{"key":"29_CR8","doi-asserted-by":"crossref","unstructured":"Grimal, P., Le Borgne, H., Ferret, O., Tourille, J.: TIAM - a metric for evaluating alignment in text-to-image generation. In: WACV, pp. 2890\u20132899 (2024)","DOI":"10.1109\/WACV57701.2024.00287"},{"key":"29_CR9","doi-asserted-by":"crossref","unstructured":"Guo, X., Liu, J., Cui, M., Li, J., Yang, H., Huang, D.: InitNO: boosting text-to-image diffusion models via initial noise optimization. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.00896"},{"key":"29_CR10","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: Advances in Neural Information Processing Systems, vol. 33 (2020)"},{"key":"29_CR11","unstructured":"Ho, J., Salimans, T.: Classifier-free diffusion guidance. In: NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications (2021)"},{"key":"29_CR12","unstructured":"Labs, B.F.: Flux (2024). https:\/\/github.com\/black-forest-labs\/flux"},{"key":"29_CR13","unstructured":"Li, Y., Keuper, M., Zhang, D., Khoreva, A.: Divide & bind your attention for improved generative semantic nursing. In: BMVC (2023)"},{"key":"29_CR14","doi-asserted-by":"crossref","unstructured":"Lin, S., Liu, B., Li, J., Yang, X.: Common diffusion noise schedules and sample steps are flawed. In: WACV, pp, 5404\u20135411 (2024)","DOI":"10.1109\/WACV57701.2024.00532"},{"key":"29_CR15","doi-asserted-by":"crossref","unstructured":"Lin, Z., et al.: Evaluating text-to-visual generation with image-to-text generation. In: 18th European Conference on Computer Vision (ECCV 2024), pp. 366\u2013384 (2024)","DOI":"10.1007\/978-3-031-72673-6_20"},{"key":"29_CR16","unstructured":"Lipman, Y., Chen, R.T.Q., Ben-Hamu, H., Nickel, M., Le, M.: Flow matching for generative modeling. In: ICLR (2023)"},{"key":"29_CR17","doi-asserted-by":"crossref","unstructured":"Park, Y.H., Kwon, M., Choi, J., Jo, J., Uh, Y.: Understanding the latent space of diffusion models through the lens of Riemannian geometry. In: Thirty-Seventh Conference on Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-1048"},{"key":"29_CR18","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"29_CR19","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, vol. 139, pp. 8748\u20138763 (2021)"},{"key":"29_CR20","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(1) (2020)"},{"key":"29_CR21","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical Text-Conditional Image Generation with CLIP Latents. arXiv:2204.06125 (2022)"},{"key":"29_CR22","doi-asserted-by":"crossref","unstructured":"Rassin, R., Hirsch, E., Glickman, D., Ravfogel, S., Goldberg, Y., Chechik, G.: Linguistic binding in diffusion models: enhancing attribute correspondence through attention map alignment. In: NeurIPS (2023)","DOI":"10.52202\/075280-0157"},{"key":"29_CR23","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"29_CR24","doi-asserted-by":"crossref","unstructured":"Saharia, C., et al.: Photorealistic text-to-image diffusion models with deep language understanding. In: Advances in Neural Information Processing Systems (2022)","DOI":"10.52202\/068431-2643"},{"key":"29_CR25","doi-asserted-by":"crossref","unstructured":"Schuhmann, C., et al.: LAION-5B: an open large-scale dataset for training next generation image-text models. In: Advances in Neural Information Processing Systems, vol. 35, pp. 25278\u201325294 (2022)","DOI":"10.52202\/068431-1833"},{"key":"29_CR26","doi-asserted-by":"crossref","unstructured":"Tang, R., et al.: What the DAAM: interpreting stable diffusion using cross attention. In: Proceedings of the Association for Computational Linguistics (2023)","DOI":"10.18653\/v1\/2023.acl-long.310"},{"key":"29_CR27","doi-asserted-by":"crossref","unstructured":"Xie, J., et al.: Boxdiff: text-to-image synthesis with training-free box-constrained diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2023)","DOI":"10.1109\/ICCV51070.2023.00685"},{"key":"29_CR28","unstructured":"Yuksekgonul, M., Bianchi, F., Kalluri, P., Jurafsky, D., Zou, J.: When and why vision-language models behave like bags-of-words, and what to do about it? In: The Eleventh International Conference on Learning Representations (2023)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-31666-0_29","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T05:47:19Z","timestamp":1785649639000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-31666-0_29"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,3]]},"ISBN":["9783032316653","9783032316660"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-31666-0_29","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,3]]},"assertion":[{"value":"3 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lyon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 August 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}