{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T05:21:30Z","timestamp":1783056090636,"version":"3.54.6"},"reference-count":86,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62337001"],"award-info":[{"award-number":["62337001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U25A20440"],"award-info":[{"award-number":["U25A20440"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["2025C02022"],"award-info":[{"award-number":["2025C02022"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.eswa.2026.133019","type":"journal-article","created":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T16:34:41Z","timestamp":1780331681000},"page":"133019","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Context-aware latent space mediation for inference-time unbiased semantic alignment in text-to-image models"],"prefix":"10.1016","volume":"330","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0794-1204","authenticated-orcid":false,"given":"Jili","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5128-6692","authenticated-orcid":false,"given":"Huicheng","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1371-2608","authenticated-orcid":false,"given":"Changqin","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5041-6093","authenticated-orcid":false,"given":"Qionghao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2853-531X","authenticated-orcid":false,"given":"Zilong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6084-1851","authenticated-orcid":false,"given":"Xiaodi","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133019_bib0001","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"2283","article-title":"A-star: Test-time attention segregation and retention for text-to-image synthesis","author":"Agarwal","year":"2023"},{"key":"10.1016\/j.eswa.2026.133019_bib0002","unstructured":"Bai, L., Sugiyama, M., & Xie, Z. (2025). Weak-to-strong diffusion with reflection. arXiv: 2502.00473."},{"key":"10.1016\/j.eswa.2026.133019_bib0003","unstructured":"Balaji, Y., Nah, S., Huang, X., Vahdat, A., Song, J., Zhang, Q., Kreis, K., Aittala, M., Aila, T., Laine, S. et al. (2022). ediff-i: Text-to-image diffusion models with an ensemble of expert denoisers. arXiv: 2211.01324."},{"issue":"3","key":"10.1016\/j.eswa.2026.133019_bib0004","first-page":"8","article-title":"Improving image generation with better captions","volume":"2","author":"Betker","year":"2023","journal-title":"Computer Science"},{"key":"10.1016\/j.eswa.2026.133019_bib0005","unstructured":"Black, K., Janner, M., Du, Y., Kostrikov, I., & Levine, S. (2023). Training diffusion models with reinforcement learning. arXiv: 2305.13301."},{"issue":"4","key":"10.1016\/j.eswa.2026.133019_bib0006","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3592116","article-title":"Attend-and-excite: Attention-based semantic guidance for text-to-image diffusion models","volume":"42","author":"Chefer","year":"2023","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"10.1016\/j.eswa.2026.133019_bib0007","unstructured":"Chen, C.-S., & Kuo, E.-J. (2025). Quantum reinforcement learning-guided diffusion model for image synthesis via hybrid quantum-classical generative model architectures. arXiv: 2509.14163."},{"key":"10.1016\/j.eswa.2026.133019_bib0008","doi-asserted-by":"crossref","first-page":"57944","DOI":"10.52202\/079017-1847","article-title":"A cat is a cat (not a dog!): Unraveling information mix-ups in text-to-image encoders through causal analysis and embedding optimization","volume":"37","author":"Chen","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0009","unstructured":"Chen, J., Yu, J., Ge, C., Yao, L., Xie, E., Wu, Y., Wang, Z., Kwok, J., Luo, P., Lu, H. et al. (2023). Pixart-alpha: Fast training of diffusion transformer for photorealistic text-to-image synthesis. arXiv: 2310.00426."},{"key":"10.1016\/j.eswa.2026.133019_bib0010","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6146","article-title":"Multimodal representation alignment for image generation: Text-image interleaved control is easier than you think","author":"Chen","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0011","unstructured":"Clark, K., Vicol, P., Swersky, K., & Fleet, D. J. (2023). Directly fine-tuning diffusion models on differentiable rewards. arXiv: 2309.17400."},{"key":"10.1016\/j.eswa.2026.133019_bib0012","unstructured":"Dong, H., Xiong, W., Goyal, D., Zhang, Y., Chow, W., Pan, R., Diao, S., Zhang, J., Shum, K., & Zhang, T. (2023). Raft: Reward ranked finetuning for generative foundation model alignment. arXiv: 2304.06767."},{"key":"10.1016\/j.eswa.2026.133019_bib0013","doi-asserted-by":"crossref","first-page":"58648","DOI":"10.52202\/075280-2556","article-title":"Stable diffusion is unstable","volume":"36","author":"Du","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0014","series-title":"Forty-first international conference on machine learning","article-title":"Scaling rectified flow transformers for high-resolution image synthesis","author":"Esser","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0015","series-title":"The thirteenth international conference on learning representations","article-title":"Online reward-weighted fine-tuning of flow matching with wasserstein regularization","author":"Fan","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0016","doi-asserted-by":"crossref","first-page":"79858","DOI":"10.52202\/075280-3497","article-title":"DPOK: Reinforcement learning for fine-tuning text-to-image diffusion models","volume":"36","author":"Fan","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0017","unstructured":"Feng, W., He, X., Fu, T.-J., Jampani, V., Akula, A., Narayana, P., Basu, S., Wang, X. E., & Wang, W. Y. (2022). Training-free structured diffusion guidance for compositional text-to-image synthesis. arXiv: 2212.05032."},{"key":"10.1016\/j.eswa.2026.133019_bib0018","doi-asserted-by":"crossref","first-page":"52132","DOI":"10.52202\/075280-2270","article-title":"Geneval: An object-focused framework for evaluating text-to-image alignment","volume":"36","author":"Ghosh","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0019","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"678","article-title":"ShortFT: Diffusion model alignment via shortcut-based fine-tuning","author":"Guo","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0020","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9380","article-title":"Initno: Boosting text-to-image diffusion models via initial noise optimization","author":"Guo","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0021","unstructured":"He, H., Ye, Y., Liu, J., Liang, J., Wang, Z., Yuan, Z., Wang, X., Mao, H., Wan, P., & Pan, L. (2025a). Gardo: Reinforcing diffusion models without reward hacking. arXiv: 2512.24138."},{"key":"10.1016\/j.eswa.2026.133019_bib0022","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"17123","article-title":"Mars: Mixture of auto-regressive models for fine-grained text-to-image synthesis","volume":"vol. 39","author":"He","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0023","unstructured":"He, X., Fu, S., Zhao, Y., Li, W., Yang, J., Yin, D., Rao, F., & Zhang, B. (2025c). Tempflow-GRPO: When timing matters for GRPO in flow models. arXiv: 2508.04324."},{"key":"10.1016\/j.eswa.2026.133019_bib0024","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., & Cohen-Or, D. (2022). Prompt-to-prompt image editing with cross attention control. arXiv: 2208.01626."},{"key":"10.1016\/j.eswa.2026.133019_bib0025","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0026","unstructured":"Ho, J., & Salimans, T. (2022). Classifier-free diffusion guidance. arXiv: 2207.12598."},{"key":"10.1016\/j.eswa.2026.133019_bib0027","series-title":"Conference on empirical methods in natural language processing, EMNLP 2015","first-page":"1373","article-title":"An improved non-monotonic transition system for dependency parsing","author":"Honnibal","year":"2015"},{"key":"10.1016\/j.eswa.2026.133019_bib0028","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"23604","article-title":"Towards better alignment: Training diffusion models with reinforcement learning against sparse rewards","author":"Hu","year":"2025"},{"issue":"5","key":"10.1016\/j.eswa.2026.133019_bib0029","doi-asserted-by":"crossref","first-page":"3563","DOI":"10.1109\/TPAMI.2025.3531907","article-title":"T2i-compbench++: An enhanced and comprehensive benchmark for compositional text-to-image generation","volume":"47","author":"Huang","year":"2025","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133019_bib0030","doi-asserted-by":"crossref","first-page":"78723","DOI":"10.52202\/075280-3443","article-title":"T2i-compbench: A comprehensive benchmark for open-world compositional text-to-image generation","volume":"36","author":"Huang","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0031","first-page":"18661","article-title":"Supervised contrastive learning","volume":"33","author":"Khosla","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0032","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"8031","article-title":"Text embedding is not all you need: Attention control for text-to-image semantic alignment with text self-attention maps","author":"Kim","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0033","unstructured":"Kingma, D. P., Welling, M. (2013). Auto-encoding variational Bayes. arXiv: 1312.6114."},{"key":"10.1016\/j.eswa.2026.133019_bib0034","doi-asserted-by":"crossref","first-page":"36652","DOI":"10.52202\/075280-1594","article-title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation","volume":"36","author":"Kirstain","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0035","unstructured":"Lee, K., Liu, H., Ryu, M., Watkins, O., Du, Y., Boutilier, C., Abbeel, P., Ghavamzadeh, M., & Gu, S. S. (2023). Aligning text-to-image models using human feedback. arXiv: 2302.12192."},{"key":"10.1016\/j.eswa.2026.133019_bib0036","series-title":"International conference on machine learning","first-page":"12888","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"10.1016\/j.eswa.2026.133019_bib0037","unstructured":"Li, Y., Keuper, M., Zhang, D., & Khoreva, A. (2023). Divide & bind your attention for improved generative semantic nursing. arXiv: 2307.10864."},{"key":"10.1016\/j.eswa.2026.133019_bib0038","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"13199","article-title":"Aesthetic post-training diffusion models from generic preferences with step-by-step preference optimization","author":"Liang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0039","unstructured":"Liu, J., Liu, G., Liang, J., Li, Y., Liu, J., Wang, X., Wan, P., Zhang, D., & Ouyang, W. (2025a). Flow-GRPO: Training flow matching models via online RL. arXiv: 2505.05470."},{"key":"10.1016\/j.eswa.2026.133019_bib0040","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"5523","article-title":"LLM4GEN: Leveraging semantic representation of LLMs for text-to-image generation","volume":"vol. 39","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0041","series-title":"European conference on computer vision","first-page":"423","article-title":"Compositional visual generation with composable diffusion models","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133019_bib0042","unstructured":"Luo, Y., Du, P., Li, B., Du, S., Zhang, T., Chang, Y., Wu, K., Gai, K., & Wang, X. (2025a). Sample by step, optimize by chunk: Chunk-level GRPO for text-to-image generation. arXiv: 2510.21583."},{"key":"10.1016\/j.eswa.2026.133019_bib0043","unstructured":"Luo, Y., Hu, X., Fan, K., Sun, H., Chen, Z., Xia, B., Zhang, T., Chang, Y., & Wang, X. (2025b). Reinforcement learning meets masked generative models: Mask-GRPO for text-to-image generation. arXiv: 2510.13418."},{"key":"10.1016\/j.eswa.2026.133019_bib0044","doi-asserted-by":"crossref","unstructured":"Ma, N., Tong, S., Jia, H., Hu, H., Su, Y.-C., Zhang, M., Yang, X., Li, Y., Jaakkola, T., Jia, X. et al. (2025a). Inference-time scaling for diffusion models beyond scaling denoising steps. arXiv: 2501.09732.","DOI":"10.1109\/CVPR52734.2025.00241"},{"key":"10.1016\/j.eswa.2026.133019_bib0045","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"7739","article-title":"Janusflow: Harmonizing autoregression and rectified flow for unified multimodal understanding and generation","author":"Ma","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0046","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9005","article-title":"Conform: Contrast is all you need for high-fidelity text-to-image diffusion models","author":"Meral","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0047","series-title":"2012 IEEE conference on computer vision and pattern recognition","first-page":"2408","article-title":"Ava: A large-scale database for aesthetic visual analysis","author":"Murray","year":"2012"},{"key":"10.1016\/j.eswa.2026.133019_bib0048","series-title":"Proceedings of the 2nd conference of the Asia-Pacific chapter of the association for computational linguistics and the 12th international joint conference on natural language processing: System demonstrations","first-page":"48","article-title":"F-coref: Fast, accurate and easy to use coreference resolution","author":"Otmazgin","year":"2022"},{"key":"10.1016\/j.eswa.2026.133019_bib0049","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"4195","article-title":"Scalable diffusion models with transformers","author":"Peebles","year":"2023"},{"key":"10.1016\/j.eswa.2026.133019_bib0050","series-title":"Proceedings of the 24th international conference on machine learning","first-page":"745","article-title":"Reinforcement learning by reward-weighted regression for operational space control","author":"Peters","year":"2007"},{"key":"10.1016\/j.eswa.2026.133019_bib0051","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J., & Rombach, R. (2023). SDXL: Improving latent diffusion models for high-resolution image synthesis. arXiv: 2307.01952."},{"key":"10.1016\/j.eswa.2026.133019_bib0052","unstructured":"Prabhudesai, M., Goyal, A., Pathak, D., & Fragkiadaki, K. (2023). Aligning text-to-image diffusion models with reward backpropagation. arXiv: 2310.03739."},{"key":"10.1016\/j.eswa.2026.133019_bib0053","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.133019_bib0054","doi-asserted-by":"crossref","first-page":"53728","DOI":"10.52202\/075280-2338","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume":"36","author":"Rafailov","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0055","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., & Chen, M. (2022). Hierarchical text-conditional image generation with clip latents. arXiv: 2204.06125, 1(2), 3."},{"key":"10.1016\/j.eswa.2026.133019_bib0056","doi-asserted-by":"crossref","first-page":"3536","DOI":"10.52202\/075280-0157","article-title":"Linguistic binding in diffusion models: Enhancing attribute correspondence through attention map alignment","volume":"36","author":"Rassin","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0057","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10684","article-title":"High-resolution image synthesis with latent diffusion models","author":"Rombach","year":"2022"},{"key":"10.1016\/j.eswa.2026.133019_bib0058","doi-asserted-by":"crossref","first-page":"36479","DOI":"10.52202\/068431-2643","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0059","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms. arXiv: 1707.06347."},{"key":"10.1016\/j.eswa.2026.133019_bib0060","unstructured":"Shen, X., Li, Z., Yang, Z., Zhang, S., Zhang, Y., Li, D., Wang, C., Lu, Q., & Tang, Y. (2025). Directly aligning the full diffusion trajectory with fine-grained human preference. arXiv: 2509.06942."},{"key":"10.1016\/j.eswa.2026.133019_bib0061","unstructured":"Song, J., Meng, C., & Ermon, S. (2020). Denoising diffusion implicit models. arXiv: 2010.02502."},{"key":"10.1016\/j.eswa.2026.133019_sbref0062","article-title":"Diffusion-rainbowpa: Improvements integrated preference alignment for diffusion-based text-to-image generation","author":"Sun","year":"2025","journal-title":"Transactions on Machine Learning Research"},{"key":"10.1016\/j.eswa.2026.133019_bib0063","unstructured":"Sun, H., Wu, J., Xia, B., Luo, Y., Zhao, Y., Qin, K., Lv, X., Zhang, T., Chang, Y., & Wang, X. (2025b). Reinforcement fine-tuning powers reasoning capability of multimodal large language models. arXiv: 2505.18536."},{"key":"10.1016\/j.eswa.2026.133019_bib0064","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"27644","article-title":"Generalizing alignment paradigm of text-to-image generation with preferences through f-divergence minimization","volume":"vol. 39","author":"Sun","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0065","series-title":"Icassp 2025-2025 IEEE international conference on acoustics, speech and signal processing (ICASSP)","first-page":"1","article-title":"Identical human preference alignment paradigm for text-to-image models","author":"Sun","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0066","series-title":"Icassp 2025-2025 IEEE international conference on acoustics, speech and signal processing (ICASSP)","first-page":"1","article-title":"Positive enhanced preference alignment for text-to-image models","author":"Sun","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0067","unstructured":"Tang, Z., Peng, J., Tang, J., Hong, M., Wang, F., & Chang, T.-H. (2024). Inference-time alignment of diffusion models with direct noise optimization. arXiv: 2405.18881."},{"key":"10.1016\/j.eswa.2026.133019_bib0068","unstructured":"Tian, Y., Xia, X., Ren, Y., Lin, S., Wang, X., Xiao, X., Tong, Y., Yang, L., & Cui, B. (2025). Training-free diffusion acceleration with bottleneck sampling. arXiv: 2503.18940."},{"key":"10.1016\/j.eswa.2026.133019_bib0069","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8228","article-title":"Diffusion model alignment using direct preference optimization","author":"Wallace","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0070","unstructured":"Wang, J., Liang, J., Liu, J., Liu, H., Liu, G., Zheng, J., Pang, W., Ma, A., Xie, Z., Wang, X. et al. (2025a). GRPO-guard: Mitigating implicit over-optimization in flow matching via regulated clipping. arXiv: 2510.22319."},{"key":"10.1016\/j.eswa.2026.133019_bib0071","unstructured":"Wang, X., Zhang, X., Luo, Z., Sun, Q., Cui, Y., Wang, J., Zhang, F., Wang, Y., Li, Z., Yu, Q. et al. (2024). Emu3: Next-token prediction is all you need. arXiv: 2409.18869."},{"key":"10.1016\/j.eswa.2026.133019_bib0072","unstructured":"Wang, Y., Li, Z., Zang, Y., Zhou, Y., Bu, J., Wang, C., Lu, Q., Jin, C., & Wang, J. (2025b). Pref-GRPO: Pairwise preference reward-based grpo for stable text-to-image reinforcement learning. arXiv: 2508.20751."},{"key":"10.1016\/j.eswa.2026.133019_bib0073","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"12966","article-title":"Janus: Decoupling visual encoding for unified multimodal understanding and generation","author":"Wu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0074","unstructured":"Wu, X., Hao, Y., Sun, K., Chen, Y., Zhu, F., Zhao, R., & Li, H. (2023). Human preference score v2: A solid benchmark for evaluating human preferences of text-to-image synthesis. arXiv: 2306.09341."},{"key":"10.1016\/j.eswa.2026.133019_bib0075","series-title":"European conference on computer vision","first-page":"108","article-title":"Deep reward supervisions for tuning text-to-image diffusion models","author":"Wu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0076","unstructured":"Xie, J., Mao, W., Bai, Z., Zhang, D. J., Wang, W., Lin, K. Q., Gu, Y., Chen, Z., Yang, Z., & Shou, M. Z. (2024). Show-o: One single transformer to unify multimodal understanding and generation. arXiv: 2408.12528."},{"key":"10.1016\/j.eswa.2026.133019_bib0077","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"13220","article-title":"DyMO: Training-free diffusion model alignment with dynamic multi-objective scheduling","author":"Xie","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0078","doi-asserted-by":"crossref","first-page":"15903","DOI":"10.52202\/075280-0700","article-title":"ImageReward: Learning and evaluating human preferences for text-to-image generation","volume":"36","author":"Xu","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133019_bib0079","unstructured":"Xue, Z., Wu, J., Gao, Y., Kong, F., Zhu, L., Chen, M., Liu, Z., Liu, W., Guo, Q., Huang, W. et al. (2025). DanceGRPO: Unleashing GRPO on visual generation. arXiv: 2505.07818."},{"key":"10.1016\/j.eswa.2026.133019_bib0080","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8941","article-title":"Using human feedback to fine-tune diffusion models without any reward model","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0081","unstructured":"Zhai, K., Singh, U., Thatipelli, A., Chakraborty, S., Sahu, A. K., Huang, F., Bedi, A. S., & Shah, M. (2025). Mira: Towards mitigating reward hacking in inference-time alignment of T2I diffusion models. arXiv: 2510.01549."},{"key":"10.1016\/j.eswa.2026.133019_bib0082","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3836","article-title":"Adding conditional control to text-to-image diffusion models","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133019_bib0083","series-title":"European conference on computer vision","first-page":"1","article-title":"Large-scale reinforcement learning for diffusion models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0084","series-title":"European conference on computer vision","first-page":"55","article-title":"Object-conditioned energy-based attention map alignment in text-to-image diffusion models","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133019_bib0085","series-title":"The thirty-ninth annual conference on neural information processing systems","article-title":"Enhancing text-to-image diffusion transformer via split-text conditioning","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133019_bib0086","series-title":"The thirteenth international conference on learning representations","article-title":"DSPO: Direct score preference optimization for diffusion model alignment","author":"Zhu","year":"2025"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426019305?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426019305?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T05:05:10Z","timestamp":1783055110000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426019305"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":86,"alternative-id":["S0957417426019305"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133019","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Context-aware latent space mediation for inference-time unbiased semantic alignment in text-to-image models","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133019","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133019"}}