{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T14:55:53Z","timestamp":1761404153906,"version":"build-2065373602"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746262.3761978","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T14:52:23Z","timestamp":1761403943000},"page":"30-38","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Text-to-Image Generation Post-Training with Pixel-Space Loss"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5126-7386","authenticated-orcid":false,"given":"Christina","family":"Zhang","sequence":"first","affiliation":[{"name":"Princeton University, Princeton, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5726-5267","authenticated-orcid":false,"given":"Simran","family":"Motwani","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4172-8041","authenticated-orcid":false,"given":"Matthew","family":"Yu","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5244-8953","authenticated-orcid":false,"given":"Ji","family":"Hou","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0857-8611","authenticated-orcid":false,"given":"Felix","family":"Juefei-Xu","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7220-9108","authenticated-orcid":false,"given":"Sam","family":"Tsai","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2031-4678","authenticated-orcid":false,"given":"Peter","family":"Vajda","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8426-3080","authenticated-orcid":false,"given":"Zijian","family":"He","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1148-9963","authenticated-orcid":false,"given":"Jialiang","family":"Wang","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,26]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","unstructured":"Yogesh Balaji Seungjun Nah Xun Huang Arash Vahdat Jiaming Song Qinsheng Zhang Karsten Kreis Miika Aittala Timo Aila Samuli Laine et al. 2022. ediff-i: Text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324 (2022)."},{"key":"e_1_3_2_1_3_1","unstructured":"James Betker Gabriel Goh Li Jing Tim Brooks Jianfeng Wang Linjie Li Long Ouyang Juntang Zhuang Joyce Lee Yufei Guo et al. 2023. Improving image generation with better captions. OpenAI. https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf Vol. 2 3 (2023) 8."},{"key":"e_1_3_2_1_4_1","volume-title":"Training diffusion models with reinforcement learning. arXiv preprint arXiv:2305.13301","author":"Black Kevin","year":"2023","unstructured":"Kevin Black, Michael Janner, Yilun Du, Ilya Kostrikov, and Sergey Levine. 2023. Training diffusion models with reinforcement learning. arXiv preprint arXiv:2305.13301 (2023)."},{"key":"e_1_3_2_1_5_1","volume-title":"Tutorial on Diffusion Models for Imaging and Vision. arXiv preprint arXiv:2403.18103","author":"Chan Stanley H","year":"2024","unstructured":"Stanley H Chan. 2024. Tutorial on Diffusion Models for Imaging and Vision. arXiv preprint arXiv:2403.18103 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"Muse: Text-to-image generation via masked generative transformers. arXiv preprint arXiv:2301.00704","author":"Chang Huiwen","year":"2023","unstructured":"Huiwen Chang, Han Zhang, Jarred Barber, AJ Maschinot, Jose Lezama, Lu Jiang, Ming-Hsuan Yang, Kevin Murphy, William T Freeman, Michael Rubinstein, et al., 2023. Muse: Text-to-image generation via masked generative transformers. arXiv preprint arXiv:2301.00704 (2023)."},{"key":"e_1_3_2_1_7_1","unstructured":"Junsong Chen Jincheng Yu Chongjian Ge Lewei Yao Enze Xie Yue Wu Zhongdao Wang James Kwok Ping Luo Huchuan Lu et al. 2023. Pixart-?: Fast training of diffusion transformer for photorealistic text-to-image synthesis. arXiv preprint arXiv:2310.00426 (2023)."},{"key":"e_1_3_2_1_8_1","volume-title":"Emu: Enhancing image generation models using photogenic needles in a haystack. arXiv preprint arXiv:2309.15807","author":"Dai Xiaoliang","year":"2023","unstructured":"Xiaoliang Dai, Ji Hou, Chih-Yao Ma, Sam Tsai, Jialiang Wang, Rui Wang, Peizhao Zhang, Simon Vandenhende, Xiaofang Wang, Abhimanyu Dubey, et al., 2023. Emu: Enhancing image generation models using photogenic needles in a haystack. arXiv preprint arXiv:2309.15807 (2023)."},{"key":"e_1_3_2_1_9_1","volume-title":"Scaling Rectified Flow Transformers for High-Resolution Image Synthesis. In Forty-first International Conference on Machine Learning.","author":"Esser Patrick","year":"2024","unstructured":"Patrick Esser, Sumith Kulal, Andreas Blattmann, Rahim Entezari, Jonas M\u00fcller, Harry Saini, Yam Levi, Dominik Lorenz, Axel Sauer, Frederic Boesel, et al., 2024. Scaling Rectified Flow Transformers for High-Resolution Image Synthesis. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_10_1","volume-title":"Vincent Tao Hu, and Bjorn Ommer","author":"Fuest Michael","year":"2024","unstructured":"Michael Fuest, Pingchuan Ma, Ming Gui, Johannes S Fischer, Vincent Tao Hu, and Bjorn Ommer. 2024. Diffusion Models and Representation Learning: A Survey. arXiv preprint arXiv:2407.00783 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf","author":"Team Google Imagen","year":"2024","unstructured":"Google Imagen 3 Team. 2024. Imagen 3. Google DeepMind. https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf (2024)."},{"key":"e_1_3_2_1_12_1","first-page":"36652","article-title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation","volume":"36","author":"Kirstain Yuval","year":"2023","unstructured":"Yuval Kirstain, Adam Polyak, Uriel Singer, Shahbuland Matiana, Joe Penna, and Omer Levy. 2023. Pick-a-pic: An open dataset of user preferences for text-to-image generation. Advances in Neural Information Processing Systems, Vol. 36 (2023), 36652-36663.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_13_1","volume-title":"Imagine flash: Accelerating emu diffusion models with backward distillation. arXiv preprint arXiv:2405.05224","author":"Kohler Jonas","year":"2024","unstructured":"Jonas Kohler, Albert Pumarola, Edgar Sch\u00f6nfeld, Artsiom Sanakoyeu, Roshan Sumbaly, Peter Vajda, and Ali Thabet. 2024. Imagine flash: Accelerating emu diffusion models with backward distillation. arXiv preprint arXiv:2405.05224 (2024)."},{"key":"e_1_3_2_1_14_1","volume-title":"Autoregressive Image Generation without Vector Quantization. arXiv preprint arXiv:2406.11838","author":"Li Tianhong","year":"2024","unstructured":"Tianhong Li, Yonglong Tian, He Li, Mingyang Deng, and Kaiming He. 2024. Autoregressive Image Generation without Vector Quantization. arXiv preprint arXiv:2406.11838 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Step-aware Preference Optimization: Aligning Preference with Denoising Performance at Each Step. arXiv preprint arXiv:2406.04314","author":"Liang Zhanhao","year":"2024","unstructured":"Zhanhao Liang, Yuhui Yuan, Shuyang Gu, Bohan Chen, Tiankai Hang, Ji Li, and Liang Zheng. 2024. Step-aware Preference Optimization: Aligning Preference with Denoising Performance at Each Step. arXiv preprint arXiv:2406.04314 (2024)."},{"key":"e_1_3_2_1_16_1","volume-title":"Simpo: Simple preference optimization with a reference-free reward. arXiv preprint arXiv:2405.14734","author":"Meng Yu","year":"2024","unstructured":"Yu Meng, Mengzhou Xia, and Danqi Chen. 2024. Simpo: Simple preference optimization with a reference-free reward. arXiv preprint arXiv:2405.14734 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952 (2023)."},{"key":"e_1_3_2_1_18_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2024. Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_19_1","volume-title":"Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125, Vol. 1, 2 (2022), 3."},{"key":"e_1_3_2_1_20_1","volume-title":"International conference on machine learning. PMLR, 8821-8831","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, Gabriel Goh, Scott Gray, Chelsea Voss, Alec Radford, Mark Chen, and Ilya Sutskever. 2021. Zero-shot text-to-image generation. In International conference on machine learning. PMLR, 8821-8831."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_22_1","volume-title":"Burcu Karagol Ayan, Tim Salimans, et al.","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily L Denton, Kamyar Ghasemipour, Raphael Gontijo Lopes, Burcu Karagol Ayan, Tim Salimans, et al., 2022. Photorealistic text-to-image diffusion models with deep language understanding. Advances in neural information processing systems, Vol. 35 (2022), 36479-36494."},{"key":"e_1_3_2_1_23_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_24_1","volume-title":"A picture is worth a thousand words: Principled recaptioning improves image generation. arXiv preprint arXiv:2310.16656","author":"Segalis Eyal","year":"2023","unstructured":"Eyal Segalis, Dani Valevski, Danny Lumen, Yossi Matias, and Yaniv Leviathan. 2023. A picture is worth a thousand words: Principled recaptioning improves image generation. arXiv preprint arXiv:2310.16656 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation. arXiv preprint arXiv:2406.06525","author":"Sun Peize","year":"2024","unstructured":"Peize Sun, Yi Jiang, Shoufa Chen, Shilong Zhang, Bingyue Peng, Ping Luo, and Zehuan Yuan. 2024. Autoregressive Model Beats Diffusion: Llama for Scalable Image Generation. arXiv preprint arXiv:2406.06525 (2024)."},{"key":"e_1_3_2_1_26_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_27_1","volume-title":"GenAI Media Generation Challenge Workshop @ CVPR2024","author":"Tsai Sam","year":"2024","unstructured":"Sam Tsai, Ji Hou, Bichen Wu, Xiaoliang Dai, Kevin Chih-Yao Ma, Matthew Yu, Rui Wang, Tianhe Li, Simran Motwani, Ajay Menon, Kunpeng Li, Tao Xu, and Karthik Sivakumar. 2024. GenAI Media Generation Challenge Workshop @ CVPR2024. https:\/\/gamgc.github.io\/ (2024)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00786"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00594"},{"key":"e_1_3_2_1_30_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Zhou Chunting","year":"2024","unstructured":"Chunting Zhou, Pengfei Liu, Puxin Xu, Srinivasan Iyer, Jiao Sun, Yuning Mao, Xuezhe Ma, Avia Efrat, Ping Yu, Lili Yu, et al., 2024. Lima: Less is more for alignment. Advances in Neural Information Processing Systems, Vol. 36 (2024)."}],"event":{"name":"MM '25:The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 3rd International Workshop on Rich Media With Generative AI"],"original-title":[],"deposited":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T14:53:02Z","timestamp":1761403982000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746262.3761978"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,26]]},"references-count":30,"alternative-id":["10.1145\/3746262.3761978","10.1145\/3746262"],"URL":"https:\/\/doi.org\/10.1145\/3746262.3761978","relation":{},"subject":[],"published":{"date-parts":[[2025,10,26]]},"assertion":[{"value":"2025-10-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}