{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T05:47:24Z","timestamp":1777873644006,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737223","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T21:07:39Z","timestamp":1754255259000},"page":"4414-4423","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["FoRAGe: High-CTR Food Image Synthesis with Retrieval-Augmented Diffusion Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-1190-247X","authenticated-orcid":false,"given":"Jiaxu","family":"Feng","sequence":"first","affiliation":[{"name":"Shanghai Key Lab of Data Science, College of Computer Science and Artificial Intelligence, Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9306-6106","authenticated-orcid":false,"given":"Xinyu","family":"Gao","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Data Science, College of Computer Science and Artificial Intelligence, Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3084-9852","authenticated-orcid":false,"given":"Muqi","family":"Huang","sequence":"additional","affiliation":[{"name":"Rajax Network Technology (ele.me), Alibaba Group, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7706-5346","authenticated-orcid":false,"given":"Kanjun","family":"Xu","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Data Science, College of Computer Science and Artificial Intelligence, Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8575-5415","authenticated-orcid":false,"given":"Yun","family":"Xiong","sequence":"additional","affiliation":[{"name":"Shanghai Key Lab of Data Science, College of Computer Science and Artificial Intelligence, Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9660-3863","authenticated-orcid":false,"given":"Kun","family":"Zhou","sequence":"additional","affiliation":[{"name":"Rajax Network Technology (ele.me), Alibaba Group, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9094-3065","authenticated-orcid":false,"given":"Chuan","family":"Li","sequence":"additional","affiliation":[{"name":"Rajax Network Technology (ele.me), Alibaba Group, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3455-6637","authenticated-orcid":false,"given":"Feng","family":"Shi","sequence":"additional","affiliation":[{"name":"Rajax Network Technology (ele.me), Alibaba Group, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/2187980.2188075"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1114"},{"key":"e_1_3_2_2_3_1","first-page":"446","volume-title":"Switzerland","author":"Bossard Lukas","year":"2014","unstructured":"Lukas Bossard, Matthieu Guillaumin, and Luc Van Gool. 2014. Food-101-mining discriminative components with random forests. In Computer vision-ECCV 2014: 13th European conference, zurich, Switzerland, September 6-12, 2014, proceedings, part VI 13. Springer, 446-461."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964325"},{"key":"e_1_3_2_2_5_1","volume-title":"Re-Imagen: Retrieval-Augmented Text-to-Image Generator. In The Eleventh International Conference on Learning Representations.","author":"Chen Wenhu","unstructured":"Wenhu Chen, Hexiang Hu, Chitwan Saharia, and William W Cohen. [n.d.]. Re-Imagen: Retrieval-Augmented Text-to-Image Generator. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714836"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2339530.2339652"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","unstructured":"Yiming Cui Wanxiang Che Ting Liu Bing Qin and Ziqing Yang. 2021. Pre-Training with Whole Word Masking for Chinese BERT. doi:10.1109\/TASLP.2021.3124365","DOI":"10.1109\/TASLP.2021.3124365"},{"key":"e_1_3_2_2_9_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805(2018).","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805(2018)."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00744"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539149"},{"key":"e_1_3_2_2_12_1","volume-title":"Generative adversarial nets. Advances in neural information processing systems","author":"Goodfellow Ian","year":"2014","unstructured":"Ian Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio. 2014. Generative adversarial nets. Advances in neural information processing systems, Vol. 27 (2014)."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093463"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482327"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2648584.2648589"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.3389\/frai.2022.976235"},{"key":"e_1_3_2_2_17_1","volume-title":"Denoising diffusion probabilistic models. Advances in neural information processing systems","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems, Vol. 33 (2020), 6840-6851."},{"key":"e_1_3_2_2_18_1","volume-title":"Proceedings of the Asian Conference on Computer Vision. 108-124","author":"Kim Taehun","year":"2022","unstructured":"Taehun Kim, Kunhee Kim, Joonyeong Lee, Dongmin Cha, Jiho Lee, and Daijin Kim. 2022. Revisiting image pyramid structure for high resolution salient object detection. In Proceedings of the Asian Conference on Computer Vision. 108-124."},{"key":"e_1_3_2_2_19_1","unstructured":"Diederik P Kingma. 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114(2013)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_2_21_1","unstructured":"Yueh-Ning Ku Mikhail Kuznetsov Shaunak Mishra and Paloma de Juan. 2023. Staging E-Commerce Products for Online Advertising using Retrieval Assisted Image Generation. In AdKDD@ KDD."},{"key":"e_1_3_2_2_22_1","volume-title":"Foodsam: Any food segmentation","author":"Lan Xing","year":"2023","unstructured":"Xing Lan, Jiayi Lyu, Hanyu Jiang, Kun Dong, Zehai Niu, Yi Zhang, and Jian Xue. 2023. Foodsam: Any food segmentation. IEEE Transactions on Multimedia(2023)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3679885"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3366423.3380163"},{"key":"e_1_3_2_2_25_1","volume-title":"Proceedings of the 26th ACM SIGKDD international conference on knowledge discovery & data mining. 2686-2696","author":"Liu Hu","year":"2020","unstructured":"Hu Liu, Jing Lu, Hao Yang, Xiwei Zhao, Sulong Xu, Hao Peng, Zehua Zhang, Wenjie Niu, Xiaokun Zhu, Yongjun Bao, et al., 2020. Category-Specific CNN for Visual-aware CTR Prediction at JD. com. In Proceedings of the 26th ACM SIGKDD international conference on knowledge discovery & data mining. 2686-2696."},{"key":"e_1_3_2_2_26_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692, Vol. 364 (2019)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_28_1","volume-title":"SIFT: An Algorithm for Extracting Structural Information From Taxonomies. arXiv preprint arXiv:1602.07064(2016).","author":"Martinez-Gil Jorge","year":"2016","unstructured":"Jorge Martinez-Gil. 2016. SIFT: An Algorithm for Extracting Structural Information From Taxonomies. arXiv preprint arXiv:1602.07064(2016)."},{"key":"e_1_3_2_2_29_1","volume-title":"NeurIPS Efficient Natural Language and Speech Processing Workshop.","author":"Muhamed Aashiq","year":"2021","unstructured":"Aashiq Muhamed, Iman Keivanloo, Sujan Perera, James Mracek, Yi Xu, Qingjun Cui, Santosh Rajagopalan, Belinda Zeng, and Trishul Chilimbi. 2021. CTR-BERT: Cost-effective knowledge distillation for billion-parameter teacher models. In NeurIPS Efficient Natural Language and Speech Processing Workshop."},{"key":"e_1_3_2_2_30_1","volume-title":"Fast approximate nearest neighbors with automatic algorithm configuration. VISAPP (1)","author":"Muja Marius","year":"2009","unstructured":"Marius Muja and David G Lowe. 2009. Fast approximate nearest neighbors with automatic algorithm configuration. VISAPP (1), Vol. 2, 331-340 (2009), 2."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413636"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00819"},{"key":"e_1_3_2_2_33_1","volume-title":"International conference on machine learning. PMLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748-8763."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2010.127"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/1242572.1242643"},{"key":"e_1_3_2_2_36_1","volume-title":"Foodfusion: A novel approach for food image composition via diffusion models. arXiv preprint arXiv:2408.14135(2024).","author":"Shi Chaohua","year":"2024","unstructured":"Chaohua Shi, Xuan Wang, Si Shi, Xule Wang, Mingrui Zhu, Nannan Wang, and Xinbo Gao. 2024. Foodfusion: A novel approach for food image composition via diffusion models. arXiv preprint arXiv:2408.14135(2024)."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3357925"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671636"},{"key":"e_1_3_2_2_39_1","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems(2017)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599780"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3640457.3688106"},{"key":"e_1_3_2_2_42_1","unstructured":"Youze Xue Binghui Chen Yifeng Geng Xuansong Xie Jiansheng Chen and Hongbing Ma. 2024. Strictly-ID-Preserved and Controllable Accessory Advertising Image Generation. arXiv preprint arXiv:2404.04828(2024)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557721"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589335.3648315"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-30671-1_4"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01174"}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737223","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T18:08:11Z","timestamp":1777572491000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737223"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":47,"alternative-id":["10.1145\/3711896.3737223","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737223","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}