{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T06:09:39Z","timestamp":1782281379634,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,10]]},"DOI":"10.1145\/3626772.3661350","type":"proceedings-article","created":{"date-parts":[[2024,7,11]],"date-time":"2024-07-11T12:40:05Z","timestamp":1720701605000},"page":"2855-2859","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Enhancing Baidu Multimodal Advertisement with Chinese Text-to-Image Generation via Bilingual Alignment and Caption Synthesis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-3053-3965","authenticated-orcid":false,"given":"Kang","family":"Zhao","sequence":"first","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0253-5488","authenticated-orcid":false,"given":"Xinyu","family":"Zhao","sequence":"additional","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0863-2539","authenticated-orcid":false,"given":"Zhipeng","family":"Jin","sequence":"additional","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5077-4782","authenticated-orcid":false,"given":"Yi","family":"Yang","sequence":"additional","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5481-4512","authenticated-orcid":false,"given":"Wen","family":"Tao","sequence":"additional","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6516-1139","authenticated-orcid":false,"given":"Cong","family":"Han","sequence":"additional","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7346-5258","authenticated-orcid":false,"given":"Shuanglong","family":"Li","sequence":"additional","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7305-8940","authenticated-orcid":false,"given":"Lin","family":"Liu","sequence":"additional","affiliation":[{"name":"Baidu Search Ads, Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,11]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"SPICE: Semantic Propositional Image Caption Evaluation. ArXiv abs\/1607.08822","author":"Anderson Peter","year":"2016","unstructured":"Peter Anderson, Basura Fernando, Mark Johnson, and Stephen Gould. 2016. SPICE: Semantic Propositional Image Caption Evaluation. ArXiv abs\/1607.08822 (2016)."},{"key":"e_1_3_2_1_2_1","volume-title":"Wasserstein Generative Adversarial Networks. In International Conference on Machine Learning.","author":"Arjovsky Mart\u00edn","year":"2017","unstructured":"Mart\u00edn Arjovsky, Soumith Chintala, and L\u00e9on Bottou. 2017. Wasserstein Generative Adversarial Networks. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_3_1","volume-title":"Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. ArXiv abs\/2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. ArXiv abs\/2308.12966 (2023)."},{"key":"e_1_3_2_1_4_1","volume-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Bao Fan","year":"2022","unstructured":"Fan Bao, Shen Nie, Kaiwen Xue, Yue Cao, Chongxuan Li, Hang Su, and Jun Zhu. 2022. All are Worth Words: A ViT Backbone for Diffusion Models. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022), 22669--22679."},{"key":"e_1_3_2_1_5_1","unstructured":"James Betker Gabriel Goh Li Jing ? Tim Brooks Jianfeng Wang Linjie Li ? LongOuyang ? Juntang Zhuang ? Joyce Lee ? Yufei Guo ? Wesam Manassra ? Prafulla Dhariwal ? Casey Chu ? Yunxin Jiao and Aditya Ramesh. [n. d.]. Improving Image Generation with Better Captions."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1242572.1242644"},{"key":"e_1_3_2_1_7_1","volume-title":"Muse: Text-To-Image Generation via Masked Generative Transformers. ArXiv abs\/2301.00704","author":"Chang Huiwen","year":"2023","unstructured":"Huiwen Chang, Han Zhang, Jarred Barber, AJ Maschinot, Jos\u00e9 Lezama, Lu Jiang, Ming Yang, Kevin P. Murphy, William T. Freeman, Michael Rubinstein, Yuanzhen Li, and Dilip Krishnan. 2023. Muse: Text-To-Image Generation via Masked Generative Transformers. ArXiv abs\/2301.00704 (2023)."},{"key":"e_1_3_2_1_8_1","volume-title":"AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities. ArXiv abs\/2211.06679","author":"Chen Zhongzhi","year":"2022","unstructured":"Zhongzhi Chen, Guangyi Liu, Bo Zhang, Fulong Ye, Qinghong Yang, and Ledell Yu Wu. 2022. AltCLIP: Altering the Language Encoder in CLIP for Extended Language Capabilities. ArXiv abs\/2211.06679 (2022)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2959100.2959190"},{"key":"e_1_3_2_1_10_1","volume-title":"Generative adversarial nets. Advances in neural information processing systems 27","author":"Goodfellow Ian","year":"2014","unstructured":"Ian Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio. 2014. Generative adversarial nets. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_2_1_11_1","volume-title":"Learning Instance-Level Representation for Large-Scale Multi-Modal Pretraining in E-Commerce. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Jin Yang","year":"2023","unstructured":"Yang Jin, Yongzhi Li, Zehuan Yuan, and Yadong Mu. 2023. Learning Instance-Level Representation for Large-Scale Multi-Modal Pretraining in E-Commerce. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2023), 11060--11069."},{"key":"e_1_3_2_1_12_1","volume-title":"International Conference on Machine Learning.","author":"Li Junnan","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven C. H. Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_13_1","volume-title":"Translation-Enhanced Multilingual Text-to-Image Generation. In Annual Meeting of the Association for Computational Linguistics.","author":"Li Yaoyiran","year":"2023","unstructured":"Yaoyiran Li, Ching-Yun Chang, Stephen Rawls, Ivan Vulic, and Anna Korhonen. 2023. Translation-Enhanced Multilingual Text-to-Image Generation. In Annual Meeting of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_14_1","unstructured":"Romain Lopez Pierre Boyeau Nir Yosef Michael I. Jordan and Jeffrey Regier. 2020. AUTO-ENCODING VARIATIONAL BAYES."},{"key":"e_1_3_2_1_15_1","volume-title":"Scalable Diffusion Models with Trans-formers. 2023 IEEE\/CVF International Conference on Computer Vision (ICCV)","author":"William","year":"2022","unstructured":"William S. Peebles and Saining Xie. 2022. Scalable Diffusion Models with Trans-formers. 2023 IEEE\/CVF International Conference on Computer Vision (ICCV) (2022), 4172--4182."},{"key":"e_1_3_2_1_16_1","volume-title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. ArXiv abs\/2307.01952","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, A. Blattmann, Tim Dockhorn, Jonas Muller, Joe Penna, and Robin Rombach. 2023. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. ArXiv abs\/2307.01952 (2023)."},{"key":"e_1_3_2_1_17_1","volume-title":"Zero-Shot Text-to-Image Generation. ArXiv abs\/2102.12092","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, Gabriel Goh, Scott Gray, Chelsea Voss, Alec Rad-ford, Mark Chen, and Ilya Sutskever. 2021. Zero-Shot Text-to-Image Generation. ArXiv abs\/2102.12092 (2021)."},{"key":"e_1_3_2_1_18_1","volume-title":"Dream Booth: Fine Tuning Text-to-Image Diffusion Models for Subject-Driven Generation. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Ruiz Nataniel","year":"2022","unstructured":"Nataniel Ruiz, Yuanzhen Li, Varun Jampani, Yael Pritch, Michael Rubinstein, and Kfir Aberman. 2022. Dream Booth: Fine Tuning Text-to-Image Diffusion Models for Subject-Driven Generation. 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022), 22500--22510."},{"key":"e_1_3_2_1_19_1","volume-title":"LAION-5B: An open large-scale dataset for training next generation image-text models. ArXiv abs\/2210.08402","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, Patrick Schramowski, Srivatsa Kundurthy, Katherine Crowson, Ludwig Schmidt, Robert Kaczmarczyk, and Jenia Jitsev. 2022. LAION-5B: An open large-scale dataset for training next generation image-text models. ArXiv abs\/2210.08402 (2022)."},{"key":"e_1_3_2_1_20_1","volume-title":"2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Vedantam Ramakrishna","year":"2014","unstructured":"Ramakrishna Vedantam, C. Lawrence Zitnick, and Devi Parikh. 2014. CIDEr: Consensus-based image description evaluation. 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2014), 4566--4575."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591844"},{"key":"e_1_3_2_1_22_1","volume-title":"Taiyi-Diffusion-XL: Advancing Bilingual Text-to-Image Generation with Large Vision-Language Model Support. ArXiv abs\/2401.14688","author":"Wu Xiaojun","year":"2024","unstructured":"Xiaojun Wu, Di Zhang, Ruyi Gan, Junyu Lu, Ziwei Wu, Renliang Sun, Jiaxing Zhang, Pingjian Zhang, and Yan Song. 2024. Taiyi-Diffusion-XL: Advancing Bilingual Text-to-Image Generation with Large Vision-Language Model Support. ArXiv abs\/2401.14688 (2024)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3505243"},{"key":"e_1_3_2_1_24_1","volume-title":"Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, Benton C. Hutchinson, Wei Han, Zarana Parekh, Xin Li, Han Zhang, Jason Baldridge, and Yonghui Wu.","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Yuanzhong Xu, Jing Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, Benton C. Hutchinson, Wei Han, Zarana Parekh, Xin Li, Han Zhang, Jason Baldridge, and Yonghui Wu. 2022. Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. Trans. Mach. Learn. Res. 2022 (2022)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigData55660.2022.10020786"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557653"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539061"},{"key":"e_1_3_2_1_28_1","volume-title":"TIRA in Baidu Image Advertising. 2021 IEEE 37th International Conference on Data Engineering (ICDE)","author":"Yu Tan","year":"2021","unstructured":"Tan Yu, Xuemeng Yang, Yan Jiang, Hongfang Zhang, Weijie Zhao, and Ping Li. 2021. TIRA in Baidu Image Advertising. 2021 IEEE 37th International Conference on Data Engineering (ICDE) (2021), 2207--2212."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462924"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591859"},{"key":"e_1_3_2_1_32_1","volume-title":"Kaleido-BERT: Vision-Language Pre-training on Fashion Domain. 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Zhuge Mingchen","year":"2021","unstructured":"Mingchen Zhuge, Dehong Gao, Deng-Ping Fan, Linbo Jin, Ben Chen, Hao Zhou, Minghui Qiu, and Ling Shao. 2021. Kaleido-BERT: Vision-Language Pre-training on Fashion Domain. 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2021), 12642--12652"}],"event":{"name":"SIGIR 2024: The 47th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Washington DC USA","acronym":"SIGIR 2024","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626772.3661350","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3626772.3661350","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T05:23:55Z","timestamp":1755840235000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626772.3661350"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,10]]},"references-count":32,"alternative-id":["10.1145\/3626772.3661350","10.1145\/3626772"],"URL":"https:\/\/doi.org\/10.1145\/3626772.3661350","relation":{},"subject":[],"published":{"date-parts":[[2024,7,10]]},"assertion":[{"value":"2024-07-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}