{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T16:40:07Z","timestamp":1755880807073,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":101,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,8]],"date-time":"2024-12-08T00:00:00Z","timestamp":1733616000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,8]]},"DOI":"10.1145\/3673791.3698424","type":"proceedings-article","created":{"date-parts":[[2024,12,8]],"date-time":"2024-12-08T06:24:16Z","timestamp":1733639056000},"page":"42-53","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Offline Evaluation of Set-Based Text-to-Image Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4411-7089","authenticated-orcid":false,"given":"Negar","family":"Arabzadeh","sequence":"first","affiliation":[{"name":"University of Waterloo, Waterloo, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2345-1288","authenticated-orcid":false,"given":"Fernando","family":"Diaz","sequence":"additional","affiliation":[{"name":"Google, Pittsburgh, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5465-5659","authenticated-orcid":false,"given":"Junfeng","family":"He","sequence":"additional","affiliation":[{"name":"Google Inc., Mountain View, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"9th Workshop of the Cross-Language Evaluation Forum (CLEF","author":"Ah-Pine JM","year":"2008","unstructured":"JM Ah-Pine, CM Cifarelli, SM Clinchant, GM Csurka, and Jean-Michel Renders. 2008. XRCE's Participation to ImageCLEF 2008. In 9th Workshop of the Cross-Language Evaluation Forum (CLEF 2008)."},{"key":"e_1_3_2_1_2_1","volume-title":"Charles LA Clarke, and Mark Sanderson","author":"Alaofi Marwah","year":"2024","unstructured":"Marwah Alaofi, Negar Arabzadeh, Charles LA Clarke, and Mark Sanderson. 2024. Generative Information Retrieval Evaluation. arXiv preprint arXiv:2404.08137 (2024)."},{"key":"e_1_3_2_1_3_1","volume-title":"A Comparison of Methods for Evaluating Generative IR. arXiv preprint arXiv:2404.04044","author":"Arabzadeh Negar","year":"2024","unstructured":"Negar Arabzadeh and Charles LA Clarke. 2024. A Comparison of Methods for Evaluating Generative IR. arXiv preprint arXiv:2404.04044 (2024)."},{"key":"e_1_3_2_1_4_1","volume-title":"arXiv preprint arXiv:2401.17543","author":"Arabzadeh Negar","year":"2024","unstructured":"Negar Arabzadeh and Charles LA Clarke. 2024. Fr\\'echet Distance for Offline Evaluation of Information Retrieval Systems with Sparse Labels. arXiv preprint arXiv:2401.17543 (2024)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578337.3605115"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10791-022-09411-0"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01834"},{"key":"e_1_3_2_1_8_1","volume-title":"A note on the inception score. arXiv preprint arXiv:1801.01973","author":"Barratt Shane","year":"2018","unstructured":"Shane Barratt and Rishi Sharma. 2018. A note on the inception score. arXiv preprint arXiv:1801.01973 (2018)."},{"key":"e_1_3_2_1_9_1","unstructured":"Miko\u0142aj Bi\u0144kowski Danica J. Sutherland Michael Arbel and Arthur Gretton. 2021. Demystifying MMD GANs. rXiv:1801.01401 [stat.ML]"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2021.103329"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1108\/EUM0000000007198"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3336191.3371844"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2815601"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2009916.2010037"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1321440.1321564"},{"key":"e_1_3_2_1_16_1","volume-title":"Predicting visual attention in graphic design documents","author":"Chakraborty Souradeep","year":"2022","unstructured":"Souradeep Chakraborty, Zijun Wei, Conor Kelton, Seoyoung Ahn, Aruna Balasubramanian, Gregory J. Zelinsky, and Dimitris Samaras. 2022. Predicting visual attention in graphic design documents. IEEE Transactions on Multimedia (2022), 1--1."},{"key":"e_1_3_2_1_17_1","volume-title":"Beyond Accuracy: Grounding Evaluation Metrics for Human-Machine Learning Systems. In Advances in Neural Information Processing Systems.","author":"Chandar Praveen","year":"2020","unstructured":"Praveen Chandar, Fernando Diaz, and Brian St. Thomas. 2020. Beyond Accuracy: Grounding Evaluation Metrics for Human-Machine Learning Systems. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/1645953.1646033"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/1526709.1526711"},{"key":"e_1_3_2_1_20_1","volume-title":"Re-imagen: Retrieval-augmented text-to-image generator. arXiv preprint arXiv:2209.14491","author":"Chen Wenhu","year":"2022","unstructured":"Wenhu Chen, Hexiang Hu, Chitwan Saharia, and William W Cohen. 2022. Re-imagen: Retrieval-augmented text-to-image generator. arXiv preprint arXiv:2209.14491 (2022)."},{"key":"e_1_3_2_1_21_1","volume-title":"Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325","author":"Chen Xinlei","year":"2015","unstructured":"Xinlei Chen, Hao Fang, Tsung-Yi Lin, Ramakrishna Vedantam, Saurabh Gupta, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2015. Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325 (2015)."},{"key":"e_1_3_2_1_22_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Cho Jaemin","year":"2021","unstructured":"Jaemin Cho, Jie Lei, Hao Tan, and Mohit Bansal. 2021. Unifying vision-and-language tasks via text generation. In International Conference on Machine Learning. PMLR, 1931--1942."},{"key":"e_1_3_2_1_23_1","volume-title":"DALL-Eval: Probing the Reasoning Skills and Social Biases of Text-to-Image Generative Transformers. CoRR abs\/2202.04053","author":"Cho Jaemin","year":"2022","unstructured":"Jaemin Cho, Abhay Zala, and Mohit Bansal. 2022. DALL-Eval: Probing the Reasoning Skills and Social Biases of Text-to-Image Generative Transformers. CoRR abs\/2202.04053 (2022). arXiv:2202.04053 https:\/\/arxiv.org\/abs\/2202.04053"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00283"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00611"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00611"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/1341531.1341545"},{"key":"e_1_3_2_1_28_1","volume-title":"How to Prompt? Opportunities and Challenges of Zero-and Few-Shot Learning for Human-AI Interaction in Creative Applications of Generative Models. arXiv preprint arXiv:2209.01390","author":"Dang Hai","year":"2022","unstructured":"Hai Dang, Lukas Mecke, Florian Lehmann, Sven Goller, and Daniel Buschek. 2022. How to Prompt? Opportunities and Challenges of Zero-and Few-Shot Learning for Human-AI Interaction in Creative Applications of Generative Models. arXiv preprint arXiv:2209.01390 (2022)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3576840.3578327"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2505515.2505717"},{"key":"e_1_3_2_1_31_1","first-page":"19822","article-title":"Cogview: Mastering text-to-image generation via transformers","volume":"34","author":"Ding Ming","year":"2021","unstructured":"Ming Ding, Zhuoyi Yang, Wenyi Hong, Wendi Zheng, Chang Zhou, Da Yin, Junyang Lin, Xu Zou, Zhou Shao, Hongxia Yang, et al. 2021. Cogview: Mastering text-to-image generation via transformers. Advances in Neural Information Processing Systems 34 (2021), 19822--19835.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2021.07.019"},{"key":"e_1_3_2_1_33_1","volume-title":"Make-a-scene: Scene-based text-to-image generation with human priors. arXiv preprint arXiv:2203.13131","author":"Gafni Oran","year":"2022","unstructured":"Oran Gafni, Adam Polyak, Oron Ashual, Shelly Sheynin, Devi Parikh, and Yaniv Taigman. 2022. Make-a-scene: Scene-based text-to-image generation with human priors. arXiv preprint arXiv:2203.13131 (2022)."},{"key":"e_1_3_2_1_34_1","volume-title":"Caponimage: Context-driven dense-captioning on image. arXiv preprint arXiv:2204.12974","author":"Gao Yiqi","year":"2022","unstructured":"Yiqi Gao, Xinglin Hou, Yuanmeng Zhang, Tiezheng Ge, Yuning Jiang, and Peng Wang. 2022. Caponimage: Context-driven dense-captioning on image. arXiv preprint arXiv:2204.12974 (2022)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1037\/h0063487"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/1835449.1835639"},{"key":"e_1_3_2_1_37_1","volume-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems 30","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.3021209"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01866"},{"key":"e_1_3_2_1_40_1","first-page":"78723","article-title":"T2i-compbench: A comprehensive benchmark for open-world compositional text-to-image generation","volume":"36","author":"Huang Kaiyi","year":"2023","unstructured":"Kaiyi Huang, Kaiyue Sun, Enze Xie, Zhenguo Li, and Xihui Liu. 2023. T2i-compbench: A comprehensive benchmark for open-world compositional text-to-image generation. Advances in Neural Information Processing Systems 36 (2023), 78723--78747.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445093"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298710"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN48605.2020.9206644"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3377325.3377522"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/2591677"},{"key":"e_1_3_2_1_46_1","first-page":"36652","article-title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation","volume":"36","author":"Kirstain Yuval","year":"2023","unstructured":"Yuval Kirstain, Adam Polyak, Uriel Singer, Shahbuland Matiana, Joe Penna, and Omer Levy. 2023. Pick-a-pic: An open dataset of user preferences for text-to-image generation. Advances in Neural Information Processing Systems 36 (2023), 36652--36663.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581641.3584078"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300863"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00028"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01765"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.3311\/PPtr.11480"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_54_1","volume-title":"An improved evaluation framework for generative adversarial networks. arXiv preprint arXiv:1803.07474","author":"Liu Shaohui","year":"2018","unstructured":"Shaohui Liu, Yi Wei, Jiwen Lu, and Jie Zhou. 2018. An improved evaluation framework for generative adversarial networks. arXiv preprint arXiv:1803.07474 (2018)."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00036"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-04447-2_89"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/1416950.1416952"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.5555\/1613984.1614008"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"crossref","unstructured":"Vidhya Navalpakkam Ravi Kumar Lihong Li and D. Sivakumar. 2012. Attention and Selection in Online Choice Tasks. In User Modeling Adaptation and Personalization Judith Masthoff Bamshad Mobasher Michel C. Desmarais and Roger Nkambou (Eds.). Springer Berlin Heidelberg Berlin Heidelberg 200--211.","DOI":"10.1007\/978-3-642-31454-4_17"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-63322-6_8"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3569219.3569352"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01372"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01372"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"crossref","unstructured":"Ville Paananen Jonas Oppenlaender and Aku Visuri. 2023. Using Text-to-Image Generation for Architectural Design Ideation. arXiv:2304.10182 [cs.HC]","DOI":"10.1177\/14780771231222783"},{"key":"e_1_3_2_1_65_1","volume-title":"Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 1).","author":"Park Dong Huk","year":"2021","unstructured":"Dong Huk Park, Samaneh Azadi, Xihui Liu, Trevor Darrell, and Anna Rohrbach. 2021. Benchmark for compositional text-to-image synthesis. In Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 1)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01112"},{"key":"e_1_3_2_1_67_1","volume-title":"Best Prompts for Text-to-Image Models and How to Find Them. arXiv preprint arXiv:2209.11711","author":"Pavlichenko Nikita","year":"2022","unstructured":"Nikita Pavlichenko and Dmitry Ustalov. 2022. Best Prompts for Text-to-Image Models and How to Find Them. arXiv preprint arXiv:2209.11711 (2022)."},{"key":"e_1_3_2_1_68_1","volume-title":"NeurIPS 2022 Workshop on Human Evaluation of Generative Models (HEGM).","author":"Petsiuk Vitali","year":"2022","unstructured":"Vitali Petsiuk, Alexander E. Siemenn, Saisamrit Surbehera, Zad Chin, Keith Tyser, Gregory Hunter, Arvind Raghavan, Yann Hicke, Bryan A. Plummer, Ori Kerret, Tonio Buonassisi, Kate Saenko, Armando Solar-Lezama, and Iddo Drori. 2022. Human Evaluation of Text-to-Image Models on a Multi-Task Benchmark. In NeurIPS 2022 Workshop on Human Evaluation of Generative Models (HEGM)."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.2307\/2346567"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_38"},{"key":"e_1_3_2_1_71_1","volume-title":"International Conference on Machine Learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, JongWook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_72_1","volume-title":"Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)."},{"key":"e_1_3_2_1_73_1","volume-title":"International Conference on Machine Learning. PMLR, 8821--8831","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, Gabriel Goh, Scott Gray, Chelsea Voss, Alec Radford, Mark Chen, and Ilya Sutskever. 2021. Zero-shot text-to-image generation. In International Conference on Machine Learning. PMLR, 8821--8831."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"crossref","unstructured":"Navyasri Reddy Samyak Jain Pradeep Yarlagadda and Vineet Gandhi. 2020. Tidying Deep Saliency Prediction Architectures. In IROS.","DOI":"10.1109\/IROS45743.2020.9341574"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/1242572.1242643"},{"volume-title":"2023 ACM Conference on Fairness, Accountability, and Transparency (FAccT '23)","author":"Cynthia","key":"e_1_3_2_1_76_1","unstructured":"Cynthia L. Bennett Emily Denton Rida Qadri, Renee Shelby. 2023. AI's Regimes of Representation: A Community-centered Study of Text-to-Image Models in South Asia. In 2023 ACM Conference on Fairness, Accountability, and Transparency (FAccT '23)."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_78_1","volume-title":"Raphael Gontijo-Lopes, Burcu Karagol Ayan, Tim Salimans, Jonathan Ho, David J. Fleet, and Mohammad Norouzi.","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily Denton, Seyed Kamyar Seyed Ghasemipour, Raphael Gontijo-Lopes, Burcu Karagol Ayan, Tim Salimans, Jonathan Ho, David J. Fleet, and Mohammad Norouzi. 2022. Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding. In Advances in Neural Information Processing Systems, Alice H. Oh, Alekh Agarwal, Danielle Belgrave, and Kyunghyun Cho (Eds.). https:\/\/openreview.net\/forum?id=08Yk-n5l2Al"},{"key":"e_1_3_2_1_79_1","volume-title":"Improved techniques for training gans. Advances in neural information processing systems 29","author":"Salimans Tim","year":"2016","unstructured":"Tim Salimans, Ian Goodfellow, Wojciech Zaremba, Vicki Cheung, Alec Radford, and Xi Chen. 2016. Improved techniques for training gans. Advances in neural information processing systems 29 (2016)."},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0142-694X(02)00034-0"},{"key":"e_1_3_2_1_81_1","volume-title":"Samira Ebrahimi Kahou, and Yoshua Bengio","author":"Sharma Shikhar","year":"2018","unstructured":"Shikhar Sharma, Dendi Suhubdy, Vincent Michalski, Samira Ebrahimi Kahou, and Yoshua Bengio. 2018. Chatpainter: Improving text to image generation using dialogue. arXiv preprint arXiv:1802.08216 (2018)."},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"crossref","unstructured":"Chengyao Shen and Qi Zhao. 2014. Webpage Saliency. In ECCV. 33--46.","DOI":"10.1007\/978-3-319-10584-0_3"},{"key":"e_1_3_2_1_83_1","volume-title":"Appraisals of salient visual elements in web page design. Advances in Human-Computer Interaction 2016","author":"Silvennoinen Johanna M","year":"2016","unstructured":"Johanna M Silvennoinen and Jussi PP Jokinen. 2016. Appraisals of salient visual elements in web page design. Advances in Human-Computer Interaction 2016 (2016)."},{"key":"e_1_3_2_1_84_1","volume-title":"Hannah Rose Kirk, Aleksandar Shtedritski, and Max Bain.","author":"Smith Brandon","year":"2023","unstructured":"Brandon Smith, Miguel Farinha, Siobhan Mackenzie Hall, Hannah Rose Kirk, Aleksandar Shtedritski, and Max Bain. 2023. Balancing the Picture: Debiasing Vision-Language Datasets with Synthetic Contrast Sets. arXiv:2305.15407 [cs.CV]"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2020.3031111"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1145\/1460096.1460109"},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1080\/14626268.2023.2174557"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531728"},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2946000"},{"key":"e_1_3_2_1_91_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01761"},{"key":"e_1_3_2_1_92_1","volume-title":"SimVLM: Simple Visual Language Model Pretraining with Weak Supervision. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=GUrhfTuf_3","author":"Wang Zirui","year":"2022","unstructured":"Zirui Wang, Jiahui Yu, Adams Wei Yu, Zihang Dai, Yulia Tsvetkov, and Yuan Cao. 2022. SimVLM: Simple Visual Language Model Pretraining with Weak Supervision. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=GUrhfTuf_3"},{"key":"e_1_3_2_1_93_1","volume-title":"Some Moral and Technical Consequences of Automation. Science 131, 3410","author":"Wiener Norbert","year":"1960","unstructured":"Norbert Wiener. 1960. Some Moral and Technical Consequences of Automation. Science 131, 3410 (1960), 1355--1358."},{"key":"e_1_3_2_1_94_1","volume-title":"2017 2nd international conference on image, vision and computing (ICIVC). IEEE, 783--787","author":"Xia Xiaoling","year":"2017","unstructured":"Xiaoling Xia, Cui Xu, and Bing Nan. 2017. Inception-v3 for flower classification. In 2017 2nd international conference on image, vision and computing (ICIVC). IEEE, 783--787."},{"key":"e_1_3_2_1_95_1","volume-title":"Imagereward: Learning and evaluating human preferences for text-to-image generation. Advances in Neural Information Processing Systems 36","author":"Xu Jiazheng","year":"2024","unstructured":"Jiazheng Xu, Xiao Liu, YuchenWu, Yuxuan Tong, Qinkai Li, Ming Ding, Jie Tang, and Yuxiao Dong. 2024. Imagereward: Learning and evaluating human preferences for text-to-image generation. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_96_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-018-3849-7"},{"key":"e_1_3_2_1_97_1","volume-title":"What you see is what you read? improving text-image alignment evaluation. Advances in Neural Information Processing Systems 36","author":"Yarom Michal","year":"2024","unstructured":"Michal Yarom, Yonatan Bitton, Soravit Changpinyo, Roee Aharoni, Jonathan Herzig, Oran Lang, Eran Ofek, and Idan Szpektor. 2024. What you see is what you read? improving text-image alignment evaluation. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_98_1","volume-title":"Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, Ben Hutchinson, Wei Han, Zarana Parekh, Xin Li, Han Zhang, Jason Baldridge, and Yonghui Wu.","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Yuanzhong Xu, Jing Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, Ben Hutchinson, Wei Han, Zarana Parekh, Xin Li, Han Zhang, Jason Baldridge, and Yonghui Wu. 2022. Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. Trans. Mach. Learn. Res. 2022 (2022). https:\/\/openreview.net\/forum?id=AFDcYJKhND"},{"key":"e_1_3_2_1_99_1","volume-title":"Rodrygo LT Santos, and Henning M\u00fcller","author":"Zaharieva Maia","year":"2017","unstructured":"Maia Zaharieva, Bogdan Ionescu, Alexandru-Lucian G\u00eensca, Rodrygo LT Santos, and Henning M\u00fcller. 2017. Retrieving Diverse Social Images at MediaEval 2017: Challenges, Dataset and Evaluation. In MediaEval."},{"key":"e_1_3_2_1_100_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00089"},{"key":"e_1_3_2_1_101_1","unstructured":"Yufan Zhou Ruiyi Zhang Changyou Chen Chunyuan Li Chris Tensmeyer Tong Yu Jiuxiang Gu Jinhui Xu and Tong Sun. [n.d.]. LAFITE: Towards Language-Free Training for Text-to-Image Generation. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) ([n.d.]). https:\/\/par.nsf.gov\/biblio\/10351124"}],"event":{"name":"SIGIR-AP 2024: Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"],"location":"Tokyo Japan","acronym":"SIGIR-AP 2024"},"container-title":["Proceedings of the 2024 Annual International ACM SIGIR Conference on Research and Development in Information Retrieval in the Asia Pacific Region"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673791.3698424","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3673791.3698424","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T16:24:10Z","timestamp":1755879850000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3673791.3698424"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,8]]},"references-count":101,"alternative-id":["10.1145\/3673791.3698424","10.1145\/3673791"],"URL":"https:\/\/doi.org\/10.1145\/3673791.3698424","relation":{},"subject":[],"published":{"date-parts":[[2024,12,8]]},"assertion":[{"value":"2024-12-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}