{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:44:59Z","timestamp":1787017499311,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":70,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T00:00:00Z","timestamp":1745280000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["No.2023ZD0121104"],"award-info":[{"award-number":["No.2023ZD0121104"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.62222213, No.U22B2059, No.92370204"],"award-info":[{"award-number":["No.62222213, No.U22B2059, No.92370204"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,4,22]]},"DOI":"10.1145\/3696410.3714777","type":"proceedings-article","created":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T18:57:28Z","timestamp":1745348248000},"page":"1666-1682","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["<i>ImageScope:<\/i>\n                    Unifying Language-Guided Image Retrieval via Large Multimodal Model Collective Reasoning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0889-7660","authenticated-orcid":false,"given":"Pengfei","family":"Luo","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, State Key Laboratory of Cognitive Intelligence, University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2677-7021","authenticated-orcid":false,"given":"Jingbo","family":"Zhou","sequence":"additional","affiliation":[{"name":"Business Intelligence Lab, Baidu Research, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4246-5386","authenticated-orcid":false,"given":"Tong","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, State Key Laboratory of Cognitive Intelligence, University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3993-3632","authenticated-orcid":false,"given":"Yuan","family":"Xia","sequence":"additional","affiliation":[{"name":"Baidu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0227-3793","authenticated-orcid":false,"given":"Linli","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, State Key Laboratory of Cognitive Intelligence, University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4835-4102","authenticated-orcid":false,"given":"Enhong","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, State Key Laboratory of Cognitive Intelligence, University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,4,22]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval. CoRR","author":"Agnolucci Lorenzo","year":"2024","unstructured":"Lorenzo Agnolucci, Alberto Baldrati, Marco Bertini, and Alberto Del Bimbo. 2024. iSEARLE: Improving Textual Inversion for Zero-Shot Composed Image Retrieval. CoRR, Vol. abs\/2405.02951 (2024)."},{"key":"e_1_3_2_1_2_1","unstructured":"AI@Meta. 2024. Llama 3 Model Card. (2024). https:\/\/github.com\/meta-llama\/llama3\/blob\/main\/MODEL_CARD.md"},{"key":"e_1_3_2_1_3_1","volume-title":"VQA: Visual Question Answering","author":"Antol Stanislaw","year":"2015","unstructured":"Stanislaw Antol, Aishwarya Agrawal, Jiasen Lu, Margaret Mitchell, Dhruv Batra, C. Lawrence Zitnick, and Devi Parikh. 2015. VQA: Visual Question Answering. In ICCV. IEEE Computer Society, 2425--2433."},{"key":"e_1_3_2_1_4_1","volume-title":"Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. CoRR","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-VL: A Frontier Large Vision-Language Model with Versatile Abilities. CoRR, Vol. abs\/2308.12966 (2023)."},{"key":"e_1_3_2_1_5_1","volume-title":"Zero-Shot Composed Image Retrieval with Textual Inversion","author":"Baldrati Alberto","unstructured":"Alberto Baldrati, Lorenzo Agnolucci, Marco Bertini, and Alberto Del Bimbo. 2023. Zero-Shot Composed Image Retrieval with Textual Inversion. In ICCV. IEEE, 15292--15301."},{"key":"e_1_3_2_1_6_1","volume-title":"Effective conditioned and composed image retrieval combining CLIP-based features","author":"Baldrati Alberto","unstructured":"Alberto Baldrati, Marco Bertini, Tiberio Uricchio, and Alberto Del Bimbo. 2022. Effective conditioned and composed image retrieval combining CLIP-based features. In CVPR. IEEE, 21434--21442."},{"key":"e_1_3_2_1_7_1","volume-title":"Graph of Thoughts: Solving Elaborate Problems with Large Language Models","author":"Besta Maciej","unstructured":"Maciej Besta, Nils Blach, Ales Kubicek, Robert Gerstenberger, Michal Podstawski, Lukas Gianinazzi, Joanna Gajda, Tomasz Lehmann, Hubert Niewiadomski, Piotr Nyczyk, and Torsten Hoefler. 2024. Graph of Thoughts: Solving Elaborate Problems with Large Language Models. In AAAI. AAAI Press, 17682--17690."},{"key":"e_1_3_2_1_8_1","unstructured":"Lucas Beyer Andreas Steiner Andr\u00e9 Susano Pinto Alexander Kolesnikov Xiao Wang Daniel Salz Maxim Neumann Ibrahim Alabdulmohsin Michael Tschannen Emanuele Bugliarello Thomas Unterthiner Daniel Keysers Skanda Koppula Fangyu Liu Adam Grycner Alexey A. Gritsenko Neil Houlsby Manoj Kumar Keran Rong Julian Eisenschlos Rishabh Kabra Matthias Bauer Matko Bosnjak Xi Chen Matthias Minderer Paul Voigtlaender Ioana Bica Ivana Balazevic Joan Puigcerver Pinelopi Papalampidi Olivier J. H\u00e9naff Xi Xiong Radu Soricut Jeremiah Harmsen and Xiaohua Zhai. 2024. PaliGemma: A versatile 3B VLM for transfer. CoRR Vol. abs\/2407.07726 (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"Image-text Retrieval: A Survey on Recent Research and Development. In IJCAI. ijcai.org, 5410--5417.","author":"Cao Min","year":"2022","unstructured":"Min Cao, Shiping Li, Juntao Li, Liqiang Nie, and Min Zhang. 2022. Image-text Retrieval: A Survey on Recent Research and Development. In IJCAI. ijcai.org, 5410--5417."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Ben Chen Linbo Jin Xinxin Wang Dehong Gao Wen Jiang and Wei Ning. 2023a. Unified Vision-Language Representation Modeling for E-Commerce Same-style Products Retrieval. In WWW (Companion Volume). ACM 381--385.","DOI":"10.1145\/3543873.3584632"},{"key":"e_1_3_2_1_11_1","article-title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning","volume":"2023","author":"Chen Wenhu","year":"2023","unstructured":"Wenhu Chen, Xueguang Ma, Xinyi Wang, and William W. Cohen. 2023b. Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks. Trans. Mach. Learn. Res., Vol. 2023 (2023).","journal-title":"Tasks. Trans. Mach. Learn. Res."},{"key":"e_1_3_2_1_12_1","volume-title":"Image Search With Text Feedback by Visiolinguistic Attention Learning","author":"Chen Yanbei","unstructured":"Yanbei Chen, Shaogang Gong, and Loris Bazzani. 2020. Image Search With Text Feedback by Visiolinguistic Attention Learning. In CVPR. Computer Vision Foundation \/ IEEE, 2998--3008."},{"key":"e_1_3_2_1_13_1","volume-title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites. CoRR","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Weiyun Wang, Hao Tian, Shenglong Ye, Zhangwei Gao, Erfei Cui, Wenwen Tong, Kongzhi Hu, Jiapeng Luo, Zheng Ma, Ji Ma, Jiaqi Wang, Xiaoyi Dong, Hang Yan, Hewei Guo, Conghui He, Botian Shi, Zhenjiang Jin, Chao Xu, Bin Wang, Xingjian Wei, Wei Li, Wenjian Zhang, Bo Zhang, Pinlong Cai, Licheng Wen, Xiangchao Yan, Min Dou, Lewei Lu, Xizhou Zhu, Tong Lu, Dahua Lin, Yu Qiao, Jifeng Dai, and Wenhai Wang. 2024. How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites. CoRR, Vol. abs\/2404.16821 (2024)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Mingmin Chi Peiwu Zhang Yingbin Zhao Rui Feng and Xiangyang Xue. 2009. Web image retrieval reranking with multi-view clustering. In WWW. ACM 1189--1190.","DOI":"10.1145\/1526709.1526922"},{"key":"e_1_3_2_1_15_1","volume-title":"Fluffy'': Personalizing Frozen Vision-Language Representations. In ECCV (20) (Lecture Notes in Computer Science","author":"Cohen Niv","unstructured":"Niv Cohen, Rinon Gal, Eli A. Meirom, Gal Chechik, and Yuval Atzmon. 2022. ''This Is My Unicorn, Fluffy'': Personalizing Frozen Vision-Language Representations. In ECCV (20) (Lecture Notes in Computer Science, Vol. 13680). Springer, 558--577."},{"key":"e_1_3_2_1_16_1","volume-title":"Visual Dialog","author":"Das Abhishek","unstructured":"Abhishek Das, Satwik Kottur, Khushi Gupta, Avi Singh, Deshraj Yadav, Jos\u00e9 M. F. Moura, Devi Parikh, and Dhruv Batra. 2017a. Visual Dialog. In CVPR. IEEE Computer Society, 1080--1089."},{"key":"e_1_3_2_1_17_1","volume-title":"Learning Cooperative Visual Dialog Agents with Deep Reinforcement Learning","author":"Das Abhishek","unstructured":"Abhishek Das, Satwik Kottur, Jos\u00e9 M. F. Moura, Stefan Lee, and Dhruv Batra. 2017b. Learning Cooperative Visual Dialog Agents with Deep Reinforcement Learning. In ICCV. IEEE Computer Society, 2970--2979."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Yue Gao Meng Wang Huan-Bo Luan Jialie Shen Shuicheng Yan and Dacheng Tao. 2011. Tag-based social image search with visual-text joint hypergraph learning. In ACM Multimedia. ACM 1517--1520.","DOI":"10.1145\/2072298.2072054"},{"key":"e_1_3_2_1_19_1","volume-title":"FashionVLP: Vision Language Transformer for Fashion Retrieval with Feedback","author":"Goenka Sonam","unstructured":"Sonam Goenka, Zhaoheng Zheng, Ayush Jaiswal, Rakesh Chada, Yue Wu, Varsha Hedau, and Pradeep Natarajan. 2022. FashionVLP: Vision Language Transformer for Fashion Retrieval with Feedback. In CVPR. IEEE, 14085--14095."},{"key":"e_1_3_2_1_20_1","volume-title":"CRITIC: Large Language Models Can Self-Correct with Tool-Interactive Critiquing. In ICLR. OpenReview.net.","author":"Gou Zhibin","year":"2024","unstructured":"Zhibin Gou, Zhihong Shao, Yeyun Gong, Yelong Shen, Yujiu Yang, Nan Duan, and Weizhu Chen. 2024. CRITIC: Large Language Models Can Self-Correct with Tool-Interactive Critiquing. In ICLR. OpenReview.net."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1116-0"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01256"},{"key":"e_1_3_2_1_23_1","first-page":"18","article-title":"Content-Based Image Retrieval Systems - Guest Editors","volume":"28","author":"Gudivada Venkat N.","year":"1995","unstructured":"Venkat N. Gudivada and Vijay V. Raghavan. 1995. Content-Based Image Retrieval Systems - Guest Editors' Introduction. Computer, Vol. 28, 9 (1995), 18--22.","journal-title":"Introduction. Computer"},{"key":"e_1_3_2_1_24_1","unstructured":"Xiaoxiao Guo Hui Wu Yu Cheng Steven Rennie Gerald Tesauro and Rog\u00e9rio Schmidt Feris. 2018. Dialog-based Interactive Image Retrieval. In NeurIPS. 676--686."},{"key":"e_1_3_2_1_25_1","volume-title":"ECCV (35) (Lecture Notes in Computer Science","author":"Han Xiao","unstructured":"Xiao Han, Licheng Yu, Xiatian Zhu, Li Zhang, Yi-Zhe Song, and Tao Xiang. 2022. FashionViL: Fashion-Focused Vision-and-Language Representation Learning. In ECCV (35) (Lecture Notes in Computer Science, Vol. 13695). Springer, 634--651."},{"key":"e_1_3_2_1_26_1","volume-title":"Composed Query Image Retrieval Using Locally Bounded Features","author":"Hosseinzadeh Mehrdad","unstructured":"Mehrdad Hosseinzadeh and Yang Wang. 2020. Composed Query Image Retrieval Using Locally Bounded Features. In CVPR. Computer Vision Foundation \/ IEEE, 3593--3602."},{"key":"e_1_3_2_1_27_1","volume-title":"Language Guided Local Infiltration for Interactive Image Retrieval. In CVPR Workshops. IEEE, 6104--6113","author":"Huang Fuxiang","year":"2023","unstructured":"Fuxiang Huang and Lei Zhang. 2023. Language Guided Local Infiltration for Interactive Image Retrieval. In CVPR Workshops. IEEE, 6104--6113."},{"key":"e_1_3_2_1_28_1","unstructured":"Jinfeng Huang Qiaoqiao She Wenbin Jiang Hua Wu Yang Hao Tong Xu and Feng Wu. 2024. QDMR-based Planning-and-Solving Prompting for Complex Reasoning Tasks. In LREC\/COLING. ELRA and ICCL 13395--13406."},{"key":"e_1_3_2_1_29_1","volume-title":"Taxonomy, Challenges, and Open Questions. CoRR","author":"Huang Lei","year":"2023","unstructured":"Lei Huang, Weijiang Yu, Weitao Ma, Weihong Zhong, Zhangyin Feng, Haotian Wang, Qianglong Chen, Weihua Peng, Xiaocheng Feng, Bing Qin, and Ting Liu. 2023. A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions. CoRR, Vol. abs\/2311.05232 (2023)."},{"key":"e_1_3_2_1_30_1","volume-title":"ICML (Proceedings of Machine Learning Research","volume":"4916","author":"Jia Chao","year":"2021","unstructured":"Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc V. Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig. 2021. Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision. In ICML (Proceedings of Machine Learning Research, Vol. 139). PMLR, 4904--4916."},{"key":"e_1_3_2_1_31_1","volume-title":"HyCIR: Boosting Zero-Shot Composed Image Retrieval with Synthetic Labels. CoRR","author":"Jiang Yingying","year":"2024","unstructured":"Yingying Jiang, Hanchao Jia, Xiaobing Wang, and Peng Hao. 2024. HyCIR: Boosting Zero-Shot Composed Image Retrieval with Synthetic Labels. CoRR, Vol. abs\/2407.05795 (2024)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Xin Jin Jiebo Luo Jie Yu Gang Wang Dhiraj Joshi and Jiawei Han. 2010. iRIN: image retrieval in image-rich information networks. In WWW. ACM 1261--1264.","DOI":"10.1145\/1772690.1772897"},{"key":"e_1_3_2_1_33_1","unstructured":"Shyamgopal Karthik Karsten Roth Massimiliano Mancini and Zeynep Akata. 2024. Vision-by-Language for Training-Free Compositional Image Retrieval. In ICLR. OpenReview.net."},{"key":"e_1_3_2_1_34_1","volume-title":"Dual Compositional Learning in Interactive Image Retrieval","author":"Kim Jongseok","unstructured":"Jongseok Kim, Youngjae Yu, Hoeseong Kim, and Gunhee Kim. 2021. Dual Compositional Learning in Interactive Image Retrieval. In AAAI. AAAI Press, 1771--1779."},{"key":"e_1_3_2_1_35_1","volume-title":"ACL (1)","author":"Lee Saehyung","unstructured":"Saehyung Lee, Sangwon Yu, Junsung Park, Jihun Yi, and Sungroh Yoon. 2024. Interactive Text-to-Image Retrieval with Large Language Models: A Plug-and-Play Approach. In ACL (1). Association for Computational Linguistics, 791--809."},{"key":"e_1_3_2_1_36_1","unstructured":"Matan Levy Rami Ben-Ari Nir Darshan and Dani Lischinski. 2023. Chatting Makes Perfect: Chat-based Image Retrieval. In NeurIPS."},{"key":"e_1_3_2_1_37_1","volume-title":"Hoi","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven C. H. Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In ICML (Proceedings of Machine Learning Research, Vol. 162). PMLR, 12888--12900."},{"key":"e_1_3_2_1_38_1","unstructured":"Junnan Li Ramprasaath R. Selvaraju Akhilesh Gotmare Shafiq R. Joty Caiming Xiong and Steven Chu-Hong Hoi. 2021. Align before Fuse: Vision and Language Representation Learning with Momentum Distillation. In NeurIPS. 9694--9705."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Haoqiang Lin Haokun Wen Xuemeng Song Meng Liu Yupeng Hu and Liqiang Nie. 2024. Fine-grained Textual Inversion Network for Zero-Shot Composed Image Retrieval. In SIGIR. ACM 240--250.","DOI":"10.1145\/3626772.3657831"},{"key":"e_1_3_2_1_40_1","volume-title":"ECCV (5) (Lecture Notes in Computer Science","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge J. Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV (5) (Lecture Notes in Computer Science, Vol. 8693). Springer, 740--755."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"e_1_3_2_1_42_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong Jae Lee. 2024b. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01--30-llava-next\/"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2006.04.045"},{"key":"e_1_3_2_1_44_1","volume-title":"Damien Teney, and Stephen Gould.","author":"Liu Zheyuan","year":"2021","unstructured":"Zheyuan Liu, Cristian Rodriguez Opazo, Damien Teney, and Stephen Gould. 2021. Image Retrieval on Real-life Images with Pre-trained Vision-and-Language Models. In ICCV. IEEE, 2105--2114."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Pengfei Luo Tong Xu Che Liu Suojuan Zhang Linli Xu Minglei Li and Enhong Chen. 2024. Bridging Gaps in Content and Knowledge for Multimodal Entity Linking. In ACM Multimedia. ACM 9311--9320.","DOI":"10.1145\/3664647.3681661"},{"key":"e_1_3_2_1_46_1","volume-title":"Katherine Hermann, Sean Welleck, Amir Yazdanbakhsh, and Peter Clark.","author":"Madaan Aman","year":"2023","unstructured":"Aman Madaan, Niket Tandon, Prakhar Gupta, Skyler Hallinan, Luyu Gao, Sarah Wiegreffe, Uri Alon, Nouha Dziri, Shrimai Prabhumoye, Yiming Yang, Shashank Gupta, Bodhisattwa Prasad Majumder, Katherine Hermann, Sean Welleck, Amir Yazdanbakhsh, and Peter Clark. 2023. Self-Refine: Iterative Refinement with Self-Feedback. In NeurIPS."},{"key":"e_1_3_2_1_47_1","volume-title":"EMNLP\/IJCNLP (1)","author":"Murahari Vishvak","unstructured":"Vishvak Murahari, Prithvijit Chattopadhyay, Dhruv Batra, Devi Parikh, and Abhishek Das. 2019. Improving Generative Visual Dialog by Answering Diverse Questions. In EMNLP\/IJCNLP (1). Association for Computational Linguistics, 1449--1454."},{"key":"e_1_3_2_1_48_1","volume-title":"ICML (Proceedings of Machine Learning Research","volume":"8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In ICML (Proceedings of Machine Learning Research, Vol. 139). PMLR, 8748--8763."},{"key":"e_1_3_2_1_49_1","unstructured":"Morgane Rivi\u00e8re Shreya Pathak Pier Giuseppe Sessa Cassidy Hardin Surya Bhupatiraju L\u00e9onard Hussenot Thomas Mesnard Bobak Shahriari Alexandre Ram\u00e9 Johan Ferret Peter Liu Pouya Tafti Abe Friesen Michelle Casbon Sabela Ramos Ravin Kumar Charline Le Lan Sammy Jerome Anton Tsitsulin Nino Vieillard Piotr Stanczyk Sertan Girgin Nikola Momchev Matt Hoffman Shantanu Thakoor Jean-Bastien Grill Behnam Neyshabur Olivier Bachem Alanna Walton Aliaksei Severyn Alicia Parrish Aliya Ahmad Allen Hutchison Alvin Abdagic Amanda Carl Amy Shen Andy Brock Andy Coenen Anthony Laforge Antonia Paterson Ben Bastian Bilal Piot Bo Wu Brandon Royal Charlie Chen Chintu Kumar Chris Perry Chris Welty Christopher A. Choquette-Choo Danila Sinopalnikov David Weinberger Dimple Vijaykumar Dominika Rogozinska Dustin Herbison Elisa Bandy Emma Wang Eric Noland Erica Moreira Evan Senter Evgenii Eltyshev Francesco Visin Gabriel Rasskin Gary Wei Glenn Cameron Gus Martins Hadi Hashemi Hanna Klimczak-Plucinska Harleen Batra Harsh Dhand Ivan Nardini Jacinda Mein Jack Zhou James Svensson Jeff Stanway Jetha Chan Jin Peng Zhou Joana Carrasqueira Joana Iljazi Jocelyn Becker Joe Fernandez Joost van Amersfoort Josh Gordon Josh Lipschultz Josh Newlan Ju-yeong Ji Kareem Mohamed Kartikeya Badola Kat Black Katie Millican Keelin McDonell Kelvin Nguyen Kiranbir Sodhia Kish Greene Lars Lowe Sj\u00f6sund Lauren Usui Laurent Sifre Lena Heuermann Leticia Lago and Lilly McNealus. 2024. Gemma 2: Improving Open Language Models at a Practical Size. CoRR Vol. abs\/2408.00118 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval","author":"Saito Kuniaki","year":"1930","unstructured":"Kuniaki Saito, Kihyuk Sohn, Xiang Zhang, Chun-Liang Li, Chen-Yu Lee, Kate Saenko, and Tomas Pfister. 2023. Pic2Word: Mapping Pictures to Words for Zero-shot Composed Image Retrieval. In CVPR. IEEE, 19305--19314."},{"key":"e_1_3_2_1_51_1","volume-title":"Toolformer: Language Models Can Teach Themselves to Use Tools. In NeurIPS.","author":"Schick Timo","year":"2023","unstructured":"Timo Schick, Jane Dwivedi-Yu, Roberto Dess\u00ec, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom. 2023. Toolformer: Language Models Can Teach Themselves to Use Tools. In NeurIPS."},{"key":"e_1_3_2_1_52_1","volume-title":"FLAVA: A Foundational Language And Vision Alignment Model","author":"Singh Amanpreet","year":"2022","unstructured":"Amanpreet Singh, Ronghang Hu, Vedanuj Goswami, Guillaume Couairon, Wojciech Galuba, Marcus Rohrbach, and Douwe Kiela. 2022. FLAVA: A Foundational Language And Vision Alignment Model. In CVPR. IEEE, 15617--15629."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1002\/asi.21659"},{"key":"e_1_3_2_1_54_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In NIPS. 5998--6008."},{"key":"e_1_3_2_1_55_1","volume-title":"Composing Text and Image for Image Retrieval - an Empirical Odyssey","author":"Vo Nam","unstructured":"Nam Vo, Lu Jiang, Chen Sun, Kevin Murphy, Li-Jia Li, Li Fei-Fei, and James Hays. 2019. Composing Text and Image for Image Retrieval - an Empirical Odyssey. In CVPR. Computer Vision Foundation \/ IEEE, 6439--6448."},{"key":"e_1_3_2_1_56_1","volume-title":"Pengcheng Wu, Jianke Zhu, Yongdong Zhang, and Jintao Li.","author":"Wan Ji","year":"2014","unstructured":"Ji Wan, Dayong Wang, Steven Chu-Hong Hoi, Pengcheng Wu, Jianke Zhu, Yongdong Zhang, and Jintao Li. 2014. Deep Learning for Content-Based Image Retrieval: A Comprehensive Study. In ACM Multimedia. ACM, 157--166."},{"key":"e_1_3_2_1_57_1","volume-title":"Quoc V. Le, and Denny Zhou.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed H. Chi, Quoc V. Le, and Denny Zhou. 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In NeurIPS."},{"key":"e_1_3_2_1_58_1","volume-title":"Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback","author":"Wu Hui","year":"2021","unstructured":"Hui Wu, Yupeng Gao, Xiaoxiao Guo, Ziad Al-Halah, Steven Rennie, Kristen Grauman, and Rog\u00e9rio Feris. 2021. Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback. In CVPR. Computer Vision Foundation \/ IEEE, 11307--11317."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.124"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i24.34743"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"crossref","unstructured":"Xiaohui Xie Yiqun Liu Maarten de Rijke Jiyin He Min Zhang and Shaoping Ma. 2018. Why People Search for Images using Web Search Engines. In WSDM. ACM 655--663.","DOI":"10.1145\/3159652.3159686"},{"key":"e_1_3_2_1_62_1","unstructured":"Derong Xu Ziheng Zhang Zhenxi Lin Xian Wu Zhihong Zhu Tong Xu Xiangyu Zhao Yefeng Zheng and Enhong Chen. 2024. Multi-perspective Improvement of Knowledge Graph Completion with Large Language Models. In LREC\/COLING. ELRA and ICCL 11956--11968."},{"key":"e_1_3_2_1_64_1","volume-title":"LDRE: LLM-based Divergent Reasoning and Ensemble for Zero-Shot Composed Image Retrieval. In SIGIR. ACM, 80--90.","author":"Yang Zhenyu","year":"2024","unstructured":"Zhenyu Yang, Dizhan Xue, Shengsheng Qian, Weiming Dong, and Changsheng Xu. 2024a. LDRE: LLM-based Divergent Reasoning and Ensemble for Zero-Shot Composed Image Retrieval. In SIGIR. ACM, 80--90."},{"key":"e_1_3_2_1_65_1","unstructured":"Shunyu Yao Dian Yu Jeffrey Zhao Izhak Shafran Tom Griffiths Yuan Cao and Karthik Narasimhan. 2023. Tree of Thoughts: Deliberate Problem Solving with Large Language Models. In NeurIPS."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_2_1_67_1","article-title":"CoCa: Contrastive Captioners are Image-Text Foundation","volume":"2022","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Zirui Wang, Vijay Vasudevan, Legg Yeung, Mojtaba Seyedhosseini, and Yonghui Wu. 2022. CoCa: Contrastive Captioners are Image-Text Foundation Models. Trans. Mach. Learn. Res., Vol. 2022 (2022).","journal-title":"Models. Trans. Mach. Learn. Res."},{"key":"e_1_3_2_1_68_1","unstructured":"Aohan Zeng Bin Xu Bowen Wang Chenhui Zhang Da Yin Diego Rojas Guanyu Feng Hanlin Zhao Hanyu Lai Hao Yu Hongning Wang Jiadai Sun Jiajie Zhang Jiale Cheng Jiayi Gui Jie Tang Jing Zhang Juanzi Li Lei Zhao Lindong Wu Lucen Zhong Mingdao Liu Minlie Huang Peng Zhang Qinkai Zheng Rui Lu Shuaiqi Duan Shudan Zhang Shulin Cao Shuxun Yang Weng Lam Tam Wenyi Zhao Xiao Liu Xiao Xia Xiaohan Zhang Xiaotao Gu Xin Lv Xinghan Liu Xinyi Liu Xinyue Yang Xixuan Song Xunkai Zhang Yifan An Yifan Xu Yilin Niu Yuantao Yang Yueyan Li Yushi Bai Yuxiao Dong Zehan Qi Zhaoyu Wang Zhen Yang Zhengxiao Du Zhenyu Hou and Zihan Wang. 2024. ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools. CoRR Vol. abs\/2406.12793 (2024)."},{"key":"e_1_3_2_1_69_1","article-title":"Multimodal Chain-of-Thought Reasoning in Language","volume":"2024","author":"Zhang Zhuosheng","year":"2024","unstructured":"Zhuosheng Zhang, Aston Zhang, Mu Li, Hai Zhao, George Karypis, and Alex Smola. 2024. Multimodal Chain-of-Thought Reasoning in Language Models. Trans. Mach. Learn. Res., Vol. 2024 (2024).","journal-title":"Models. Trans. Mach. Learn. Res."},{"key":"e_1_3_2_1_70_1","volume-title":"MAKE: Vision-Language Pre-training based Product Retrieval in Taobao Search. In WWW (Companion Volume). ACM, 356--360.","author":"Zheng Xiaoyang","year":"2023","unstructured":"Xiaoyang Zheng, Zilong Wang, Sen Li, Ke Xu, Tao Zhuang, Qingwen Liu, and Xiaoyi Zeng. 2023. MAKE: Vision-Language Pre-training based Product Retrieval in Taobao Search. In WWW (Companion Volume). ACM, 356--360."},{"key":"e_1_3_2_1_71_1","volume-title":"Chi","author":"Zhou Denny","year":"2023","unstructured":"Denny Zhou, Nathanael Sch\u00e4rli, Le Hou, Jason Wei, Nathan Scales, Xuezhi Wang, Dale Schuurmans, Claire Cui, Olivier Bousquet, Quoc V. Le, and Ed H. Chi. 2023. Least-to-Most Prompting Enables Complex Reasoning in Large Language Models. In ICLR. OpenReview.net."}],"event":{"name":"WWW '25: The ACM Web Conference 2025","location":"Sydney NSW Australia","acronym":"WWW '25","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714777","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696410.3714777","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T21:18:41Z","timestamp":1750281521000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714777"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,22]]},"references-count":70,"alternative-id":["10.1145\/3696410.3714777","10.1145\/3696410"],"URL":"https:\/\/doi.org\/10.1145\/3696410.3714777","relation":{},"subject":[],"published":{"date-parts":[[2025,4,22]]},"assertion":[{"value":"2025-04-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}