{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:37:31Z","timestamp":1776883051652,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754845","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:56:44Z","timestamp":1761375404000},"page":"2860-2869","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Breaking the Modality Barrier: Universal Embedding Learning with Multimodal LLMs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-2302-246X","authenticated-orcid":false,"given":"Tiancheng","family":"Gu","sequence":"first","affiliation":[{"name":"The University of Sydney, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6073-9014","authenticated-orcid":false,"given":"Kaicheng","family":"Yang","sequence":"additional","affiliation":[{"name":"DeepGlint, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8689-8366","authenticated-orcid":false,"given":"Ziyong","family":"Feng","sequence":"additional","affiliation":[{"name":"DeepGlint, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8543-1390","authenticated-orcid":false,"given":"Xingjun","family":"Wang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6581-7783","authenticated-orcid":false,"given":"Yanzhao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hanghzou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6570-9406","authenticated-orcid":false,"given":"Dingkun","family":"Long","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7358-819X","authenticated-orcid":false,"given":"Yingda","family":"Chen","sequence":"additional","affiliation":[{"name":"Alibaba Group, Bellevue, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3706-8896","authenticated-orcid":false,"given":"Weidong","family":"Cai","sequence":"additional","affiliation":[{"name":"The University of Sydney, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3709-6216","authenticated-orcid":false,"given":"Jiankang","family":"Deng","sequence":"additional","affiliation":[{"name":"Imperial College London, London, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al.","author":"Abdin Marah","year":"2024","unstructured":"Marah Abdin, Jyoti Aneja, Hany Awadalla, Ahmed Awadallah, Ammar Ahmad Awan, Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al., 2024. Phi-3 technical report: A highly capable language model locally on your phone. arXiv:2404.14219 (2024)."},{"key":"e_1_3_2_1_2_1","unstructured":"Jinze Bai Shuai Bai Yunfei Chu Zeyu Cui Kai Dang Xiaodong Deng Yang Fan Wenbin Ge Yu Han Fei Huang et al. 2023. Qwen technical report. arXiv:2309.16609 (2023)."},{"key":"e_1_3_2_1_3_1","first-page":"15338","article-title":"Zero-shot composed image retrieval with textual inversion","author":"Baldrati Alberto","year":"2023","unstructured":"Alberto Baldrati, Lorenzo Agnolucci, Marco Bertini, and Alberto Del Bimbo. 2023. Zero-shot composed image retrieval with textual inversion. In ICCV. 15338-15347.","journal-title":"ICCV."},{"key":"e_1_3_2_1_4_1","volume-title":"Llm2vec: Large language models are secretly powerful text encoders. COLM","author":"BehnamGhader Parishad","year":"2024","unstructured":"Parishad BehnamGhader, Vaibhav Adlakha, Marius Mosbach, Dzmitry Bahdanau, Nicolas Chapados, and Siva Reddy. 2024. Llm2vec: Large language models are secretly powerful text encoders. COLM (2024)."},{"key":"e_1_3_2_1_5_1","first-page":"7734","article-title":"Gallerygpt: Analyzing paintings with large multimodal models","author":"Bin Yi","year":"2024","unstructured":"Yi Bin, Wenhao Shi, Yujuan Ding, Zhiqiang Hu, Zheng Wang, Yang Yang, See-Kiong Ng, and Heng Tao Shen. 2024. Gallerygpt: Analyzing paintings with large multimodal models. In ACMMM. 7734-7743.","journal-title":"ACMMM."},{"key":"e_1_3_2_1_6_1","volume-title":"FLAME: Frozen Large Language Models Enable Data-Efficient Language-Image Pre-training. arXiv:2411.11927","author":"Cao Anjia","year":"2024","unstructured":"Anjia Cao, Xing Wei, and Zhiheng Ma. 2024. FLAME: Frozen Large Language Models Enable Data-Efficient Language-Image Pre-training. arXiv:2411.11927 (2024)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Lin Chen Jinsong Li Xiaoyi Dong Pan Zhang Conghui He Jiaqi Wang Feng Zhao and Dahua Lin. 2024. Sharegpt4v: Improving large multi-modal models with better captions. In ECCV.","DOI":"10.1007\/978-3-031-72643-9_22"},{"key":"e_1_3_2_1_8_1","volume-title":"Reproducible scaling laws for contrastive language-image learning. arXiv:2212.07143","author":"Cherti Mehdi","year":"2022","unstructured":"Mehdi Cherti, Romain Beaumont, Ross Wightman, Mitchell Wortsman, Gabriel Ilharco, Cade Gordon, Christoph Schuhmann, Ludwig Schmidt, and Jenia Jitsev. 2022. Reproducible scaling laws for contrastive language-image learning. arXiv:2212.07143 (2022)."},{"key":"e_1_3_2_1_9_1","volume-title":"Rafael Sampaio De Rezende, Yannis Kalantidis, and Diane Larlus.","author":"Chun Sanghyuk","year":"2021","unstructured":"Sanghyuk Chun, Seong Joon Oh, Rafael Sampaio De Rezende, Yannis Kalantidis, and Diane Larlus. 2021. Probabilistic embeddings for cross-modal retrieval. In CVPR."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Xin Cong Bowen Yu Mengcheng Fang Tingwen Liu Haiyang Yu Zhongkai Hu Fei Huang Yongbin Li and Bin Wang. 2023. Universal Information Extraction with Meta-Pretrained Self-Retrieval. In ACL.","DOI":"10.18653\/v1\/2023.findings-acl.251"},{"key":"e_1_3_2_1_11_1","volume-title":"Qlora: Efficient finetuning of quantized llms. NeurIPS","author":"Dettmers Tim","year":"2023","unstructured":"Tim Dettmers, Artidoro Pagnoni, Ari Holtzman, and Luke Zettlemoyer. 2023. Qlora: Efficient finetuning of quantized llms. NeurIPS (2023)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Aniket Didolkar Andrii Zadaianchuk Rabiul Awal Maximilian Seitzer Efstratios Gavves and Aishwarya Agrawal. 2025. CTRL-O: Language-Controllable Object-Centric Visual Representation Learning. In CVPR.","DOI":"10.1109\/CVPR52734.2025.02749"},{"key":"e_1_3_2_1_13_1","volume-title":"Colpali: Efficient document retrieval with vision language models. In ICLR.","author":"Faysse Manuel","year":"2024","unstructured":"Manuel Faysse, Hugues Sibille, Tony Wu, Bilel Omrani, Gautier Viaud, C\u00e9line Hudelot, and Pierre Colombo. 2024. Colpali: Efficient document retrieval with vision language models. In ICLR."},{"key":"e_1_3_2_1_14_1","volume-title":"Scaling deep contrastive learning batch size under memory limited setup. arXiv:2101.06983","author":"Gao Luyu","year":"2021","unstructured":"Luyu Gao, Yunyi Zhang, Jiawei Han, and Jamie Callan. 2021b. Scaling deep contrastive learning batch size under memory limited setup. arXiv:2101.06983 (2021)."},{"key":"e_1_3_2_1_15_1","volume-title":"Simcse: Simple contrastive learning of sentence embeddings. arXiv:2104.08821","author":"Gao Tianyu","year":"2021","unstructured":"Tianyu Gao, Xingcheng Yao, and Danqi Chen. 2021a. Simcse: Simple contrastive learning of sentence embeddings. arXiv:2104.08821 (2021)."},{"key":"e_1_3_2_1_16_1","volume-title":"Conceptbert: Concept-aware representation for visual question answering. In EMNLP.","author":"Gard\u00e8res Fran\u00e7ois","year":"2020","unstructured":"Fran\u00e7ois Gard\u00e8res, Maryam Ziaeefard, Baptiste Abeloos, and Freddy Lecue. 2020. Conceptbert: Concept-aware representation for visual question answering. In EMNLP."},{"key":"e_1_3_2_1_17_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"Rwkv-clip: A robust vision-language representation learner. In EMNLP.","author":"Gu Tiancheng","year":"2024","unstructured":"Tiancheng Gu, Kaicheng Yang, Xiang An, Ziyong Feng, Dongnan Liu, Weidong Cai, and Jiankang Deng. 2024. Rwkv-clip: A robust vision-language representation learner. In EMNLP."},{"key":"e_1_3_2_1_19_1","volume-title":"Sugarcrepe: Fixing hackable benchmarks for vision-language compositionality. NeurIPS","author":"Hsieh Cheng-Yu","year":"2023","unstructured":"Cheng-Yu Hsieh, Jieyu Zhang, Zixian Ma, Aniruddha Kembhavi, and Ranjay Krishna. 2023. Sugarcrepe: Fixing hackable benchmarks for vision-language compositionality. NeurIPS (2023)."},{"key":"e_1_3_2_1_20_1","volume-title":"Decoupled Global-Local Alignment for Improving Compositional Understanding. arXiv preprint arXiv:2504.16801","author":"Hu Xiaoxing","year":"2025","unstructured":"Xiaoxing Hu, Kaicheng Yang, Jun Wang, Haoran Xu, Ziyong Feng, and Yupei Wang. 2025. Decoupled Global-Local Alignment for Improving Compositional Understanding. arXiv preprint arXiv:2504.16801 (2025)."},{"key":"e_1_3_2_1_21_1","unstructured":"Weiquan Huang Aoqi Wu Yifan Yang Xufang Luo Yuqing Yang Liang Hu Qi Dai Xiyang Dai Dongdong Chen Chong Luo et al. 2024. Llm2clip: Powerful language model unlock richer visual representation. arXiv:2411.04997 (2024)."},{"key":"e_1_3_2_1_22_1","first-page":"525","article-title":"Hal-eval: A universal and fine-grained hallucination evaluation framework for large vision language models","author":"Jiang Chaoya","year":"2024","unstructured":"Chaoya Jiang, Hongrui Jia, Mengfan Dong, Wei Ye, Haiyang Xu, Ming Yan, Ji Zhang, and Shikun Zhang. 2024a. Hal-eval: A universal and fine-grained hallucination evaluation framework for large vision language models. In ACMMM. 525-534.","journal-title":"ACMMM."},{"key":"e_1_3_2_1_23_1","volume-title":"E5-v: Universal embeddings with multimodal large language models. arXiv:2407.12580","author":"Jiang Ting","year":"2024","unstructured":"Ting Jiang, Minghui Song, Zihan Zhang, Haizhen Huang, Weiwei Deng, Feng Sun, Qi Zhang, Deqing Wang, and Fuzhen Zhuang. 2024b. E5-v: Universal embeddings with multimodal large language models. arXiv:2407.12580 (2024)."},{"key":"e_1_3_2_1_24_1","volume-title":"Vlm2vec: Training vision-language models for massive multimodal embedding tasks. ICLR","author":"Jiang Ziyan","year":"2025","unstructured":"Ziyan Jiang, Rui Meng, Xinyi Yang, Semih Yavuz, Yingbo Zhou, and Wenhu Chen. 2025. Vlm2vec: Training vision-language models for massive multimodal embedding tasks. ICLR (2025)."},{"key":"e_1_3_2_1_25_1","first-page":"7969","article-title":"Active retrieval augmented generation","author":"Jiang Zhengbao","year":"2023","unstructured":"Zhengbao Jiang, Frank F Xu, Luyu Gao, Zhiqing Sun, Qian Liu, Jane Dwivedi-Yu, Yiming Yang, Jamie Callan, and Graham Neubig. 2023. Active retrieval augmented generation. In EMNLP. 7969-7992.","journal-title":"EMNLP."},{"key":"e_1_3_2_1_26_1","volume-title":"Noe Pion, Philippe Weinzaepfel, and Diane Larlus.","author":"Kalantidis Yannis","year":"2020","unstructured":"Yannis Kalantidis, Mert Bulent Sariyildiz, Noe Pion, Philippe Weinzaepfel, and Diane Larlus. 2020. Hard negative mixing for contrastive learning. NeurIPS (2020)."},{"key":"e_1_3_2_1_27_1","volume-title":"Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih.","author":"Karpukhin Vladimir","year":"2020","unstructured":"Vladimir Karpukhin, Barlas Oguz, Sewon Min, Patrick SH Lewis, Ledell Wu, Sergey Edunov, Danqi Chen, and Wen-tau Yih. 2020. Dense Passage Retrieval for Open-Domain Question Answering.. In EMNLP."},{"key":"e_1_3_2_1_28_1","volume-title":"Nv-embed: Improved techniques for training llms as generalist embedding models. ICLR","author":"Lee Chankyu","year":"2024","unstructured":"Chankyu Lee, Rajarshi Roy, Mengyao Xu, Jonathan Raiman, Mohammad Shoeybi, Bryan Catanzaro, and Wei Ping. 2024. Nv-embed: Improved techniques for training llms as generalist embedding models. ICLR (2024)."},{"key":"e_1_3_2_1_29_1","volume-title":"Llava-onevision: Easy visual task transfer. arXiv:2408.03326","author":"Li Bo","year":"2024","unstructured":"Bo Li, Yuanhan Zhang, Dong Guo, Renrui Zhang, Feng Li, Hao Zhang, Kaichen Zhang, Peiyuan Zhang, Yanwei Li, Ziwei Liu, et al., 2024. Llava-onevision: Easy visual task transfer. arXiv:2408.03326 (2024)."},{"key":"e_1_3_2_1_30_1","unstructured":"Junnan Li Dongxu Li Silvio Savarese and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In ICML."},{"key":"e_1_3_2_1_31_1","volume-title":"Vila: On pre-training for visual language models. In CVPR.","author":"Lin Ji","year":"2024","unstructured":"Ji Lin, Hongxu Yin, Wei Ping, Pavlo Molchanov, Mohammad Shoeybi, and Song Han. 2024. Vila: On pre-training for visual language models. In CVPR."},{"key":"e_1_3_2_1_32_1","unstructured":"Tsung-Yi Lin Michael Maire Serge Belongie James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1r and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In ECCV."},{"key":"e_1_3_2_1_33_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li and Yong Jae Lee. 2024b. Improved baselines with visual instruction tuning. In CVPR."},{"key":"e_1_3_2_1_34_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong Jae Lee. 2024c. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"e_1_3_2_1_35_1","volume-title":"Visual instruction tuning. NeurIPS","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong Jae Lee. 2023. Visual instruction tuning. NeurIPS (2023)."},{"key":"e_1_3_2_1_36_1","volume-title":"LamRA: Large Multimodal Model as Your Advanced Retrieval Assistant. CVPR","author":"Liu Yikun","year":"2024","unstructured":"Yikun Liu, Pingan Chen, Jiayin Cai, Xiaolong Jiang, Yao Hu, Jiangchao Yao, Yanfeng Wang, and Weidi Xie. 2024a. LamRA: Large Multimodal Model as Your Advanced Retrieval Assistant. CVPR (2024)."},{"key":"e_1_3_2_1_37_1","unstructured":"Xueguang Ma Liang Wang Nan Yang Furu Wei and Jimmy Lin. 2024. Fine-tuning llama for multi-stage text retrieval. In SIGIR."},{"key":"e_1_3_2_1_38_1","volume-title":"Ullme: A unified framework for large language model embeddings with generation-augmented learning. In EMNLP.","author":"Man Hieu","year":"2024","unstructured":"Hieu Man, Nghia Ngo, Franck Dernoncourt, and Thien Nguyen. 2024. Ullme: A unified framework for large language model embeddings with generation-augmented learning. In EMNLP."},{"key":"e_1_3_2_1_39_1","volume-title":"MTEB: Massive Text Embedding Benchmark. arXiv:2210.07316","author":"Muennighoff Niklas","year":"2022","unstructured":"Niklas Muennighoff, Nouamane Tazi, Lo\u00efc Magne, and Nils Reimers. 2022. MTEB: Massive Text Embedding Benchmark. arXiv:2210.07316 (2022)."},{"key":"e_1_3_2_1_40_1","volume-title":"Representation learning with contrastive predictive coding. arXiv:1807.03748","author":"van den Oord Aaron","year":"2018","unstructured":"Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv:1807.03748 (2018)."},{"key":"e_1_3_2_1_41_1","volume-title":"Brais Martinez, and Georgios Tzimiropoulos.","author":"Ouali Yassine","year":"2024","unstructured":"Yassine Ouali, Adrian Bulat, Alexandros Xenos, Anestis Zaganidis, Ioannis Maniadis Metaxas, Brais Martinez, and Georgios Tzimiropoulos. 2024. Discriminative Fine-tuning of LVLMs. arXiv:2412.04378 (2024)."},{"key":"e_1_3_2_1_42_1","volume-title":"Kosmos-2: Grounding multimodal large language models to the world. arXiv:2306.14824","author":"Peng Zhiliang","year":"2023","unstructured":"Zhiliang Peng, Wenhui Wang, Li Dong, Yaru Hao, Shaohan Huang, Shuming Ma, and Furu Wei. 2023. Kosmos-2: Grounding multimodal large language models to the world. arXiv:2306.14824 (2023)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Bryan A Plummer Liwei Wang Chris M Cervantes Juan C Caicedo Julia Hockenmaier and Svetlana Lazebnik. 2015. Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. In ICCV.","DOI":"10.1109\/ICCV.2015.303"},{"key":"e_1_3_2_1_44_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In ICML."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"e_1_3_2_1_46_1","unstructured":"Joshua David Robinson Ching-Yao Chuang Suvrit Sra and Stefanie Jegelka. 2020. Contrastive Learning with Hard Negative Samples. In ICLR."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Jungkyoo Shin Bumsoo Kim and Eunwoo Kim. 2025. Generative Modeling of Class Probability for Multi-Modal Representation Learning. In CVPR.","DOI":"10.1109\/CVPR52734.2025.01931"},{"key":"e_1_3_2_1_48_1","volume-title":"Notes on kullback-leibler divergence and likelihood. arXiv:1404.2000","author":"Shlens Jonathon","year":"2014","unstructured":"Jonathon Shlens. 2014. Notes on kullback-leibler divergence and likelihood. arXiv:1404.2000 (2014)."},{"key":"e_1_3_2_1_49_1","volume-title":"Eva-clip: Improved training techniques for clip at scale. arXiv:2303.15389","author":"Sun Quan","year":"2023","unstructured":"Quan Sun, Yuxin Fang, Ledell Wu, Xinlong Wang, and Yue Cao. 2023a. Eva-clip: Improved training techniques for clip at scale. arXiv:2303.15389 (2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"EVA-CLIP-18B: Scaling CLIP to 18 Billion Parameters. arXiv:2402.04252","author":"Sun Quan","year":"2023","unstructured":"Quan Sun, Jinsheng Wang, Qiying Yu, Yufeng Cui, Fan Zhang, Xiaosong Zhang, and Xinlong Wang. 2023b. EVA-CLIP-18B: Scaling CLIP to 18 Billion Parameters. arXiv:2402.04252 (2023)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"crossref","unstructured":"Yuanmin Tang Jing Yu Keke Gai Jiamin Zhuang Gang Xiong Gaopeng Gou and Qi Wu. 2025. Missing Target-Relevant Information Prediction with World Model for Accurate Zero-Shot Composed Image Retrieval. In CVPR.","DOI":"10.1109\/CVPR52734.2025.02308"},{"key":"e_1_3_2_1_52_1","volume-title":"Llama: Open and efficient foundation language models. arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023a. Llama: Open and efficient foundation language models. arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_53_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023b. Llama 2: Open foundation and fine-tuned chat models. arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_54_1","volume-title":"Image captioners are scalable vision learners too. NeurIPS","author":"Tschannen Michael","year":"2023","unstructured":"Michael Tschannen, Manoj Kumar, Andreas Steiner, Xiaohua Zhai, Neil Houlsby, and Lucas Beyer. 2023. Image captioners are scalable vision learners too. NeurIPS (2023)."},{"key":"e_1_3_2_1_55_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge et al. 2024a. Qwen2-vl: Enhancing vision-language model's perception of the world at any resolution. arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_56_1","first-page":"7803","article-title":"Large multi-modality model assisted ai-generated image quality assessment","author":"Wang Puyi","year":"2024","unstructured":"Puyi Wang, Wei Sun, Zicheng Zhang, Jun Jia, Yanwei Jiang, Zhichao Zhang, Xiongkuo Min, and Guangtao Zhai. 2024b. Large multi-modality model assisted ai-generated image quality assessment. In ACMMM. 7803-7812.","journal-title":"ACMMM."},{"key":"e_1_3_2_1_57_1","unstructured":"Weihan Wang Qingsong Lv Wenmeng Yu Wenyi Hong Ji Qi Yan Wang Junhui Ji Zhuoyi Yang Lei Zhao Xixuan Song Jiazheng Xu Bin Xu Juanzi Li Yuxiao Dong Ming Ding and Jie Tang. 2023. CogVLM: Visual Expert for Pretrained Language Models. arXiv:2311.03079 [cs.CV]"},{"key":"e_1_3_2_1_58_1","volume-title":"Uniir: Training and benchmarking universal multimodal information retrievers","author":"Wei Cong","year":"2024","unstructured":"Cong Wei, Yang Chen, Haonan Chen, Hexiang Hu, Ge Zhang, Jie Fu, Alan Ritter, and Wenhu Chen. 2024. Uniir: Training and benchmarking universal multimodal information retrievers. In ECCV. Springer, 387-404."},{"key":"e_1_3_2_1_59_1","volume-title":"Weihao Wang, Kevin Qinghong Lin, Yuchao Gu, Zhijie Chen, Zhenheng Yang, and Mike Zheng Shou.","author":"Xie Jinheng","year":"2024","unstructured":"Jinheng Xie, Weijia Mao, Zechen Bai, David Junhao Zhang, Weihao Wang, Kevin Qinghong Lin, Yuchao Gu, Zhijie Chen, Zhenheng Yang, and Mike Zheng Shou. 2024a. Show-o: One single transformer to unify multimodal understanding and generation. arXiv:2408.12528 (2024)."},{"key":"e_1_3_2_1_60_1","volume-title":"Croc: Pretraining Large Multimodal Models with Cross-Modal Comprehension. arXiv:2410.14332","author":"Xie Yin","year":"2024","unstructured":"Yin Xie, Kaicheng Yang, Ninghua Yang, Weimo Deng, Xiangzi Dai, Tiancheng Gu, Yumeng Wang, Xiang An, Yongle Zhao, Ziyong Feng, et al., 2024b. Croc: Pretraining Large Multimodal Models with Cross-Modal Comprehension. arXiv:2410.14332 (2024)."},{"key":"e_1_3_2_1_61_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei et al. 2024. Qwen2. 5 technical report. arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_62_1","first-page":"2922","article-title":"Alip: Adaptive language-image pre-training with synthetic caption","author":"Yang Kaicheng","year":"2023","unstructured":"Kaicheng Yang, Jiankang Deng, Xiang An, Jiawei Li, Ziyong Feng, Jia Guo, Jing Yang, and Tongliang Liu. 2023. Alip: Adaptive language-image pre-training with synthetic caption. In ICCV. 2922-2931.","journal-title":"ICCV."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i20.35505"},{"key":"e_1_3_2_1_64_1","volume-title":"When and why vision-language models behave like bags-of-words, and what to do about it? arXiv:2210.01936","author":"Yuksekgonul Mert","year":"2022","unstructured":"Mert Yuksekgonul, Federico Bianchi, Pratyusha Kalluri, Dan Jurafsky, and James Zou. 2022. When and why vision-language models behave like bags-of-words, and what to do about it? arXiv:2210.01936 (2022)."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"crossref","unstructured":"Xiaohua Zhai Basil Mustafa Alexander Kolesnikov and Lucas Beyer. 2023. Sigmoid loss for language image pre-training. In ICCV.","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"e_1_3_2_1_66_1","volume-title":"Long-clip: Unlocking the long-text capability of clip. In ECCV.","author":"Zhang Beichen","year":"2024","unstructured":"Beichen Zhang, Pan Zhang, Xiaoyi Dong, Yuhang Zang, and Jiaqi Wang. 2024c. Long-clip: Unlocking the long-text capability of clip. In ECCV."},{"key":"e_1_3_2_1_67_1","volume-title":"Llava-grounding: Grounded visual chat with large multimodal models. In ECCV.","author":"Zhang Hao","year":"2024","unstructured":"Hao Zhang, Hongyang Li, Feng Li, Tianhe Ren, Xueyan Zou, Shilong Liu, Shijia Huang, Jianfeng Gao, Leizhang, Chunyuan Li, et al., 2024a. Llava-grounding: Grounded visual chat with large multimodal models. In ECCV."},{"key":"e_1_3_2_1_68_1","volume-title":"Magiclens: Self-supervised image retrieval with open-ended instructions. arXiv:2403.19651","author":"Zhang Kai","year":"2024","unstructured":"Kai Zhang, Yi Luan, Hexiang Hu, Kenton Lee, Siyuan Qiao, Wenhu Chen, Yu Su, and Ming-Wei Chang. 2024b. Magiclens: Self-supervised image retrieval with open-ended instructions. arXiv:2403.19651 (2024)."},{"key":"e_1_3_2_1_69_1","volume-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv:2304.10592","author":"Zhu Deyao","year":"2023","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2023. Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv:2304.10592 (2023)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754845","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:42:47Z","timestamp":1765309367000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754845"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":69,"alternative-id":["10.1145\/3746027.3754845","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754845","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}