{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T15:49:29Z","timestamp":1781884169930,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,9,22]]},"DOI":"10.1145\/3705328.3748064","type":"proceedings-article","created":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T10:46:13Z","timestamp":1757155573000},"page":"482-491","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["VL-CLIP: Enhancing Multimodal Recommendations via Visual Grounding and LLM-Augmented CLIP Embeddings"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5261-4069","authenticated-orcid":false,"given":"Ramin","family":"Giahi","sequence":"first","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7459-795X","authenticated-orcid":false,"given":"Kehui","family":"Yao","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Bellevue, Washington, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6268-4265","authenticated-orcid":false,"given":"Sriram","family":"Kollipara","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1040-0211","authenticated-orcid":false,"given":"Kai","family":"Zhao","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0300-5344","authenticated-orcid":false,"given":"Vahid","family":"Mirjalili","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3702-528X","authenticated-orcid":false,"given":"Jianpeng","family":"Xu","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0239-0606","authenticated-orcid":false,"given":"Topojoy","family":"Biswas","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7754-3652","authenticated-orcid":false,"given":"Evren","family":"Korpeoglu","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9186-3175","authenticated-orcid":false,"given":"Kannan","family":"Achan","sequence":"additional","affiliation":[{"name":"Walmart Global Tech, Sunnyvale, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,9,7]]},"reference":[{"key":"e_1_3_3_3_2_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403311"},{"key":"e_1_3_3_3_3_2","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared\u00a0D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et\u00a0al. 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020) 1877\u20131901."},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"crossref","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Chen Yen-Chun","year":"2020","unstructured":"Yen-Chun Chen, Linjie Li, Licheng Yu, Ahmed El\u00a0Kholy, Faisal Ahmed, Zhe Gan, Yu Cheng, and Jingjing Liu. 2020. UNITER: UNiversal Image-TExt Representation Learning. In Computer Vision \u2013 ECCV 2020, Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.). Springer International Publishing, Cham, 104\u2013120."},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52729.2023.00276"},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"publisher","unstructured":"Patrick\u00a0John Chia Giuseppe Attanasio Federico Bianchi Silvia Terragni Ana\u00a0Rita Magalh\u00a0aes Diogo Goncalves Ciro Greco and Jacopo Tagliabue. 2022. Contrastive language and Vision Learning of General Fashion Concepts. Scientific Reports 12 1 (2022) 18958. 10.1038\/s41598-022-23052-9","DOI":"10.1038\/s41598-022-23052-9"},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-76878-1_6"},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-26438-2_7"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","unstructured":"H. Huang O. Zheng D. Wang et\u00a0al. 2023. ChatGPT for Shaping the Future of Dentistry: The Potential of Multi-modal Large Language Model. International Journal of Oral Science 15 (2023) 29. 10.1038\/s41368-023-00239-y","DOI":"10.1038\/s41368-023-00239-y"},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01064"},{"key":"e_1_3_3_3_12_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.469"},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"publisher","unstructured":"Xiang Li Congcong Wen Yuan Hu and Nan Zhou. 2023. RS-CLIP: Zero-shot Remote Sensing Scene Classification via Contrastive Cision-Language Supervision. International Journal of Applied Earth Observation and Geoinformation 124 (2023) 103497. 10.1016\/j.jag.2023.103497","DOI":"10.1016\/j.jag.2023.103497"},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3701716.3717567"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-63"},{"key":"e_1_3_3_3_17_2","volume-title":"ViLBERT: Pretraining Task-agnostic Visiolinguistic Representations for Vision-and-Language Tasks","author":"Lu Jiasen","year":"2019","unstructured":"Jiasen Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. ViLBERT: Pretraining Task-agnostic Visiolinguistic Representations for Vision-and-Language Tasks. Curran Associates Inc., Red Hook, NY, USA."},{"key":"e_1_3_3_3_18_2","unstructured":"Luyi Ma Xiaohan Li Zezhong Fan Kai Zhao Jianpeng Xu Jason Cho Praveen Kanumala Kaushiki Nag Sushant Kumar and Kannan Achan. 2024. Triple modality fusion: Aligning visual textual and graph data with large language models for multi-behavior recommendations. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.12228 (2024)."},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547910"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"crossref","unstructured":"Yu\u00a0A Malkov and Dmitry\u00a0A Yashunin. 2018. Efficient and Robust Approximate Nearest Neighbor Search using Hierarchical Navigable Small World Graphs. IEEE transactions on pattern analysis and machine intelligence 42 4 (2018) 824\u2013836.","DOI":"10.1109\/TPAMI.2018.2889473"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","unstructured":"Bertalan Mesk\u00f3. 2023. The Impact of Multimodal Large Language Models on Health Care\u2019s Future. Journal of Medical Internet Research 25 (2023) e52865. 10.2196\/52865","DOI":"10.2196\/52865"},{"key":"e_1_3_3_3_22_2","unstructured":"Ron Mokady Amir Hertz and Amit\u00a0H. Bermano. 2021. ClipCap: CLIP Prefix for Image Captioning. arxiv:https:\/\/arXiv.org\/abs\/2111.09734\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2111.09734"},{"key":"e_1_3_3_3_23_2","first-page":"802","volume-title":"Proceedings of the 17th International Conference on Information Systems for Crisis Response and Management (ISCRAM)","author":"Ofli Ferda","year":"2020","unstructured":"Ferda Ofli, Firoj Alam, and Muhammad Imran. 2020. Analysis of Social Media Data using Multimodal Deep Learning for Disaster Response. In Proceedings of the 17th International Conference on Information Systems for Crisis Response and Management (ISCRAM). ISCRAM, 802\u2013811."},{"key":"e_1_3_3_3_24_2","volume-title":"Proceedings of the International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Ceyuan Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pam Mishkin, Jack Clark, et\u00a0al. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_3_3_25_2","series-title":"(NIPS \u201922)","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, Patrick Schramowski, Srivatsa Kundurthy, Katherine Crowson, Ludwig Schmidt, Robert Kaczmarczyk, and Jenia Jitsev. 2022. LAION-5B: An Open Large-scale Dataset for Training Next Generation Image-Text Models. In Proceedings of the 36th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201922). Curran Associates Inc., Red Hook, NY, USA, Article 1833, 17\u00a0pages."},{"key":"e_1_3_3_3_26_2","unstructured":"Christoph Schuhmann Richard Vencu Romain Beaumont Robert Kaczmarczyk Clayton Mullis Aarush Katta Theo Coombes Jenia Jitsev and Aran Komatsuzaki. 2021. Laion-400m: Open Dataset of Clip-filtered 400 Million Image-Text Pairs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.02114 (2021)."},{"key":"e_1_3_3_3_27_2","volume-title":"International Conference on Learning Representations (ICLR)","author":"Su Weijie","year":"2020","unstructured":"Weijie Su, Xizhou Zhu, Yue Cao, Bin Li, Lewei Lu, Furu Wei, and Jifeng Dai. 2020. VL-BERT: Pre-training of Generic Visual-Linguistic Representations. In International Conference on Learning Representations (ICLR). https:\/\/openreview.net\/forum?id=SygXPaEYvH"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"e_1_3_3_3_30_2","volume-title":"International Conference on Learning Representations","author":"Yao Lewei","year":"2022","unstructured":"Lewei Yao, Runhui Huang, Lu Hou, Guansong Lu, Minzhe Niu, Hang Xu, Xiaodan Liang, Zhenguo Li, Xin Jiang, and Chunjing Xu. 2022. FILIP: Fine-grained Interactive Language-Image Pre-Training. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=cpDhcsEDC2"},{"key":"e_1_3_3_3_31_2","unstructured":"Christoph Zauner. 2010. Implementation and Benchmarking of Perceptual Image Hash Functions. (2010)."},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330739"},{"key":"e_1_3_3_3_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3701716.3715227"},{"key":"e_1_3_3_3_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671640"}],"event":{"name":"RecSys '25: Nineteenth ACM Conference on Recommender Systems","location":"Prague Czech Republic","acronym":"RecSys '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction","SIGAI ACM Special Interest Group on Artificial Intelligence","SIGIR ACM Special Interest Group on Information Retrieval","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the Nineteenth ACM Conference on Recommender Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3705328.3748064","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T11:43:27Z","timestamp":1757159007000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3705328.3748064"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,7]]},"references-count":34,"alternative-id":["10.1145\/3705328.3748064","10.1145\/3705328"],"URL":"https:\/\/doi.org\/10.1145\/3705328.3748064","relation":{},"subject":[],"published":{"date-parts":[[2025,9,7]]},"assertion":[{"value":"2025-09-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}