{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T16:02:47Z","timestamp":1784131367447,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,10]]},"DOI":"10.1145\/3746252.3761170","type":"proceedings-article","created":{"date-parts":[[2025,11,8]],"date-time":"2025-11-08T00:36:36Z","timestamp":1762562196000},"page":"1788-1797","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["UniECS: Unified Multimodal E-Commerce Search Framework with Gated Cross-modal Fusion"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-2373-1439","authenticated-orcid":false,"given":"Zihan","family":"Liang","sequence":"first","affiliation":[{"name":"Kuaishou Technology, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5075-0099","authenticated-orcid":false,"given":"Yufei","family":"Ma","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8066-8188","authenticated-orcid":false,"given":"Zhipeng","family":"Qian","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3844-8359","authenticated-orcid":false,"given":"Huangyu","family":"Dai","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, HangZhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2032-4762","authenticated-orcid":false,"given":"Zihan","family":"Wang","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4495-8686","authenticated-orcid":false,"given":"Ben","family":"Chen","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6287-3673","authenticated-orcid":false,"given":"Chenyi","family":"Lei","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Hangzhou, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8092-4646","authenticated-orcid":false,"given":"Yuqing","family":"Ding","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9801-9292","authenticated-orcid":false,"given":"Han","family":"Li","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,10]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Pravesh Agrawal Szymon Antoniak Emma Bou Hanna Baptiste Bout Devendra Chaplot Jessica Chudnovsky Diogo Costa Baudouin De Monicault Saurabh Garg Theophile Gervet Soham Ghosh Am\u00e9lie H\u00e9liou Paul Jacob Albert Q. Jiang Kartik Khandelwal Timoth\u00e9e Lacroix Guillaume Lample Diego Las Casas Thibaut Lavril Teven Le Scao Andy Lo William Marshall Louis Martin Arthur Mensch Pavankumar Muddireddy Valera Nemychnikova Marie Pellat Patrick Von Platen Nikhil Raghuraman Baptiste Rozi\u00e8re Alexandre Sablayrolles Lucile Saulnier Romain Sauvestre Wendy Shang Roman Soletskyi Lawrence Stewart Pierre Stock Joachim Studnia Sandeep Subramanian Sagar Vaze Thomas Wang and Sophia Yang. 2024. Pixtral 12B. arXiv:2410.07073 [cs.CV]"},{"key":"e_1_3_2_1_2_1","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katie Millican Malcolm Reynolds Roman Ring Eliza Rutherford Serkan Cabi Tengda Han Zhitao Gong Sina Samangooei Marianne Monteiro Jacob Menick Sebastian Borgeaud Andrew Brock Aida Nematzadeh Sahand Sharifzadeh Mikolaj Binkowski Ricardo Barreira Oriol Vinyals Andrew Zisserman and Karen Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. arXiv:2204.14198 [cs.CV]"},{"key":"e_1_3_2_1_3_1","volume-title":"Unified Vision-Language Representation Modeling for E-Commerce Same-style Products Retrieval. In Companion Proceedings of the ACM Web Conference 2023","author":"Chen Ben","year":"2023","unstructured":"Ben Chen, Linbo Jin, Xinxin Wang, Dehong Gao, Wen Jiang, and Wei Ning. 2023. Unified Vision-Language Representation Modeling for E-Commerce Same-style Products Retrieval. In Companion Proceedings of the ACM Web Conference 2023 (Austin, TX, USA). 381-385."},{"key":"e_1_3_2_1_4_1","volume-title":"NVLM: Open Frontier-Class Multimodal LLMs. arXiv:2409.11402 [cs.CL]","author":"Dai Wenliang","year":"2024","unstructured":"Wenliang Dai, Nayeon Lee, Boxin Wang, Zhuolin Yang, Zihan Liu, Jon Barker, Tuomas Rintamaki, Mohammad Shoeybi, Bryan Catanzaro, and Wei Ping. 2024. NVLM: Open Frontier-Class Multimodal LLMs. arXiv:2409.11402 [cs.CL]"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Deng Xiang","year":"2023","unstructured":"Xiang Deng, Yu Gu, Boyuan Zheng, Shijie Chen, Samuel Stevens, Boshi Wang, Huan Sun, and Yu Su. 2023. MIND2WEB: towards a generalist agent for the web. In Proceedings of the 37th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS '23)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.100"},{"key":"e_1_3_2_1_7_1","volume-title":"Automatic Spatially-Aware Fashion Concept Discovery. In 2017 IEEE International Conference on Computer Vision (ICCV). 1472-1480","author":"Han Xintong","unstructured":"Xintong Han, Zuxuan Wu, Phoenix X. Huang, Xiao Zhang, Menglong Zhu, Yuan Li, Yang Zhao, and Larry S. Davis. 2017. Automatic Spatially-Aware Fashion Concept Discovery. In 2017 IEEE International Conference on Computer Vision (ICCV). 1472-1480."},{"key":"e_1_3_2_1_8_1","volume-title":"MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework. In The Twelfth International Conference on Learning Representations.","author":"Hong Sirui","year":"2024","unstructured":"Sirui Hong, Mingchen Zhuge, Jonathan Chen, Xiawu Zheng, Yuheng Cheng, Jinlin Wang, Ceyao Zhang, Zili Wang, Steven Ka Shing Yau, Zijuan Lin, Liyang Zhou, Chenyu Ran, Lingfeng Xiao, Chenglin Wu, and J\u00fcrgen Schmidhuber. 2024. MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_9_1","unstructured":"Chao Jia Yinfei Yang Ye Xia Yi-Ting Chen Zarana Parekh Hieu Pham Quoc V. Le Yunhsuan Sung Zhen Li and Tom Duerig. 2021. Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision. arXiv:2102.05918 [cs.CV]"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"Ting Jiang Minghui Song Zihan Zhang Haizhen Huang Weiwei Deng Feng Sun Qi Zhang Deqing Wang and Fuzhen Zhuang. 2024. E5-V: Universal Embeddings with Multimodal Large Language Models. arXiv:2407.12580 [cs.CL]","DOI":"10.18653\/v1\/2024.findings-emnlp.181"},{"key":"e_1_3_2_1_11_1","volume-title":"Proxy Anchor Loss for Deep Metric Learning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 3235-3244","author":"Kim Sungyeon","year":"2020","unstructured":"Sungyeon Kim, Dongwon Kim, Minsu Cho, and Suha Kwak. 2020. Proxy Anchor Loss for Deep Metric Learning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 3235-3244."},{"key":"e_1_3_2_1_12_1","volume-title":"MDAgents: An Adaptive Collaboration of LLMs for Medical Decision-Making. In The Thirty-eighth Annual Conference on Neural Information Processing Systems.","author":"Kim Yubin","year":"2024","unstructured":"Yubin Kim, Chanwoo Park, Hyewon Jeong, Yik Siu Chan, Xuhai Xu, Daniel McDuff, Hyeonhoon Lee, Marzyeh Ghassemi, Cynthia Breazeal, and Hae Won Park. 2024. MDAgents: An Adaptive Collaboration of LLMs for Medical Decision-Making. In The Thirty-eighth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_13_1","first-page":"0","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In Proceedings of the 40th International Conference on Machine Learning (Honolulu, Hawaii, USA). 19730-19742."},{"key":"e_1_3_2_1_14_1","volume-title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. arXiv:2201.12086 [cs.CV]","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. arXiv:2201.12086 [cs.CV]"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"Li Junnan","unstructured":"Junnan Li, Ramprasaath R. Selvaraju, Akhilesh D. Gotmare, Shafiq Joty, Caiming Xiong, and Steven C.H. Hoi. 2021. Align before fuse: vision and language representation learning with momentum distillation. In Proceedings of the 35th International Conference on Neural Information Processing Systems. Red Hook, NY, USA."},{"key":"e_1_3_2_1_16_1","first-page":"3030","volume-title":"Self-Renewal Prompt Optimizing with Implicit Reasoning. In Findings of the Association for Computational Linguistics: EMNLP","author":"Liang Zihan","year":"2024","unstructured":"Zihan Liang, Ben Chen, Zhuoran Ran, Zihan Wang, Huangyu Dai, Yufei Ma, Dehong Gao, Xiaoyan Cai, and Libin Yang. 2024. Self-Renewal Prompt Optimizing with Implicit Reasoning. In Findings of the Association for Computational Linguistics: EMNLP 2024. Miami, Florida, USA, 3030-3041."},{"key":"e_1_3_2_1_17_1","volume-title":"MM-EMBED: UNIVERSAL MULTIMODAL RETRIEVAL WITH MULTIMODAL LLMS. In The Thirteenth International Conference on Learning Representations.","author":"Lin Sheng-Chieh","year":"2025","unstructured":"Sheng-Chieh Lin, Chankyu Lee, Mohammad Shoeybi, Jimmy Lin, Bryan Catanzaro, and Wei Ping. 2025. MM-EMBED: UNIVERSAL MULTIMODAL RETRIEVAL WITH MULTIMODAL LLMS. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_18_1","volume-title":"Universal Vision-Language Dense Retrieval: Learning A Unified Representation Space for Multi-Modal Retrieval. In The Eleventh International Conference on Learning Representations.","author":"Liu Zhenghao","year":"2023","unstructured":"Zhenghao Liu, Chenyan Xiong, Yuanhuiyi Lv, Zhiyuan Liu, and Ge Yu. 2023. Universal Vision-Language Dense Retrieval: Learning A Unified Representation Space for Multi-Modal Retrieval. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Mao Anqi","year":"2023","unstructured":"Anqi Mao, Mehryar Mohri, and Yutao Zhong. 2023. Cross-entropy loss functions: theoretical analysis and applications. In Proceedings of the 40th International Conference on Machine Learning (Honolulu, Hawaii, USA)."},{"key":"e_1_3_2_1_20_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. arXiv:2103.00020 [cs.CV]"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"e_1_3_2_1_22_1","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Yang Fan Kai Dang Mengfei Du Xuancheng Ren Rui Men Dayiheng Liu Chang Zhou Jingren Zhou and Junyang Lin. 2024. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv:2409.12191 [cs.CV]"},{"key":"e_1_3_2_1_23_1","volume-title":"Computer Vision - ECCV 2024: 18th European Conference (Milan, Italy). 387-404.","author":"Wei Cong","unstructured":"Cong Wei, Yang Chen, Haonan Chen, Hexiang Hu, Ge Zhang, Jie Fu, Alan Ritter, and Wenhu Chen. 2024. UniIR: Training and Benchmarking Universal Multimodal Information Retrievers. In Computer Vision - ECCV 2024: 18th European Conference (Milan, Italy). 387-404."},{"key":"e_1_3_2_1_24_1","volume-title":"Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback. arXiv:1905.12794 [cs.CV]","author":"Wu Hui","year":"2020","unstructured":"Hui Wu, Yupeng Gao, Xiaoxiao Guo, Ziad Al-Halah, Steven Rennie, Kristen Grauman, and Rogerio Feris. 2020. Fashion IQ: A New Dataset Towards Retrieving Images by Natural Language Feedback. arXiv:1905.12794 [cs.CV]"},{"key":"e_1_3_2_1_25_1","unstructured":"Zhiyu Wu Xiaokang Chen Zizheng Pan Xingchao Liu Wen Liu Damai Dai Huazuo Gao Yiyang Ma Chengyue Wu Bingxuan Wang Zhenda Xie Yu Wu Kai Hu Jiawei Wang Yaofeng Sun Yukun Li Yishi Piao Kang Guan Aixin Liu Xin Xie Yuxiang You Kai Dong Xingkai Yu Haowei Zhang Liang Zhao Yisong Wang and Chong Ruan. 2024. DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding. arXiv:2412.10302 [cs.CV]"},{"key":"e_1_3_2_1_26_1","unstructured":"Zhengyuan Yang Linjie Li Kevin Lin Jianfeng Wang Chung-Ching Lin Zicheng Liu and Lijuan Wang. 2023. The Dawn of LMMs: Preliminary Explorations with GPT-4V(ision). arXiv:2309.17421 [cs.CV]"},{"key":"e_1_3_2_1_27_1","volume-title":"ReAct: Synergizing Reasoning and Acting in Language Models. In The Eleventh International Conference on Learning Representations.","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik R Narasimhan, and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_28_1","volume-title":"GME: Improving Universal Multimodal Retrieval by Multimodal LLMs. arXiv:2412.16855 [cs.CL]","author":"Zhang Xin","year":"2024","unstructured":"Xin Zhang, Yanzhao Zhang, Wen Xie, Mingxin Li, Ziqi Dai, Dingkun Long, Pengjun Xie, Meishan Zhang, Wenjie Li, and Min Zhang. 2024. GME: Improving Universal Multimodal Retrieval by Multimodal LLMs. arXiv:2412.16855 [cs.CL]"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29936"},{"key":"e_1_3_2_1_30_1","volume-title":"Defu Lian, and Yongping Xiong.","author":"Zhou Junjie","year":"2024","unstructured":"Junjie Zhou, Zheng Liu, Ze Liu, Shitao Xiao, Yueze Wang, Bo Zhao, Chen Jason Zhang, Defu Lian, and Yongping Xiong. 2024a. MegaPairs: Massive Data Synthesis For Universal Multimodal Retrieval. arXiv:2412.14475 [cs.CV]"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.175"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.783"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671640"}],"event":{"name":"CIKM '25: The 34th ACM International Conference on Information and Knowledge Management","location":"Seoul Republic of Korea","acronym":"CIKM '25","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the 34th ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746252.3761170","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T02:05:20Z","timestamp":1765505120000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746252.3761170"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,10]]},"references-count":33,"alternative-id":["10.1145\/3746252.3761170","10.1145\/3746252"],"URL":"https:\/\/doi.org\/10.1145\/3746252.3761170","relation":{},"subject":[],"published":{"date-parts":[[2025,11,10]]},"assertion":[{"value":"2025-11-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}