{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:10:14Z","timestamp":1784139014775,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808442","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"4880-4885","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Preliminary Study of an Evaluation Benchmark for Vision\u2013Language Models in Fashion E-Commerce"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4841-1824","authenticated-orcid":false,"given":"Ryotaro","family":"Shimizu","sequence":"first","affiliation":[{"name":"ZOZO Research, Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8803-5842","authenticated-orcid":false,"given":"Sai","family":"Htaungkham","sequence":"additional","affiliation":[{"name":"ZOZO Research, Tokyo, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4828-8555","authenticated-orcid":false,"given":"Shion","family":"Sakurai","sequence":"additional","affiliation":[{"name":"ZOZO, Inc., Chiba, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6738-0274","authenticated-orcid":false,"given":"Yuki","family":"Shimizu","sequence":"additional","affiliation":[{"name":"ZOZO, Inc., Chiba, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.744"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_2_1_3_1","volume-title":"Localization, Text Reading, and Beyond. arXiv preprint arXiv:2308.12966","author":"Bai Jinze","year":"2023","unstructured":"Jinze Bai, Shuai Bai, Shusheng Yang, Shijie Wang, Sinan Tan, Peng Wang, Junyang Lin, Chang Zhou, and Jingren Zhou. 2023. Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond. arXiv preprint arXiv:2308.12966 (2023)."},{"key":"e_1_3_2_1_4_1","volume-title":"arXiv preprint arXiv:2502.13923","author":"Bai Shuai","year":"2025","unstructured":"Shuai Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Sibo Song, Kai Dang, Peng Wang, Shijie Wang, Jun Tang, Humen Zhong, Yuanzhi Zhu, Mingkun Yang, Zhaohai Li, Jianqiang Wan, Pengfei Wang, Wei Ding, Zheren Fu, Yiheng Xu, Jiabo Ye, Xi Zhang, Tianbao Xie, Zesen Cheng, Hang Zhang, Zhibo Yang, Haiyang Xu, and Junyang Lin. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_5_1","first-page":"1877","article-title":"Language Models are Few-Shot Learners","volume":"33","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. In Advances in Neural Information Processing Systems, Vol. 33. 1877-1901.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_6_1","volume-title":"Junqi Zhao, Weisheng Wang, Boyang Li, Pascale Fung, and Steven Hoi.","author":"Dai Wenliang","year":"2023","unstructured":"Wenliang Dai, Junnan Li, Dongxu Li, Anthony Meng Huat Tiong, Junqi Zhao, Weisheng Wang, Boyang Li, Pascale Fung, and Steven Hoi. 2023. InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning. In Advances in Neural Information Processing Systems, Vol. 36."},{"key":"e_1_3_2_1_7_1","volume-title":"International Conference on Learning Representations.","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_8_1","volume-title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. In Advances of Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Fu Chaoyou","year":"2025","unstructured":"Chaoyou Fu, Peixian Chen, Yunhang Shen, Yulei Qin, Mengdan Zhang, Xu Lin, Jinrui Yang, Xiawu Zheng, Ke Li, Xing Sun, Yunsheng Wu, Rongrong Ji, Caifeng Shan, and Ran He. 2025. MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models. In Advances of Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00548"},{"key":"e_1_3_2_1_10_1","volume-title":"Gemini: A Family of Highly Capable Multimodal Models. arXiv preprint arXiv:2312.11805","author":"Team Gemini","year":"2023","unstructured":"Gemini Team, Google. 2023. Gemini: A Family of Highly Capable Multimodal Models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_1_11_1","volume-title":"Unlocking Multimodal Understanding Across Millions of Tokens of Context. arXiv preprint arXiv:2403.05530","author":"Team Gemini","year":"2024","unstructured":"Gemini Team, Google. 2024. Gemini 1.5: Unlocking Multimodal Understanding Across Millions of Tokens of Context. arXiv preprint arXiv:2403.05530 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"LookBench: A Live and Holistic Open Benchmark for Fashion Image Retrieval. arXiv preprint arXiv:2601.14706","author":"Gao Chao","year":"2026","unstructured":"Gensmo.ai, Chao Gao, Siqiao Xue, Jiwen Fu, Tingyi Gu, Shanshan Li, and Fan Zhou. 2026. LookBench: A Live and Holistic Open Benchmark for Fashion Image Retrieval. arXiv preprint arXiv:2601.14706 (2026)."},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 25th ACM International Conference on Multimedia. 1078-1086","author":"Han Xintong","unstructured":"Xintong Han, Zuxuan Wu, Yu-Gang Jiang, and Larry S. Davis. 2017. Learning Fashion Compatibility with Bidirectional LSTMs. In Proceedings of the 25th ACM International Conference on Multimedia. 1078-1086."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_37"},{"key":"e_1_3_2_1_15_1","volume-title":"Sai Htaung Kham, and Yuki Saito","author":"Hirakawa Yuki","year":"2024","unstructured":"Yuki Hirakawa, Takashi Wada, Kazuya Morishita, Ryotaro Shimizu, Takuya Furusawa, Sai Htaung Kham, and Yuki Saito. 2024. An Empirical Analysis of GPT-4V's Performance on Fashion Aesthetic Evaluation. In SIGGRAPH Asia 2024 Technical Communications. Article 24."},{"key":"e_1_3_2_1_16_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations.","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, yelong shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_19"},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the IEEE International Conference on Computer Vision. 3343-3351","author":"Kiapour M. Hadi","unstructured":"M. Hadi Kiapour, Xufeng Han, Svetlana Lazebnik, Alexander C. Berg, and Tamara L. Berg. 2015. Where to Buy It: Matching Street Clothing Photos in Online Shops. In Proceedings of the IEEE International Conference on Computer Vision. 3343-3351."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01263"},{"key":"e_1_3_2_1_20_1","volume-title":"LLaVA-OneVision: Easy Visual Task Transfer. Transactions on Machine Learning Research","author":"Li Bo","year":"2025","unstructured":"Bo Li, Yuanhan Zhang, Dong Guo, Renrui Zhang, Feng Li, Hao Zhang, Kaichen Zhang, Peiyuan Zhang, Yanwei Li, Ziwei Liu, and Chunyuan Li. 2025. LLaVA-OneVision: Easy Visual Task Transfer. Transactions on Machine Learning Research (2025)."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the International Conference on Machine Learning.","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023b. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Proceedings of the International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 162). 12888-12900."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"},{"key":"e_1_3_2_1_26_1","volume-title":"A Survey on Hallucination in Large Vision-Language Models. arXiv preprint arXiv:2402.00253","author":"Liu Hanchao","year":"2024","unstructured":"Hanchao Liu, Wenyuan Xue, Yifei Chen, Dapeng Chen, Xiutian Zhao, Ke Wang, Liping Hou, Rongjun Li, and Wei Peng. 2024c. A Survey on Hallucination in Large Vision-Language Models. arXiv preprint arXiv:2402.00253 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the European Conference on Computer Vision. 216-233","author":"Liu Yuan","year":"2024","unstructured":"Yuan Liu, Haodong Duan, Yuanhan Zhang, Bo Li, Songyang Zhang, Yike Yuan, Wangbo Zhao, Jiaqi Wang, Conghui He, Ziwei Liu, Kai Chen, and Dahua Lin. 2024a. MMBench: Is Your Multi-modal Model an All-around Player?. In Proceedings of the European Conference on Computer Vision. 216-233."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.124"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.556"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00681"},{"key":"e_1_3_2_1_31_1","unstructured":"OpenAI. 2023a. GPT-4 Technical Report. Technical Report. OpenAI. arXiv preprint arXiv:2303.08774."},{"key":"e_1_3_2_1_32_1","unstructured":"OpenAI. 2023b. GPT-4V(ision) System Card. Technical Report. OpenAI. Available at https:\/\/openai.com\/contributions\/gpt-4v."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_2_1_34_1","first-page":"8748","volume-title":"Proceedings of the International Conference on Machine Learning","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the International Conference on Machine Learning, Vol. 139. 8748-8763."},{"key":"e_1_3_2_1_35_1","volume-title":"International Conference on Learning Representations.","author":"Sclar Melanie","year":"2024","unstructured":"Melanie Sclar, Yejin Choi, Yulia Tsvetkov, and Alane Suhr. 2024. Quantifying Language Models' Sensitivity to Spurious Features in Prompt Design or: How I Learned to Start Worrying about Prompt Formatting. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.110791"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119167"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2009.03.002"},{"key":"e_1_3_2_1_39_1","volume-title":"Discovering Knowledge Deficiencies of Language Models on Massive Knowledge Base. In NeurIPS 2025 Workshop on Evaluating the Evolving LLM Lifecycle: Benchmarks, Emergent Abilities, and Scaling.","author":"Song Linxin","year":"2025","unstructured":"Linxin Song, Xuwei Ding, Jieyu Zhang, Taiwei Shi, Ryotaro Shimizu, Rahul Gupta, Yang Liu, Jian Kang, and Jieyu Zhao. 2025. Discovering Knowledge Deficiencies of Language Models on Massive Knowledge Base. In NeurIPS 2025 Workshop on Evaluating the Evolving LLM Lifecycle: Benchmarks, Emergent Abilities, and Scaling."},{"key":"e_1_3_2_1_40_1","first-page":"2579","article-title":"Visualizing Data Using t-SNE","volume":"9","author":"der Maaten Laurens Van","year":"2008","unstructured":"Laurens Van der Maaten and Geoffrey Hinton. 2008. Visualizing Data Using t-SNE. Journal of Machine Learning Research, Vol. 9 (2008), 2579-2605.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_41_1","unstructured":"Shijian Wang Linxin Song Jieyu Zhang Ryotaro Shimizu Ao Luo Li Yao Cunjian Chen Julian McAuley and Hanqian Wu. 2025. Template Matters: Understanding the Role of Instruction Templates in Multimodal Language Model Evaluation and Training. In ICLR 2025 Workshop on Navigating and Addressing Data Problems for Foundation Models."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1800"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01115"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2020"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:27:59Z","timestamp":1784136479000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808442"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":45,"alternative-id":["10.1145\/3805712.3808442","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808442","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}