{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T15:26:14Z","timestamp":1777735574615,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","funder":[{"name":"National Key R&D Program of China","award":["2024YFB4710900"],"award-info":[{"award-number":["2024YFB4710900"]}]},{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62202087"],"award-info":[{"award-number":["62202087"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Guangdong Basic and Applied Basic Research Foundation","award":["2024A1515010244"],"award-info":[{"award-number":["2024A1515010244"]}]},{"DOI":"10.13039\/501100006374","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["N2404011, N2404008"],"award-info":[{"award-number":["N2404011, N2404008"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733328","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:29:43Z","timestamp":1750876183000},"page":"407-415","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Ensemble CLIPs: Effective Zero-shot Classification with Hundreds of Multi-modal CLIPs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-6094-3583","authenticated-orcid":false,"given":"Bowen","family":"Han","sequence":"first","affiliation":[{"name":"College of Information Science and Engineering, Northeastern University, Shenyang, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6863-8516","authenticated-orcid":false,"given":"Shizhuo","family":"Deng","sequence":"additional","affiliation":[{"name":"College of Information Science and Engineering, Northeastern University, Shenyang, Liaoning, China and Foshan Graduate School of Innovation, Northeastern University, Foshan, Guangdong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2170-4678","authenticated-orcid":false,"given":"Zehua","family":"Gan","sequence":"additional","affiliation":[{"name":"College of Information Science and Engineering, Northeastern University, Shenyang, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-7996-3716","authenticated-orcid":false,"given":"Da","family":"Teng","sequence":"additional","affiliation":[{"name":"College of Information Science and Engineering, Northeastern University, Shenyang, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0673-6767","authenticated-orcid":false,"given":"Dongyue","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Information Science and Engineering, Northeastern University, Shenyang, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1424-798X","authenticated-orcid":false,"given":"Tong","family":"Jia","sequence":"additional","affiliation":[{"name":"College of Information Science and Engineering, Northeastern University, Shenyang, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"e_1_3_2_1_2_1","volume-title":"Bagging predictors. Machine learning","author":"Breiman Leo","year":"1996","unstructured":"Leo Breiman. 1996. Bagging predictors. Machine learning, Vol. 24 (1996), 123--140."},{"key":"e_1_3_2_1_3_1","unstructured":"Tom B Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_4_1","first-page":"70115","article-title":"Large language models are visual reasoning coordinators","volume":"36","author":"Chen Liangyu","year":"2023","unstructured":"Liangyu Chen, Bo Li, Sheng Shen, Jingkang Yang, Chunyuan Li, Kurt Keutzer, Trevor Darrell, and Ziwei Liu. 2023. Large language models are visual reasoning coordinators. Advances in Neural Information Processing Systems, Vol. 36 (2023), 70115--70140.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.461"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"e_1_3_2_1_8_1","unstructured":"Han Fang Pengfei Xiong Luhui Xu and Yu Chen. 2021. CLIP2Video: Mastering Video-Text Retrieval via Image CLIP."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2004.383"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_41"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00255"},{"key":"e_1_3_2_1_12_1","unstructured":"Yoav Freund Robert E Schapire et al. 1996. Experiments with a new boosting algorithm. In icml Vol. 96. Citeseer 148--156."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25152"},{"key":"e_1_3_2_1_14_1","volume-title":"Align: Scaling up visual and vision-language representation learning with noisy text supervision.","author":"Jia Chao","year":"2021","unstructured":"Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc V Le, Yunhsuan Sung, Zhen Li, and Tom Duerig. 2021. Align: Scaling up visual and vision-language representation learning with noisy text supervision."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.77"},{"key":"e_1_3_2_1_16_1","volume-title":"An ensemble of fine-tuned convolutional neural networks for medical image classification","author":"Kumar Ashnil","year":"2016","unstructured":"Ashnil Kumar, Jinman Kim, David Lyndon, Michael Fulham, and Dagan Feng. 2016. An ensemble of fine-tuned convolutional neural networks for medical image classification. IEEE journal of biomedical and health informatics, Vol. 21, 1 (2016), 31--40."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Tian Liu Huixin Zhang Shubham Parashar and Shu Kong. 2024. Few-Shot Recognition via Stage-Wise Retrieval-Augmented Finetuning.","DOI":"10.1109\/CVPR52734.2025.01405"},{"key":"e_1_3_2_1_18_1","volume-title":"Beyond Sole Strength: Customized Ensembles for Generalized Vision-Language Models. In International Conference on Machine Learning.","author":"Lu Zhihe","year":"2024","unstructured":"Zhihe Lu, Jiawang Bai, Xin Li, Zeyu Xiao, and Xinchao Wang. 2024. Beyond Sole Strength: Customized Ensembles for Generalized Vision-Language Models. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_19_1","volume-title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval. arXiv preprint arXiv:2104.08860","author":"Luo Huaishao","year":"2021","unstructured":"Huaishao Luo, Lei Ji, Ming Zhong, Yang Chen, Wen Lei, Nan Duan, and Tianrui Li. 2021. CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval. arXiv preprint arXiv:2104.08860 (2021)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547910"},{"key":"e_1_3_2_1_21_1","unstructured":"Subhransu Maji Esa Rahtu Juho Kannala Matthew Blaschko and Andrea Vedaldi. 2013. Fine-grained visual classification of aircraft."},{"key":"e_1_3_2_1_22_1","volume-title":"The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=jlAjNL8z5cs","author":"Menon Sachit","year":"2023","unstructured":"Sachit Menon and Carl Vondrick. 2023. Visual Classification via Description from Large Language Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=jlAjNL8z5cs"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICVGIP.2008.47"},{"key":"e_1_3_2_1_24_1","volume-title":"International Conference on Machine Learning. PMLR, 26342--26362","author":"Novack Zachary","year":"2023","unstructured":"Zachary Novack, Julian McAuley, Zachary Chase Lipton, and Saurabh Garg. 2023. Chils: Zero-shot image classification with hierarchical label sets. In International Conference on Machine Learning. PMLR, 26342--26362."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01234"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6248092"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01438"},{"key":"e_1_3_2_1_28_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pam Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_29_1","volume-title":"International conference on machine learning. Pmlr, 8821--8831","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, Gabriel Goh, Scott Gray, Chelsea Voss, Alec Radford, Mark Chen, and Ilya Sutskever. 2021. Zero-shot text-to-image generation. In International conference on machine learning. Pmlr, 8821--8831."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01443"},{"key":"e_1_3_2_1_33_1","first-page":"25278","article-title":"Laion-5b: An open large-scale dataset for training next generation image-text models","volume":"35","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, et al. 2022. Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems, Vol. 35 (2022), 25278--25294.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_34_1","first-page":"14274","article-title":"Test-time prompt tuning for zero-shot generalization in vision-language models","volume":"35","author":"Shu Manli","year":"2022","unstructured":"Manli Shu, Weili Nie, De-An Huang, Zhiding Yu, Tom Goldstein, Anima Anandkumar, and Chaowei Xiao. 2022. Test-time prompt tuning for zero-shot generalization in vision-language models. Advances in Neural Information Processing Systems, Vol. 35 (2022), 14274--14289.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Roshan Zamir, and Mubarak Shah. 2012. UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild. arxiv: 1212.0402 [cs.CV] https:\/\/arxiv.org\/abs\/1212.0402"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00257"},{"key":"e_1_3_2_1_37_1","volume-title":"International Conference on Machine Learning. PMLR, 36978--36989","author":"Weng Zejia","year":"2023","unstructured":"Zejia Weng, Xitong Yang, Ang Li, Zuxuan Wu, and Yu-Gang Jiang. 2023. Open-vclip: Transforming clip to an open-vocabulary video model via interpolated weight optimization. In International Conference on Machine Learning. PMLR, 36978--36989."},{"key":"e_1_3_2_1_38_1","volume-title":"Stacked generalization. Neural networks","author":"Wolpert David H","year":"1992","unstructured":"David H Wolpert. 1992. Stacked generalization. Neural networks, Vol. 5, 2 (1992), 241--259."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"e_1_3_2_1_40_1","volume-title":"Hallucination is inevitable: An innate limitation of large language models. arXiv preprint arXiv:2401.11817","author":"Xu Ziwei","year":"2024","unstructured":"Ziwei Xu, Sanjay Jain, and Mohan Kankanhalli. 2024. Hallucination is inevitable: An innate limitation of large language models. arXiv preprint arXiv:2401.11817 (2024)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01460"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_29"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02713"}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","location":"Chicago IL USA","acronym":"ICMR '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733328","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:06:55Z","timestamp":1755749215000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733328"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":43,"alternative-id":["10.1145\/3731715.3733328","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733328","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}