{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,7]],"date-time":"2025-10-07T08:49:37Z","timestamp":1759826977286,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3611904","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"5027-5037","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Uniformly Distributed Category Prototype-Guided Vision-Language Framework for Long-Tail Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2685-5010","authenticated-orcid":false,"given":"Xiaoxuan","family":"He","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou Shi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3257-1011","authenticated-orcid":false,"given":"Siming","family":"Fu","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou Shi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7653-1199","authenticated-orcid":false,"given":"Xinpeng","family":"Ding","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8108-0828","authenticated-orcid":false,"given":"Yuchen","family":"Cao","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou Shi, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0157-8885","authenticated-orcid":false,"given":"Hualiang","family":"Wang","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"1567","article-title":"Learning imbalanced datasets with label-distribution-aware margin loss","volume":"32","author":"Cao Kaidi","year":"2019","unstructured":"Kaidi Cao, Colin Wei, Adrien Gaidon, Nikos Arechiga, and Tengyu Ma. 2019. Learning imbalanced datasets with label-distribution-aware margin loss. Advances in Neural Information Processing Systems, Vol. 32 (2019), 1567--1578.","journal-title":"Advances in Neural Information Processing Systems"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_2_1","DOI":"10.1007\/978-3-030-58526-6_41"},{"key":"e_1_3_2_1_3_1","volume-title":"Reslt: Residual learning for long-tailed recognition","author":"Cui Jiequan","year":"2022","unstructured":"Jiequan Cui, Shu Liu, Zhuotao Tian, Zhisheng Zhong, and Jiaya Jia. 2022. Reslt: Residual learning for long-tailed recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence (2022), 3695--3706."},{"unstructured":"Jiequan Cui Zhisheng Zhong Shu Liu Bei Yu and Jiaya Jia. 2021. Parametric Contrastive Learning. arxiv: 2107.12028 [cs.CV]","key":"e_1_3_2_1_4_1"},{"volume-title":"International joint conference on artificial intelligence","author":"Elkan Charles","unstructured":"Charles Elkan. 2001. The foundations of cost-sensitive learning. In International joint conference on artificial intelligence, Vol. 17. Lawrence Erlbaum Associates Ltd, 973--978.","key":"e_1_3_2_1_5_1"},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the Asian Conference on Computer Vision (ACCV). 569--585","author":"Fu Siming","year":"2022","unstructured":"Siming Fu, Huanpeng Chu, Xiaoxuan He, Hualiang Wang, Zhenyu Yang, and Haoji Hu. 2022. Meta-Prototype Decoupled Training for Long-tailed Learning. In Proceedings of the Asian Conference on Computer Vision (ACCV). 569--585."},{"key":"e_1_3_2_1_7_1","volume-title":"Distributionally robust optimization and its tractable approximations. Operations research","author":"Goh Joel","year":"2010","unstructured":"Joel Goh and Melvyn Sim. 2010. Distributionally robust optimization and its tractable approximations. Operations research, Vol. 58, 4-part-1 (2010), 902--917."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_8_1","DOI":"10.1109\/CVPR.2016.90"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_9_1","DOI":"10.1109\/CVPR46437.2021.00656"},{"key":"e_1_3_2_1_10_1","volume-title":"International Conference on Machine Learning. PMLR, 4904--4916","author":"Jia Chao","year":"2021","unstructured":"Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig. 2021. Scaling up visual and vision-language representation learning with noisy text supervision. In International Conference on Machine Learning. PMLR, 4904--4916."},{"unstructured":"B. Kang S. Xie M. Rohrbach Z. Yan A. Gordo J. Feng and Y. Kalantidis. 2019. Decoupling Representation and Classifier for Long-Tailed Recognition. arXiv:1910.09217 (2019).","key":"e_1_3_2_1_11_1"},{"key":"e_1_3_2_1_12_1","first-page":"18661","article-title":"Supervised contrastive learning","volume":"33","author":"Khosla Prannay","year":"2020","unstructured":"Prannay Khosla, Piotr Teterwak, Chen Wang, Aaron Sarna, Yonglong Tian, Phillip Isola, Aaron Maschinot, Ce Liu, and Dilip Krishnan. 2020. Supervised contrastive learning. Advances in Neural Information Processing Systems, Vol. 33 (2020), 18661--18673.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_13_1","volume-title":"International Conference on Machine Learning. PMLR, 12888--12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022b. Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In International Conference on Machine Learning. PMLR, 12888--12900."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_14_1","DOI":"10.1109\/CVPR52688.2022.00679"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_15_1","DOI":"10.1109\/ICCV48922.2021.00067"},{"key":"e_1_3_2_1_16_1","volume-title":"Expectation-Maximization Attention Networks for Semantic Segmentation. In The IEEE International Conference on Computer Vision. 9167--9176","author":"Li Xia","year":"2019","unstructured":"Xia Li, Zhisheng Zhong, Jianlong Wu, Yibo Yang, Zhouchen Lin, and Hong Liu. 2019. Expectation-Maximization Attention Networks for Semantic Segmentation. In The IEEE International Conference on Computer Vision. 9167--9176."},{"key":"e_1_3_2_1_17_1","volume-title":"Clip surgery for better explainability with enhancement in open-vocabulary tasks. arXiv preprint arXiv:2304.05653","author":"Li Yi","year":"2023","unstructured":"Yi Li, Hualiang Wang, Yiqun Duan, and Xiaomeng Li. 2023. Clip surgery for better explainability with enhancement in open-vocabulary tasks. arXiv preprint arXiv:2304.05653 (2023)."},{"key":"e_1_3_2_1_18_1","volume-title":"Exploring Visual Interpretability for Contrastive Language-Image Pre-training. arXiv preprint arXiv:2209.07046","author":"Li Yi","year":"2022","unstructured":"Yi Li, Hualiang Wang, Yiqun Duan, Hang Xu, and Xiaomeng Li. 2022c. Exploring Visual Interpretability for Contrastive Language-Image Pre-training. arXiv preprint arXiv:2209.07046 (2022)."},{"volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition. 2537--2546","author":"Liu Ziwei","unstructured":"Ziwei Liu, Zhongqi Miao, Xiaohang Zhan, Jiayun Wang, Boqing Gong, and Stella X. Yu. 2019. Large-Scale Long-Tailed Recognition in an Open World. In Proceedings of the IEEE conference on computer vision and pattern recognition. 2537--2546.","key":"e_1_3_2_1_19_1"},{"key":"e_1_3_2_1_20_1","volume-title":"Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983","author":"Loshchilov Ilya","year":"2016","unstructured":"Ilya Loshchilov and Frank Hutter. 2016. Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016)."},{"key":"e_1_3_2_1_21_1","volume-title":"A simple long-tailed recognition baseline via vision-language model. arXiv preprint arXiv:2111.14745","author":"Ma Teli","year":"2021","unstructured":"Teli Ma, Shijie Geng, Mengmeng Wang, Jing Shao, Jiasen Lu, Hongsheng Li, Peng Gao, and Yu Qiao. 2021. A simple long-tailed recognition baseline via vision-language model. arXiv preprint arXiv:2111.14745 (2021)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_22_1","DOI":"10.1007\/978-3-030-01216-8_12"},{"key":"e_1_3_2_1_23_1","first-page":"3111","article-title":"Distributed representations of words and phrases and their compositionality","volume":"26","author":"Mikolov Tomas","year":"2013","unstructured":"Tomas Mikolov, Ilya Sutskever, Kai Chen, Greg S Corrado, and Jeff Dean. 2013. Distributed representations of words and phrases and their compositionality. Advances in Neural Information Processing Systems, Vol. 26 (2013), 3111--3119.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_24_1","volume-title":"International Conference on Machine Learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748--8763."},{"unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog Vol. 1 8 (2019) 9.","key":"e_1_3_2_1_25_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_26_1","DOI":"10.1109\/ICCV48922.2021.00936"},{"unstructured":"Christoph Schuhmann Romain Beaumont Richard Vencu Cade Gordon Ross Wightman Mehdi Cherti Theo Coombes Aarush Katta Clayton Mullis Mitchell Wortsman et al. 2022. Laion-5b: An open large-scale dataset for training next generation image-text models. arXiv preprint arXiv:2210.08402 (2022).","key":"e_1_3_2_1_27_1"},{"key":"e_1_3_2_1_28_1","volume-title":"Laion-400m: Open dataset of clip-filtered 400 million image-text pairs. arXiv preprint arXiv:2111.02114","author":"Schuhmann Christoph","year":"2021","unstructured":"Christoph Schuhmann, Richard Vencu, Romain Beaumont, Robert Kaczmarczyk, Clayton Mullis, Aarush Katta, Theo Coombes, Jenia Jitsev, and Aran Komatsuzaki. 2021. Laion-400m: Open dataset of clip-filtered 400 million image-text pairs. arXiv preprint arXiv:2111.02114 (2021)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_29_1","DOI":"10.1016\/j.patcog.2007.04.009"},{"key":"e_1_3_2_1_30_1","first-page":"1513","article-title":"Long-tailed classification by keeping the good and removing the bad momentum causal effect","volume":"33","author":"Tang Kaihua","year":"2020","unstructured":"Kaihua Tang, Jianqiang Huang, and Hanwang Zhang. 2020. Long-tailed classification by keeping the good and removing the bad momentum causal effect. Advances in Neural Information Processing Systems, Vol. 33 (2020), 1513--1524.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_31_1","volume-title":"Tel Aviv","author":"Tian Changyao","year":"2022","unstructured":"Changyao Tian, Wenhai Wang, Xizhou Zhu, Jifeng Dai, and Yu Qiao. 2022. Vl-ltr: Learning class-wise visual-linguistic representation for long-tailed visual recognition. In Computer Vision-ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXV. Springer, 73--91."},{"key":"e_1_3_2_1_32_1","volume-title":"International Conference on Machine Learning. PMLR, 10347--10357","author":"Touvron Hugo","year":"2021","unstructured":"Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herv\u00e9 J\u00e9gou. 2021. Training data-efficient image transformers & distillation through attention. In International Conference on Machine Learning. PMLR, 10347--10357."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_33_1","DOI":"10.1109\/CVPR.2018.00914"},{"key":"e_1_3_2_1_34_1","volume-title":"Tel Aviv","author":"Wang Hualiang","year":"2022","unstructured":"Hualiang Wang, Siming Fu, Xiaoxuan He, Hangxiang Fang, Zuozhu Liu, and Haoji Hu. 2022. Towards Calibrated Hyper-Sphere Representation via Distribution Overlap Coefficient for Long-Tailed Learning. In Computer Vision-ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXIV. Springer, 179--196."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_35_1","DOI":"10.1109\/CVPR46437.2021.00378"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_36_1","DOI":"10.1109\/CVPR46437.2021.00957"},{"key":"e_1_3_2_1_37_1","volume-title":"International Conference on Machine Learning. PMLR, 9929--9939","author":"Wang Tongzhou","year":"2020","unstructured":"Tongzhou Wang and Phillip Isola. 2020. Understanding contrastive representation learning through alignment and uniformity on the hypersphere. In International Conference on Machine Learning. PMLR, 9929--9939."},{"key":"e_1_3_2_1_38_1","volume-title":"International Conference on Learning Representations.","author":"Wang Xudong","year":"2021","unstructured":"Xudong Wang, Long Lian, Zhongqi Miao, Ziwei Liu, and Stella Yu. 2021a. Long-tailed Recognition by Routing Diverse Distribution-Aware Experts. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_39_1","volume-title":"Long-tailed recognition by routing diverse distribution-aware experts. arXiv preprint arXiv:2010.01809","author":"Wang Xudong","year":"2020","unstructured":"Xudong Wang, Long Lian, Zhongqi Miao, Ziwei Liu, and Stella X Yu. 2020. Long-tailed recognition by routing diverse distribution-aware experts. arXiv preprint arXiv:2010.01809 (2020)."},{"key":"e_1_3_2_1_40_1","volume-title":"Learning to model the tail. Advances in neural information processing systems","author":"Wang Yu-Xiong","year":"2017","unstructured":"Yu-Xiong Wang, Deva Ramanan, and Martial Hebert. 2017. Learning to model the tail. Advances in neural information processing systems, Vol. 30 (2017), 7032--7042."},{"key":"e_1_3_2_1_41_1","volume-title":"Zihang Dai, Yulia Tsvetkov, and Yuan Cao.","author":"Wang Zirui","year":"2021","unstructured":"Zirui Wang, Jiahui Yu, Adams Wei Yu, Zihang Dai, Yulia Tsvetkov, and Yuan Cao. 2021c. Simvlm: Simple visual language model pretraining with weak supervision. arXiv preprint arXiv:2108.10904 (2021)."},{"key":"e_1_3_2_1_42_1","volume-title":"FILIP: fine-grained interactive language-image pre-training. arXiv preprint arXiv:2111.07783","author":"Yao Lewei","year":"2021","unstructured":"Lewei Yao, Runhui Huang, Lu Hou, Guansong Lu, Minzhe Niu, Hang Xu, Xiaodan Liang, Zhenguo Li, Xin Jiang, and Chunjing Xu. 2021. FILIP: fine-grained interactive language-image pre-training. arXiv preprint arXiv:2111.07783 (2021)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_43_1","DOI":"10.1109\/CVPR46437.2021.00239"},{"key":"e_1_3_2_1_44_1","volume-title":"Test-Agnostic Long-Tailed Recognition by Test-Time Aggregating Diverse Experts with Self-Supervision. arXiv preprint arXiv:2107.09249","author":"Zhang Yifan","year":"2021","unstructured":"Yifan Zhang, Bryan Hooi, Lanqing Hong, and Jiashi Feng. 2021a. Test-Agnostic Long-Tailed Recognition by Test-Time Aggregating Diverse Experts with Self-Supervision. arXiv preprint arXiv:2107.09249 (2021)."},{"key":"e_1_3_2_1_45_1","volume-title":"Deep long-tailed learning: A survey. arXiv preprint arXiv:2110.04596","author":"Zhang Yifan","year":"2021","unstructured":"Yifan Zhang, Bingyi Kang, Bryan Hooi, Shuicheng Yan, and Jiashi Feng. 2021b. Deep long-tailed learning: A survey. arXiv preprint arXiv:2110.04596 (2021)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_46_1","DOI":"10.1109\/CVPR46437.2021.01622"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_47_1","DOI":"10.1109\/TPAMI.2017.2723009"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_48_1","DOI":"10.1109\/TKDE.2006.17"}],"event":{"sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"acronym":"MM '23","name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611904","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3611904","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:11:44Z","timestamp":1755821504000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611904"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":48,"alternative-id":["10.1145\/3581783.3611904","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3611904","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}