{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T15:44:35Z","timestamp":1781797475216,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Key Research and Development Program of China","award":["2021YFA0717700"],"award-info":[{"award-number":["2021YFA0717700"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681135","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:49Z","timestamp":1729925989000},"page":"5451-5459","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["VeCAF: Vision-language Collaborative Active Finetuning with Training Objective Awareness"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9174-1765","authenticated-orcid":false,"given":"Rongyu","family":"Zhang","sequence":"first","affiliation":[{"name":"Nanjing University &amp; Peking University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8849-8854","authenticated-orcid":false,"given":"Zefan","family":"Cai","sequence":"additional","affiliation":[{"name":"University of Wisconsin - Madison &amp; Peking University, Madison, WI, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3384-4512","authenticated-orcid":false,"given":"Huanrui","family":"Yang","sequence":"additional","affiliation":[{"name":"University of California, Berkeley &amp; University of Arizona, Berkeley, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4894-2501","authenticated-orcid":false,"given":"Zidong","family":"Liu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6829-6667","authenticated-orcid":false,"given":"Denis","family":"Gudovskiy","sequence":"additional","affiliation":[{"name":"Panasonic Corporation, Mountainview, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7525-987X","authenticated-orcid":false,"given":"Tomoyuki","family":"Okuno","sequence":"additional","affiliation":[{"name":"Panasonic Corporation, Osaka, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9838-1367","authenticated-orcid":false,"given":"Yohei","family":"Nakata","sequence":"additional","affiliation":[{"name":"Panasonic Corporation, Osaka, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3868-8501","authenticated-orcid":false,"given":"Kurt","family":"Keutzer","sequence":"additional","affiliation":[{"name":"University of California, Berkeley, Berkeley, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2824-6750","authenticated-orcid":false,"given":"Baobao","family":"Chang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5316-619X","authenticated-orcid":false,"given":"Yuan","family":"Du","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2687-6978","authenticated-orcid":false,"given":"Li","family":"Du","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4047-3526","authenticated-orcid":false,"given":"Shanghang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_9"},{"key":"e_1_3_2_1_2_1","unstructured":"Alaaeldin Ali Hugo Touvron Mathilde Caron Piotr Bojanowski Matthijs Douze Armand Joulin Ivan Laptev Natalia Neverova Gabriel Synnaeve Jakob Verbeek et al. 2021. XCiT: Cross-covariance image transformers. Advances in Neural Information Processing Systems (2021)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00976"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV).","author":"Bengar Javad Zolfaghari","unstructured":"Javad Zolfaghari Bengar, Joost van de Weijer, Bartlomiej Twardowski, and Bogdan Raducanu. 2021. Reducing label effort: Self-supervised meets active learning. In Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_1_5_1","volume-title":"Human-in-the-Loop through Chain-of-Thought. arXiv preprint arXiv:2306.07932","author":"Cai Zefan","year":"2023","unstructured":"Zefan Cai, Baobao Chang, and Wenjuan Han. 2023. Human-in-the-Loop through Chain-of-Thought. arXiv preprint arXiv:2306.07932 (2023)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"e_1_3_2_1_7_1","volume-title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic. arXiv:2306.15195","author":"Chen Keqin","year":"2023","unstructured":"Keqin Chen, Zhao Zhang, Weili Zeng, Richong Zhang, Feng Zhu, and Rui Zhao. 2023. Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic. arXiv:2306.15195 (2023)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_9_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In North American","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In North American Chapter of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_10_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_11_1","volume-title":"Active Preference Learning with Discrete Choice Data. Advances in Neural Information Processing Systems","author":"Eric Brochu","year":"2007","unstructured":"Brochu Eric, Nando Freitas, and Abhijeet Ghosh. 2007. Active Preference Learning with Discrete Choice Data. Advances in Neural Information Processing Systems (2007)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01707"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2004.383"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR).","author":"Hendrycks Dan","year":"2019","unstructured":"Dan Hendrycks and Thomas Dietterich. 2019. Benchmarking Neural Network Robustness to Common Corruptions and Perturbations. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_15_1","volume-title":"Bayesian Active Learning for Discrete Latent Variable Models. arXiv:2202.13426","author":"Jha Aditi","year":"2022","unstructured":"Aditi Jha, Zoe C Ashwood, and Jonathan W Pillow. 2022. Bayesian Active Learning for Discrete Latent Variable Models. arXiv:2202.13426 (2022)."},{"key":"e_1_3_2_1_16_1","unstructured":"Shaoxiong Ji Shirui Pan Guodong Long Xue Li Jing Jiang and Zi Huang. 2019. Learning private neural language modeling with attentive aggregation. In International joint conference on neural networks (IJCNN)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206627"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00807"},{"key":"e_1_3_2_1_19_1","unstructured":"A. Krizhevsky and G. Hinton. 2009. Learning multiple layers of features from tiny images. Master's thesis University of Toronto (2009)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01753"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"David","unstructured":"David D. Lewis and Jason Catlett. 1994. Heterogenous Uncertainty Sampling for Supervised Learning. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_22_1","volume-title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. arXiv:2301.12597","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. arXiv:2301.12597 (2023)."},{"key":"e_1_3_2_1_23_1","unstructured":"Lei Li Yuwei Yin Shicheng Li Liang Chen Peiyi Wang Shuhuai Ren Mukai Li Yazheng Yang Jingjing Xu Xu Sun et al. 2023. M^3IT: A Large-Scale Dataset towards Multi-Modal Multilingual Instruction Tuning. arXiv:2306.04387 (2023)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00705"},{"key":"e_1_3_2_1_25_1","volume-title":"Piotr Dollar, and Larry Zitnick","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Dollar, and Larry Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00914"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_28_1","volume-title":"Latent Structured Active Learning. Advances in Neural Information Processing Systems","author":"Luo Wenjie","year":"2013","unstructured":"Wenjie Luo, Alex Schwing, and Raquel Urtasun. 2013. Latent Structured Active Learning. Advances in Neural Information Processing Systems (2013)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01722"},{"key":"e_1_3_2_1_30_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Mahmood Rafid","year":"2022","unstructured":"Rafid Mahmood, Sanja Fidler, and Marc T Law. 2022. Low-Budget Active Learning via Wasserstein Distance: An Integer Programming Approach. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01192"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_33_1","unstructured":"Alec Radford Jeffrey Wu Rewon Child David Luan Dario Amodei Ilya Sutskever et al. 2019. Language models are unsupervised multitask learners. OpenAI blog Vol. 1 8 (2019) 9."},{"key":"e_1_3_2_1_34_1","first-page":"25278","article-title":"2022. Laion-5b: An open large-scale dataset for training next generation image-text models","volume":"35","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, et al. 2022. Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems, Vol. 35 (2022), 25278--25294.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR).","author":"Sener Ozan","year":"2018","unstructured":"Ozan Sener and Silvio Savarese. 2018. Active Learning for Convolutional Neural Networks: A Core-Set Approach. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_37_1","volume-title":"Support Vector Machine Active Learning with Applications to Text Classification. Journal of Machine Learning Research","author":"Tong Simon","year":"2001","unstructured":"Simon Tong and Daphne Koller. 2001. Support Vector Machine Active Learning with Applications to Text Classification. Journal of Machine Learning Research (2001)."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Touvron Hugo","year":"2021","unstructured":"Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herve Jegou. 2021. Training data-efficient image transformers & distillation through attention. In Proceedings of the International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_39_1","volume-title":"Cost-Effective Active Learning for Deep Image Classification","author":"Wang Keze","year":"2016","unstructured":"Keze Wang, Dongyu Zhang, Ya Li, Ruimao Zhang, and Liang Lin. 2016. Cost-Effective Active Learning for Deep Image Classification. IEEE Transactions on Circuits and Systems for Video Technology (2016)."},{"key":"e_1_3_2_1_40_1","volume-title":"Large language models are not fair evaluators. arXiv preprint arXiv:2305.17926","author":"Wang Peiyi","year":"2023","unstructured":"Peiyi Wang, Lei Li, Liang Chen, Dawei Zhu, Binghuai Lin, Yunbo Cao, Qi Liu, Tianyu Liu, and Zhifang Sui. 2023. Large language models are not fair evaluators. arXiv preprint arXiv:2305.17926 (2023)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02271"},{"key":"e_1_3_2_1_42_1","unstructured":"Linting Xue Noah Constant Adam Roberts Mihir Kale Rami Al-Rfou Aditya Siddhant Aditya Barua and Colin Raffel. 2020. mT5: A Massively Multilingual Pre-trained Text-to-Text Transformer. In North American Chapter of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00018"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00529"},{"key":"e_1_3_2_1_45_1","volume-title":"MMICL: Empowering Vision-language Model with Multi-Modal In-Context Learning. arXiv:2309.07915","author":"Zhao Haozhe","year":"2023","unstructured":"Haozhe Zhao, Zefan Cai, Shuzheng Si, Xiaojian Ma, Kaikai An, Liang Chen, Zixuan Liu, Sheng Wang, Wenjuan Han, and Baobao Chang. 2023. MMICL: Empowering Vision-language Model with Multi-Modal In-Context Learning. arXiv:2309.07915 (2023)."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681135","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681135","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:53Z","timestamp":1750294673000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681135"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":44,"alternative-id":["10.1145\/3664647.3681135","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681135","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}