{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T02:43:41Z","timestamp":1783737821575,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","funder":[{"name":"Science Foundation of Ministry of Education of China under Grant","award":["23JHQ090"],"award-info":[{"award-number":["23JHQ090"]}]},{"name":"National Key Research and Development Program of China under Grant","award":["2024YFE0102100"],"award-info":[{"award-number":["2024YFE0102100"]}]},{"name":"National Natural Science Foundation of China under Grants","award":["62177013 and 62277012"],"award-info":[{"award-number":["62177013 and 62277012"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,13]]},"DOI":"10.1145\/3726302.3729923","type":"proceedings-article","created":{"date-parts":[[2025,7,14]],"date-time":"2025-07-14T14:55:26Z","timestamp":1752504926000},"page":"844-853","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Class Activation Values: Lucid and Faithful Visual Interpretations for CLIP-based Text-Image Retrievals"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-0637-8184","authenticated-orcid":false,"given":"Pengxu","family":"Chen","sequence":"first","affiliation":[{"name":"Hainan University, Haikou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7813-5894","authenticated-orcid":false,"given":"Huazhong","family":"Liu","sequence":"additional","affiliation":[{"name":"Hainan University, Haikou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2939-6171","authenticated-orcid":false,"given":"Jihong","family":"Ding","sequence":"additional","affiliation":[{"name":"Hainan University, Haikou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6088-8585","authenticated-orcid":false,"given":"Xinghao","family":"Huang","sequence":"additional","affiliation":[{"name":"Hainan University, Haikou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3694-9532","authenticated-orcid":false,"given":"Shaojun","family":"Zou","sequence":"additional","affiliation":[{"name":"Hainan University, Haikou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7986-4244","authenticated-orcid":false,"given":"Laurence Tianruo","family":"Yang","sequence":"additional","affiliation":[{"name":"Zhengzhou University, Zhengzhou, China and St. Francis Xavier University, Antigonish, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,7,13]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"397","volume-title":"Lior Wolf. Generic Attention-model Explainability for Interpreting Bi-Modal and Encoder-Decoder Transformers. In Proceedings of IEEE\/CVF International Conference on Computer Vision","author":"Chefer Hila","year":"2021","unstructured":"Hila Chefer, Shir Gur, and Lior Wolf. Generic Attention-model Explainability for Interpreting Bi-Modal and Encoder-Decoder Transformers. In Proceedings of IEEE\/CVF International Conference on Computer Vision, pages 397-406, 2021."},{"key":"e_1_3_2_1_2_1","first-page":"5052","volume-title":"Advances in Neural Information Processing Systems","author":"Qiang Yao","year":"2022","unstructured":"Yao Qiang, Deng Pan, Chengyin Li, Xin Li, Rhongho Jang, and Dongxiao Zhu. AttCAT: Explaining Transformers via Attentive Class Activation Tokens. Advances in Neural Information Processing Systems, pages 5052-5064, 2022."},{"key":"e_1_3_2_1_3_1","first-page":"16009","volume-title":"Advances in Neural Information Processing Systems","author":"Wang Ying","year":"2023","unstructured":"Ying Wang, Tim G. J. Rudner, and Andrew Gordon Wilson. Visual Explanations of Image-Text Representations via Multi-Modal Information Bottleneck Attribution. Advances in Neural Information Processing Systems, pages 16009-16027, 2023."},{"key":"e_1_3_2_1_4_1","first-page":"61072","volume-title":"Proceedings of International Conference on Machine Learning","author":"Zhao Chenyang","year":"2024","unstructured":"Chenyang Zhao, Kun Wang, Xingyu Zeng, Rui Zhao, and Antoni B. Chan. Gradient-based Visual Explanation for Transformer-based CLIP. In Proceedings of International Conference on Machine Learning, pages 61072-61091, 2024."},{"key":"e_1_3_2_1_5_1","first-page":"8748","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchn Krueger, and Ilya Sutskever. Learning Transferable Visual Models from Natural Language Supervision. In International Conference on Machine Learning, pages 8748-8763, 2021."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657741"},{"key":"e_1_3_2_1_7_1","first-page":"310","volume-title":"Jiaqi. Long-CLIP: Unlocking the Long-text Capability of Clip. In Proceedings of European Conference on Computer Vision","author":"Zhang Beichen","year":"2025","unstructured":"Zhang, Beichen and Zhang, Pan and Dong, Xiaoyi and Zang, Yuhang and Wang, Jiaqi. Long-CLIP: Unlocking the Long-text Capability of Clip. In Proceedings of European Conference on Computer Vision, pages 310-325. Springer, 2025."},{"key":"e_1_3_2_1_8_1","first-page":"3635","volume-title":"Hadi. SAM-CLIP: Merging Vision Foundation Models Towards Semantic and Spatial Understanding. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Vasu Haoxiang","year":"2024","unstructured":"Wang, Haoxiang and Vasu, Pavan Kumar Anasosalu and Faghri, Fartash and Vemulapalli, Raviteja and Farajtabar, Mehrdad and Mehta, Sachin and Rastegari, Mohammad and Tuzel, Oncel and Pouransari, Hadi. SAM-CLIP: Merging Vision Foundation Models Towards Semantic and Spatial Understanding. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 3635-3647, 2024."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657831"},{"key":"e_1_3_2_1_10_1","first-page":"1","volume-title":"Nature Reviews Methods Primers","author":"Katsantoni Markus","year":"2021","unstructured":"Hafner, Markus and Katsantoni, Maria and K\u00f6ster, Tino and Marks, James and Mukherjee, Joyita and Staiger, Dorothee and Ule, Jernej and Zavolan, Mihaela. CLIP and Complementary Methods. Nature Reviews Methods Primers, pages 1-23, 2021."},{"key":"e_1_3_2_1_11_1","first-page":"1040","volume-title":"Krisztian Balog. Explainability for Transparent Conversational Information-Seeking. In Proceedings of the International ACM SIGIR Conference on Research and Development in Information Retrieval","author":"Spina Damiano","year":"2024","unstructured":"Damiano Spina, Johanne Trippas, and Krisztian Balog. Explainability for Transparent Conversational Information-Seeking. In Proceedings of the International ACM SIGIR Conference on Research and Development in Information Retrieval, page 1040--1050, 2024."},{"key":"e_1_3_2_1_12_1","first-page":"4190","volume-title":"Abnar and Willem Zuidema. Quantifying Attention Flow in Transformers. In Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"Samira","year":"2020","unstructured":"Samira Abnar and Willem Zuidema. Quantifying Attention Flow in Transformers. In Proceedings of the Annual Meeting of the Association for Computational Linguistics, pages 4190-4197, 2020."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1580"},{"key":"e_1_3_2_1_14_1","first-page":"782","volume-title":"Chefer and Shir Gur and Lior Wolf. Transformer Interpretability Beyond Attention Visualization. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Hila","year":"2021","unstructured":"Hila Chefer and Shir Gur and Lior Wolf. Transformer Interpretability Beyond Attention Visualization. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 782-791, 2021."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/174"},{"key":"e_1_3_2_1_16_1","first-page":"1","volume-title":"Text-based Decomposition. In Proceedings of International Conference on Learning Representation","author":"Gandelsman Yossi","year":"2024","unstructured":"Yossi Gandelsman, Alexei A. Efros, and Steinhardt Jacob. Interpreting CLIP's Image Representation via Text-based Decomposition. In Proceedings of International Conference on Learning Representation, pages 1-22, 2024."},{"key":"e_1_3_2_1_17_1","first-page":"2881","volume-title":"Jiaya Jia. Pyramid Scene Parsing Network. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Zhao Hengshuang","year":"2017","unstructured":"Hengshuang Zhao, Jianping Shi, Xiaojuan Qi, Xiaogang Wang, and Jiaya Jia. Pyramid Scene Parsing Network. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 2881-2890, 2017."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681707"},{"key":"e_1_3_2_1_19_1","first-page":"1","volume-title":"Advances in Neural Information Processing Systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jone, Aidan N. Gomez, Lukasa Kaiser, and Illia Polosukhin. Attention is All You Need. Advances in Neural Information Processing Systems, pages 1-15, 2017."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.519"},{"key":"e_1_3_2_1_21_1","first-page":"1","volume-title":"Proceedings of International Conference on Learning Representations","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In Proceedings of International Conference on Learning Representations, pages 1-9, 2021."},{"key":"e_1_3_2_1_22_1","first-page":"3595","volume-title":"Kuk-Jin Toon. Class Tokens Infusion for Weakly Supervised Semantic Segmentation. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Yoon Sung-Hoon","year":"2024","unstructured":"Sung-Hoon Yoon, Hoyong Kwon, Hyeonseong Kim, and Kuk-Jin Toon. Class Tokens Infusion for Weakly Supervised Semantic Segmentation. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 3595-3605, 2024."},{"key":"e_1_3_2_1_23_1","first-page":"85523","volume-title":"Advances in Neural Information Processing Systems","author":"Zou Yixiong","year":"2024","unstructured":"Yixiong Zou, Shuai Yi, Yuhua Li, and Ruixuan Li. A Closer Look at the CLS Token for Cross-Domain Few-Shot Learning. Advances in Neural Information Processing Systems, pages 85523-85545, 2024."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNSM.2021.3056912"},{"key":"e_1_3_2_1_25_1","first-page":"680","volume-title":"Dehua Chen. High-Order Faithful Interpretation Methods for Tensor-Based CNNs. In Proceedings of the IEEE International Symposium on Parallel and Distributed Processing with Applications","author":"Chen Pengxu","year":"2023","unstructured":"Pengxu Chen, Huazhong Liu, Hanning Zhang, Jihong Ding, Yunfan Zhang, and Dehua Chen. High-Order Faithful Interpretation Methods for Tensor-Based CNNs. In Proceedings of the IEEE International Symposium on Parallel and Distributed Processing with Applications, pages 680-687, 2023."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_28_1","first-page":"618","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Ramakrishna Vedantam Devi Parish Abhishek Das","year":"2017","unstructured":"Abhishek Das Ramakrishna Vedantam Devi Parish Ramprasaath R. Selvaraju, Michael Cogswell and Dhruv Batra. Grad-CAM: Visual Explanations from Deep Networks via Gradient-based Localization. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, page 618--626, 2017."},{"key":"e_1_3_2_1_29_1","first-page":"151","volume-title":"Saenko Kate. RISE: Randomized Input Sampling for Explanation of Black-models. In Proceedings of the British Machine Vision Conference","author":"Petsiuk Vitali","year":"2018","unstructured":"Vitali Petsiuk, Abir Das, and Saenko Kate. RISE: Randomized Input Sampling for Explanation of Black-models. In Proceedings of the British Machine Vision Conference, pages 151-165, 2018."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-017-1059-x"},{"key":"e_1_3_2_1_31_1","first-page":"24","volume-title":"Xia Hu. Score-CAM: Score-weighted Visual Explanations for Convolutional Neural Networks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshop","author":"Wang Haofan","year":"2020","unstructured":"Haofan Wang, Zifan Wang, Mengnan Du, Fan Yang, Zijian Zhang, Sirui Ding, Piotr Mardziel, and Xia Hu. Score-CAM: Score-weighted Visual Explanations for Convolutional Neural Networks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshop, pages 24-25, 2020."},{"key":"e_1_3_2_1_32_1","first-page":"312","volume-title":"Proceedings of the IEEE International Conference on Multimedia and Expo","author":"Xiang Xiaohong","year":"2023","unstructured":"Xiaohong Xiang, Fuyuan Zhang, Xin Deng, and Ke Hu. Multi-scale Inputs Make a Better Visual Interpretation of CNN Networks. In Proceedings of the IEEE International Conference on Multimedia and Expo, pages 312-317, 2023."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657750"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.371"},{"key":"e_1_3_2_1_35_1","first-page":"1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshop","volume":"2","author":"Qi Zhongang","year":"2019","unstructured":"Zhongang Qi, Saeed Khorram, and Fuxin Li. Visualizing Deep Networks by Optimizing with Integrated Gradients. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshop, volume 2, pages 1-4, 2019."},{"key":"e_1_3_2_1_36_1","first-page":"839","volume-title":"Vineeth N Balasubramanian. Grad-CAM: Generalized Gradient-based Visual Explanations for Deep Convolutional Networks. In Proceedings of IEEE\/CVF Winter Conference on Applications of Computer Vision","author":"Chattopadhay Aditya","year":"2018","unstructured":"Aditya Chattopadhay, Anirban Sarkar, Prantik Howlader, and Vineeth N Balasubramanian. Grad-CAM: Generalized Gradient-based Visual Explanations for Deep Convolutional Networks. In Proceedings of IEEE\/CVF Winter Conference on Applications of Computer Vision, pages 839-847, 2018."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3089943"},{"key":"e_1_3_2_1_38_1","first-page":"16322","volume-title":"Ajmal Mian. CAMERAS: Enhanced Resolution and Sanity Preserving Class Activation Mapping for Image Saliency. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Jalwana Mohammad A. A. K.","year":"2021","unstructured":"Mohammad A. A. K. Jalwana, Naveed Akhtar, Mohammed Bennamoun, and Ajmal Mian. CAMERAS: Enhanced Resolution and Sanity Preserving Class Activation Mapping for Image Saliency. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pages 16322-16331, 2021. graphy"}],"event":{"name":"SIGIR '25: The 48th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Padua Italy","acronym":"SIGIR '25","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3726302.3729923","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T18:36:23Z","timestamp":1755887783000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3726302.3729923"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,13]]},"references-count":38,"alternative-id":["10.1145\/3726302.3729923","10.1145\/3726302"],"URL":"https:\/\/doi.org\/10.1145\/3726302.3729923","relation":{},"subject":[],"published":{"date-parts":[[2025,7,13]]},"assertion":[{"value":"2025-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}