{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,10]],"date-time":"2026-05-10T10:20:28Z","timestamp":1778408428272,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":64,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,6]],"date-time":"2023-12-06T00:00:00Z","timestamp":1701820800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"R\\&D Program of Beijing Municipal Education Commission","award":["KM202110005022"],"award-info":[{"award-number":["KM202110005022"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61906007"],"award-info":[{"award-number":["61906007"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,12,6]]},"DOI":"10.1145\/3595916.3626437","type":"proceedings-article","created":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T16:34:41Z","timestamp":1704126881000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Generic Attention-model Explainability by Weighted Relevance Accumulation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6888-9622","authenticated-orcid":false,"given":"Yiming","family":"Huang","sequence":"first","affiliation":[{"name":"Faculty of Information Technology, Beijing University of Technology, CN"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9019-4087","authenticated-orcid":false,"given":"Aozhe","family":"Jia","sequence":"additional","affiliation":[{"name":"Faculty of Information Technology, Beijing University of Technology, CN"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7002-5447","authenticated-orcid":false,"given":"Xiaodan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Faculty of Information Technology, Beijing University of Technology, CN"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2292-4592","authenticated-orcid":false,"given":"Jiawei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Sensetime Research, CN"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,1]]},"reference":[{"key":"#cr-split#-e_1_3_2_2_1_1.1","doi-asserted-by":"crossref","unstructured":"Samira Abnar and Willem\u00a0H. Zuidema. 2020. Quantifying Attention Flow in Transformers. 4190-4197\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.385 10.18653\/v1","DOI":"10.18653\/v1\/2020.acl-main.385"},{"key":"#cr-split#-e_1_3_2_2_1_1.2","doi-asserted-by":"crossref","unstructured":"Samira Abnar and Willem\u00a0H. Zuidema. 2020. Quantifying Attention Flow in Transformers. 4190-4197\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.385","DOI":"10.18653\/v1\/2020.acl-main.385"},{"key":"#cr-split#-e_1_3_2_2_2_1.1","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katie Millican Malcolm Reynolds Roman Ring Eliza Rutherford Serkan Cabi Tengda Han Zhitao Gong Sina Samangooei Marianne Monteiro Jacob Menick Sebastian Borgeaud Andrew Brock Aida Nematzadeh Sahand Sharifzadeh Mikolaj Binkowski Ricardo Barreira Oriol Vinyals Andrew Zisserman and Karen Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. https:\/\/doi.org\/10.48550\/arXiv.2204.14198 arXiv:2204.14198 10.48550\/arXiv.2204.14198"},{"key":"#cr-split#-e_1_3_2_2_2_1.2","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katie Millican Malcolm Reynolds Roman Ring Eliza Rutherford Serkan Cabi Tengda Han Zhitao Gong Sina Samangooei Marianne Monteiro Jacob Menick Sebastian Borgeaud Andrew Brock Aida Nematzadeh Sahand Sharifzadeh Mikolaj Binkowski Ricardo Barreira Oriol Vinyals Andrew Zisserman and Karen Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. https:\/\/doi.org\/10.48550\/arXiv.2204.14198 arXiv:2204.14198"},{"key":"#cr-split#-e_1_3_2_2_3_1.1","doi-asserted-by":"crossref","unstructured":"Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. 6077-6086\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR.2018.00636 10.1109\/CVPR.2018.00636","DOI":"10.1109\/CVPR.2018.00636"},{"key":"#cr-split#-e_1_3_2_2_3_1.2","doi-asserted-by":"crossref","unstructured":"Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. 6077-6086\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR.2018.00636","DOI":"10.1109\/CVPR.2018.00636"},{"key":"e_1_3_2_2_4_1","volume-title":"METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments., 65\u201372\u00a0pages. https:\/\/aclanthology.org\/W05-0909\/","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie . 2005 . METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments., 65\u201372\u00a0pages. https:\/\/aclanthology.org\/W05-0909\/ Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments., 65\u201372\u00a0pages. https:\/\/aclanthology.org\/W05-0909\/"},{"key":"#cr-split#-e_1_3_2_2_5_1.1","doi-asserted-by":"crossref","unstructured":"Emanuele Bugliarello Ryan Cotterell Naoaki Okazaki and Desmond Elliott. 2021. Multimodal Pretraining Unmasked: A Meta-Analysis and a Unified Framework of Vision-and-Language BERTs. 978-994\u00a0pages. https:\/\/doi.org\/10.1162\/tacl_a_00408 10.1162\/tacl_a_00408","DOI":"10.1162\/tacl_a_00408"},{"key":"#cr-split#-e_1_3_2_2_5_1.2","doi-asserted-by":"crossref","unstructured":"Emanuele Bugliarello Ryan Cotterell Naoaki Okazaki and Desmond Elliott. 2021. Multimodal Pretraining Unmasked: A Meta-Analysis and a Unified Framework of Vision-and-Language BERTs. 978-994\u00a0pages. https:\/\/doi.org\/10.1162\/tacl_a_00408","DOI":"10.1162\/tacl_a_00408"},{"key":"#cr-split#-e_1_3_2_2_6_1.1","doi-asserted-by":"crossref","unstructured":"Hila Chefer Shir Gur and Lior Wolf. 2021. Generic Attention-model Explainability for Interpreting Bi-Modal and Encoder-Decoder Transformers. 387-396\u00a0pages. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00045 10.1109\/ICCV48922.2021.00045","DOI":"10.1109\/ICCV48922.2021.00045"},{"key":"#cr-split#-e_1_3_2_2_6_1.2","doi-asserted-by":"crossref","unstructured":"Hila Chefer Shir Gur and Lior Wolf. 2021. Generic Attention-model Explainability for Interpreting Bi-Modal and Encoder-Decoder Transformers. 387-396\u00a0pages. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00045","DOI":"10.1109\/ICCV48922.2021.00045"},{"key":"#cr-split#-e_1_3_2_2_7_1.1","doi-asserted-by":"crossref","unstructured":"Hila Chefer Shir Gur and Lior Wolf. 2021. Transformer Interpretability Beyond Attention Visualization. 782-791\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00084 10.1109\/CVPR46437.2021.00084","DOI":"10.1109\/CVPR46437.2021.00084"},{"key":"#cr-split#-e_1_3_2_2_7_1.2","doi-asserted-by":"crossref","unstructured":"Hila Chefer Shir Gur and Lior Wolf. 2021. Transformer Interpretability Beyond Attention Visualization. 782-791\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00084","DOI":"10.1109\/CVPR46437.2021.00084"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-022-1369-5"},{"key":"#cr-split#-e_1_3_2_2_9_1.1","doi-asserted-by":"crossref","unstructured":"Shuguang Chen Gustavo Aguilar Leonardo Neves and Thamar Solorio. 2021. Can images help recognize entities? A study of the role of images for Multimodal NER. 87-96\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2021.wnut-1.11 10.18653\/v1","DOI":"10.18653\/v1\/2021.wnut-1.11"},{"key":"#cr-split#-e_1_3_2_2_9_1.2","doi-asserted-by":"crossref","unstructured":"Shuguang Chen Gustavo Aguilar Leonardo Neves and Thamar Solorio. 2021. Can images help recognize entities? A study of the role of images for Multimodal NER. 87-96\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2021.wnut-1.11","DOI":"10.18653\/v1\/2021.wnut-1.11"},{"key":"e_1_3_2_2_10_1","unstructured":"Xinlei Chen Hao Fang Tsung-Yi Lin Ramakrishna Vedantam Saurabh Gupta Piotr Doll\u00e1r and C.\u00a0Lawrence Zitnick. 2015. Microsoft COCO Captions: Data Collection and Evaluation Server. arXiv:1504.00325http:\/\/arxiv.org\/abs\/1504.00325 Xinlei Chen Hao Fang Tsung-Yi Lin Ramakrishna Vedantam Saurabh Gupta Piotr Doll\u00e1r and C.\u00a0Lawrence Zitnick. 2015. Microsoft COCO Captions: Data Collection and Evaluation Server. arXiv:1504.00325http:\/\/arxiv.org\/abs\/1504.00325"},{"key":"e_1_3_2_2_11_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly Jakob Uszkoreit and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. https:\/\/openreview.net\/forum?id=YicbFdNTTy Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly Jakob Uszkoreit and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. https:\/\/openreview.net\/forum?id=YicbFdNTTy"},{"key":"#cr-split#-e_1_3_2_2_12_1.1","doi-asserted-by":"crossref","unstructured":"Yifan Du Zikang Liu Junyi Li and Wayne\u00a0Xin Zhao. 2022. A Survey of Vision-Language Pre-Trained Models. 5436-5443\u00a0pages. https:\/\/doi.org\/10.24963\/ijcai.2022\/762 10.24963\/ijcai.2022","DOI":"10.24963\/ijcai.2022\/762"},{"key":"#cr-split#-e_1_3_2_2_12_1.2","doi-asserted-by":"crossref","unstructured":"Yifan Du Zikang Liu Junyi Li and Wayne\u00a0Xin Zhao. 2022. A Survey of Vision-Language Pre-Trained Models. 5436-5443\u00a0pages. https:\/\/doi.org\/10.24963\/ijcai.2022\/762","DOI":"10.24963\/ijcai.2022\/762"},{"key":"#cr-split#-e_1_3_2_2_13_1.1","doi-asserted-by":"crossref","unstructured":"Stella Frank Emanuele Bugliarello and Desmond Elliott. 2021. Vision-and-Language or Vision-for-Language? On Cross-Modal Influence in Multimodal Transformers. 9847-9857\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2021.emnlp-main.775 10.18653\/v1","DOI":"10.18653\/v1\/2021.emnlp-main.775"},{"key":"#cr-split#-e_1_3_2_2_13_1.2","doi-asserted-by":"crossref","unstructured":"Stella Frank Emanuele Bugliarello and Desmond Elliott. 2021. Vision-and-Language or Vision-for-Language? On Cross-Modal Influence in Multimodal Transformers. 9847-9857\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2021.emnlp-main.775","DOI":"10.18653\/v1\/2021.emnlp-main.775"},{"key":"#cr-split#-e_1_3_2_2_14_1.1","doi-asserted-by":"crossref","unstructured":"Yash Goyal Tejas Khot Aishwarya Agrawal Douglas Summers-Stay Dhruv Batra and Devi Parikh. 2019. Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering. 398-414\u00a0pages. https:\/\/doi.org\/10.1007\/s11263-018-1116-0 10.1007\/s11263-018-1116-0","DOI":"10.1007\/s11263-018-1116-0"},{"key":"#cr-split#-e_1_3_2_2_14_1.2","doi-asserted-by":"crossref","unstructured":"Yash Goyal Tejas Khot Aishwarya Agrawal Douglas Summers-Stay Dhruv Batra and Devi Parikh. 2019. Making the V in VQA Matter: Elevating the Role of Image Understanding in Visual Question Answering. 398-414\u00a0pages. https:\/\/doi.org\/10.1007\/s11263-018-1116-0","DOI":"10.1007\/s11263-018-1116-0"},{"key":"#cr-split#-e_1_3_2_2_15_1.1","doi-asserted-by":"crossref","unstructured":"Jack Hessel Ari Holtzman Maxwell Forbes Ronan\u00a0Le Bras and Yejin Choi. 2021. CLIPScore: A Reference-free Evaluation Metric for Image Captioning. 7514-7528\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2021.emnlp-main.595 10.18653\/v1","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"#cr-split#-e_1_3_2_2_15_1.2","doi-asserted-by":"crossref","unstructured":"Jack Hessel Ari Holtzman Maxwell Forbes Ronan\u00a0Le Bras and Yejin Choi. 2021. CLIPScore: A Reference-free Evaluation Metric for Image Captioning. 7514-7528\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2021.emnlp-main.595","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"#cr-split#-e_1_3_2_2_16_1.1","doi-asserted-by":"crossref","unstructured":"Andrej Karpathy and Li Fei-Fei. 2017. Deep Visual-Semantic Alignments for Generating Image Descriptions. 664-676\u00a0pages. https:\/\/doi.org\/10.1109\/TPAMI.2016.2598339 10.1109\/TPAMI.2016.2598339","DOI":"10.1109\/TPAMI.2016.2598339"},{"key":"#cr-split#-e_1_3_2_2_16_1.2","doi-asserted-by":"crossref","unstructured":"Andrej Karpathy and Li Fei-Fei. 2017. Deep Visual-Semantic Alignments for Generating Image Descriptions. 664-676\u00a0pages. https:\/\/doi.org\/10.1109\/TPAMI.2016.2598339","DOI":"10.1109\/TPAMI.2016.2598339"},{"key":"e_1_3_2_2_17_1","unstructured":"Wonjae Kim Bokyung Son and Ildoo Kim. 2021. ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision. 5583\u20135594\u00a0pages. http:\/\/proceedings.mlr.press\/v139\/kim21k.html Wonjae Kim Bokyung Son and Ildoo Kim. 2021. ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision. 5583\u20135594\u00a0pages. http:\/\/proceedings.mlr.press\/v139\/kim21k.html"},{"key":"e_1_3_2_2_18_1","unstructured":"Junnan Li Ramprasaath\u00a0R. Selvaraju Akhilesh Gotmare Shafiq\u00a0R. Joty Caiming Xiong and Steven\u00a0Chu-Hong Hoi. 2021. Align before Fuse: Vision and Language Representation Learning with Momentum Distillation. 9694\u20139705\u00a0pages. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/505259756244493872b7709a8a01b536-Abstract.html Junnan Li Ramprasaath\u00a0R. Selvaraju Akhilesh Gotmare Shafiq\u00a0R. Joty Caiming Xiong and Steven\u00a0Chu-Hong Hoi. 2021. Align before Fuse: Vision and Language Representation Learning with Momentum Distillation. 9694\u20139705\u00a0pages. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/505259756244493872b7709a8a01b536-Abstract.html"},{"key":"e_1_3_2_2_19_1","unstructured":"Liunian\u00a0Harold Li Mark Yatskar Da Yin Cho-Jui Hsieh and Kai-Wei Chang. 2019. VisualBERT: A Simple and Performant Baseline for Vision and Language. arXiv:1908.03557http:\/\/arxiv.org\/abs\/1908.03557 Liunian\u00a0Harold Li Mark Yatskar Da Yin Cho-Jui Hsieh and Kai-Wei Chang. 2019. VisualBERT: A Simple and Performant Baseline for Vision and Language. arXiv:1908.03557http:\/\/arxiv.org\/abs\/1908.03557"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"e_1_3_2_2_21_1","unstructured":"Yibing Liu Haoliang Li Yangyang Guo Chenqi Kong Jing Li and Shiqi Wang. 2022. Rethinking Attention-Model Explainability through Faithfulness Violation Test. 13807\u201313824\u00a0pages. https:\/\/proceedings.mlr.press\/v162\/liu22i.html Yibing Liu Haoliang Li Yangyang Guo Chenqi Kong Jing Li and Shiqi Wang. 2022. Rethinking Attention-Model Explainability through Faithfulness Violation Test. 13807\u201313824\u00a0pages. https:\/\/proceedings.mlr.press\/v162\/liu22i.html"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3514094.3534148"},{"key":"e_1_3_2_2_23_1","unstructured":"Ron Mokady Amir Hertz and Amit\u00a0H. Bermano. 2021. ClipCap: CLIP Prefix for Image Captioning. arXiv:2111.09734https:\/\/arxiv.org\/abs\/2111.09734 Ron Mokady Amir Hertz and Amit\u00a0H. Bermano. 2021. ClipCap: CLIP Prefix for Image Captioning. arXiv:2111.09734https:\/\/arxiv.org\/abs\/2111.09734"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"crossref","unstructured":"Kishore Papineni Salim Roukos Todd Ward and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. 311\u2013318\u00a0pages. Kishore Papineni Salim Roukos Todd Ward and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. 311\u2013318\u00a0pages.","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00915"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00915"},{"key":"e_1_3_2_2_27_1","unstructured":"Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark Gretchen Krueger and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. 8748\u20138763\u00a0pages. http:\/\/proceedings.mlr.press\/v139\/radford21a.html Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark Gretchen Krueger and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. 8748\u20138763\u00a0pages. http:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"#cr-split#-e_1_3_2_2_28_1.1","doi-asserted-by":"crossref","unstructured":"Shaoqing Ren Kaiming He Ross\u00a0B. Girshick and Jian Sun. 2017. Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. 1137-1149\u00a0pages. https:\/\/doi.org\/10.1109\/TPAMI.2016.2577031 10.1109\/TPAMI.2016.2577031","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"#cr-split#-e_1_3_2_2_28_1.2","doi-asserted-by":"crossref","unstructured":"Shaoqing Ren Kaiming He Ross\u00a0B. Girshick and Jian Sun. 2017. Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. 1137-1149\u00a0pages. https:\/\/doi.org\/10.1109\/TPAMI.2016.2577031","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"#cr-split#-e_1_3_2_2_29_1.1","doi-asserted-by":"crossref","unstructured":"Marco\u00a0T\u00falio Ribeiro Sameer Singh and Carlos Guestrin. 2016. \"Why Should I Trust You?\": Explaining the Predictions of Any Classifier. 1135-1144\u00a0pages. https:\/\/doi.org\/10.1145\/2939672.2939778 10.1145\/2939672.2939778","DOI":"10.1145\/2939672.2939778"},{"key":"#cr-split#-e_1_3_2_2_29_1.2","doi-asserted-by":"crossref","unstructured":"Marco\u00a0T\u00falio Ribeiro Sameer Singh and Carlos Guestrin. 2016. \"Why Should I Trust You?\": Explaining the Predictions of Any Classifier. 1135-1144\u00a0pages. https:\/\/doi.org\/10.1145\/2939672.2939778","DOI":"10.1145\/2939672.2939778"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Emmanuelle Salin Badreddine Farah St\u00e9phane Ayache and Beno\u00eet Favre. 2022. Are Vision-Language Transformers Learning Multimodal Representations? A Probing Perspective. 11248\u201311257\u00a0pages. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/21375 Emmanuelle Salin Badreddine Farah St\u00e9phane Ayache and Beno\u00eet Favre. 2022. Are Vision-Language Transformers Learning Multimodal Representations? A Probing Perspective. 11248\u201311257\u00a0pages. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/21375","DOI":"10.1609\/aaai.v36i10.21375"},{"key":"#cr-split#-e_1_3_2_2_31_1.1","doi-asserted-by":"crossref","unstructured":"Ramprasaath\u00a0R. Selvaraju Michael Cogswell Abhishek Das Ramakrishna Vedantam Devi Parikh and Dhruv Batra. 2017. Grad-CAM: Visual Explanations from Deep Networks via Gradient-Based Localization. 618-626\u00a0pages. https:\/\/doi.org\/10.1109\/ICCV.2017.74 10.1109\/ICCV.2017.74","DOI":"10.1109\/ICCV.2017.74"},{"key":"#cr-split#-e_1_3_2_2_31_1.2","doi-asserted-by":"crossref","unstructured":"Ramprasaath\u00a0R. Selvaraju Michael Cogswell Abhishek Das Ramakrishna Vedantam Devi Parikh and Dhruv Batra. 2017. Grad-CAM: Visual Explanations from Deep Networks via Gradient-Based Localization. 618-626\u00a0pages. https:\/\/doi.org\/10.1109\/ICCV.2017.74","DOI":"10.1109\/ICCV.2017.74"},{"key":"e_1_3_2_2_32_1","unstructured":"Sheng Shen Liunian\u00a0Harold Li Hao Tan Mohit Bansal Anna Rohrbach Kai-Wei Chang Zhewei Yao and Kurt Keutzer. 2022. How Much Can CLIP Benefit Vision-and-Language Tasks?https:\/\/openreview.net\/forum?id=zf_Ll3HZWgy Sheng Shen Liunian\u00a0Harold Li Hao Tan Mohit Bansal Anna Rohrbach Kai-Wei Chang Zhewei Yao and Kurt Keutzer. 2022. How Much Can CLIP Benefit Vision-and-Language Tasks?https:\/\/openreview.net\/forum?id=zf_Ll3HZWgy"},{"key":"e_1_3_2_2_33_1","unstructured":"Vivswan Shitole Fuxin Li Minsuk Kahng Prasad Tadepalli and Alan Fern. 2021. One Explanation is Not Enough: Structured Attention Graphs for Image Classification. 11352\u201311363\u00a0pages. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/5e751896e527c862bf67251a474b3819-Abstract.html Vivswan Shitole Fuxin Li Minsuk Kahng Prasad Tadepalli and Alan Fern. 2021. One Explanation is Not Enough: Structured Attention Graphs for Image Classification. 11352\u201311363\u00a0pages. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/5e751896e527c862bf67251a474b3819-Abstract.html"},{"key":"#cr-split#-e_1_3_2_2_34_1.1","doi-asserted-by":"crossref","unstructured":"Haoyu Song Li Dong Weinan Zhang Ting Liu and Furu Wei. 2022. CLIP Models are Few-Shot Learners: Empirical Studies on VQA and Visual Entailment. 6088-6100\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2022.acl-long.421 10.18653\/v1","DOI":"10.18653\/v1\/2022.acl-long.421"},{"key":"#cr-split#-e_1_3_2_2_34_1.2","doi-asserted-by":"crossref","unstructured":"Haoyu Song Li Dong Weinan Zhang Ting Liu and Furu Wei. 2022. CLIP Models are Few-Shot Learners: Empirical Studies on VQA and Visual Entailment. 6088-6100\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/2022.acl-long.421","DOI":"10.18653\/v1\/2022.acl-long.421"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"#cr-split#-e_1_3_2_2_36_1.1","doi-asserted-by":"crossref","unstructured":"Damien Teney Peter Anderson Xiaodong He and Anton van\u00a0den Hengel. 2018. Tips and Tricks for Visual Question Answering: Learnings From the 2017 Challenge. 4223-4232\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR.2018.00444 10.1109\/CVPR.2018.00444","DOI":"10.1109\/CVPR.2018.00444"},{"key":"#cr-split#-e_1_3_2_2_36_1.2","doi-asserted-by":"crossref","unstructured":"Damien Teney Peter Anderson Xiaodong He and Anton van\u00a0den Hengel. 2018. Tips and Tricks for Visual Question Answering: Learnings From the 2017 Challenge. 4223-4232\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR.2018.00444","DOI":"10.1109\/CVPR.2018.00444"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"#cr-split#-e_1_3_2_2_38_1.1","doi-asserted-by":"crossref","unstructured":"Ramakrishna Vedantam C.\u00a0Lawrence Zitnick and Devi Parikh. 2015. CIDEr: Consensus-based image description evaluation. 4566-4575\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR.2015.7299087 10.1109\/CVPR.2015.7299087","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"#cr-split#-e_1_3_2_2_38_1.2","doi-asserted-by":"crossref","unstructured":"Ramakrishna Vedantam C.\u00a0Lawrence Zitnick and Devi Parikh. 2015. CIDEr: Consensus-based image description evaluation. 4566-4575\u00a0pages. https:\/\/doi.org\/10.1109\/CVPR.2015.7299087","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"#cr-split#-e_1_3_2_2_39_1.1","doi-asserted-by":"crossref","unstructured":"Elena Voita David Talbot Fedor Moiseev Rico Sennrich and Ivan Titov. 2019. Analyzing Multi-Head Self-Attention: Specialized Heads Do the Heavy Lifting the Rest Can Be Pruned. 5797-5808\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/p19-1580 10.18653\/v1","DOI":"10.18653\/v1\/P19-1580"},{"key":"#cr-split#-e_1_3_2_2_39_1.2","doi-asserted-by":"crossref","unstructured":"Elena Voita David Talbot Fedor Moiseev Rico Sennrich and Ivan Titov. 2019. Analyzing Multi-Head Self-Attention: Specialized Heads Do the Heavy Lifting the Rest Can Be Pruned. 5797-5808\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/p19-1580","DOI":"10.18653\/v1\/P19-1580"},{"key":"#cr-split#-e_1_3_2_2_40_1.1","doi-asserted-by":"crossref","unstructured":"Sarah Wiegreffe and Yuval Pinter. 2019. Attention is not not Explanation. 11-20\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/D19-1002 10.18653\/v1","DOI":"10.18653\/v1\/D19-1002"},{"key":"#cr-split#-e_1_3_2_2_40_1.2","doi-asserted-by":"crossref","unstructured":"Sarah Wiegreffe and Yuval Pinter. 2019. Attention is not not Explanation. 11-20\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/D19-1002","DOI":"10.18653\/v1\/D19-1002"},{"key":"#cr-split#-e_1_3_2_2_41_1.1","unstructured":"Jiahui Yu Zirui Wang Vijay Vasudevan Legg Yeung Mojtaba Seyedhosseini and Yonghui Wu. 2022. CoCa: Contrastive Captioners are Image-Text Foundation Models. https:\/\/doi.org\/10.48550\/arXiv.2205.01917 arXiv:2205.01917 10.48550\/arXiv.2205.01917"},{"key":"#cr-split#-e_1_3_2_2_41_1.2","unstructured":"Jiahui Yu Zirui Wang Vijay Vasudevan Legg Yeung Mojtaba Seyedhosseini and Yonghui Wu. 2022. CoCa: Contrastive Captioners are Image-Text Foundation Models. https:\/\/doi.org\/10.48550\/arXiv.2205.01917 arXiv:2205.01917"},{"key":"#cr-split#-e_1_3_2_2_42_1.1","doi-asserted-by":"crossref","unstructured":"Amir Zadeh Paul\u00a0Pu Liang Soujanya Poria Erik Cambria and Louis-Philippe Morency. 2018. Multimodal Language Analysis in the Wild: CMU-MOSEI Dataset and Interpretable Dynamic Fusion Graph. 2236-2246\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/P18-1208 10.18653\/v1","DOI":"10.18653\/v1\/P18-1208"},{"key":"#cr-split#-e_1_3_2_2_42_1.2","doi-asserted-by":"crossref","unstructured":"Amir Zadeh Paul\u00a0Pu Liang Soujanya Poria Erik Cambria and Louis-Philippe Morency. 2018. Multimodal Language Analysis in the Wild: CMU-MOSEI Dataset and Interpretable Dynamic Fusion Graph. 2236-2246\u00a0pages. https:\/\/doi.org\/10.18653\/v1\/P18-1208","DOI":"10.18653\/v1\/P18-1208"}],"event":{"name":"MMAsia '23: ACM Multimedia Asia","location":"Tainan Taiwan","acronym":"MMAsia '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["ACM Multimedia Asia 2023"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3595916.3626437","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3595916.3626437","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:35:56Z","timestamp":1750178156000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3595916.3626437"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,6]]},"references-count":64,"alternative-id":["10.1145\/3595916.3626437","10.1145\/3595916"],"URL":"https:\/\/doi.org\/10.1145\/3595916.3626437","relation":{},"subject":[],"published":{"date-parts":[[2023,12,6]]},"assertion":[{"value":"2024-01-01","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}