{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T13:10:08Z","timestamp":1748092208284,"version":"3.41.0"},"publisher-location":"Cham","reference-count":52,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031920882","type":"print"},{"value":"9783031920899","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-92089-9_16","type":"book-chapter","created":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T12:49:18Z","timestamp":1748090958000},"page":"247-262","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["How to\u00a0Squeeze An Explanation Out of\u00a0Your Model"],"prefix":"10.1007","author":[{"given":"Tiago","family":"Roxo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joana C.","family":"Costa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pedro R. M.","family":"In\u00e1cio","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hugo","family":"Proen\u00e7a","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"16_CR1","doi-asserted-by":"crossref","unstructured":"Bargal, S.A., Zunino, A., Kim, D., Zhang, J., Murino, V., Sclaroff, S.: Excitation backprop for RNNs. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1440\u20131449 (2018)","DOI":"10.1109\/CVPR.2018.00156"},{"key":"16_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2023.104645","volume":"132","author":"L Cascone","year":"2023","unstructured":"Cascone, L., Pero, C., Proen\u00e7a, H.: Visual and textual explainability for a biometric verification system based on piecewise facial attribute analysis. Image Vis. Comput. 132, 104645 (2023)","journal-title":"Image Vis. Comput."},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Chattopadhay, A., Sarkar, A., Howlader, P., Balasubramanian, V.N.: Grad-CAM++: generalized gradient-based visual explanations for deep convolutional networks. In: 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 839\u2013847. IEEE (2018)","DOI":"10.1109\/WACV.2018.00097"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Chefer, H., Gur, S., Wolf, L.: Transformer interpretability beyond attention visualization. In: Proceedings of the IEEE\/CVF Conference on CVPR, pp. 782\u2013791 (2021)","DOI":"10.1109\/CVPR46437.2021.00084"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Z., et al.: LCTR: on awakening the local continuity of transformer for weakly supervised object localization. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol.\u00a036, pp. 410\u2013418 (2022)","DOI":"10.1609\/aaai.v36i1.19918"},{"key":"16_CR6","doi-asserted-by":"publisher","first-page":"61113","DOI":"10.1109\/ACCESS.2024.3395118","volume":"12","author":"JC Costa","year":"2024","unstructured":"Costa, J.C., Roxo, T., Proen\u00e7a, H., In\u00e1cio, P.: How deep learning sees the world: a survey on adversarial attacks & defenses. IEEE Access 12, 61113\u201361136 (2024). https:\/\/doi.org\/10.1109\/ACCESS.2024.3395118","journal-title":"IEEE Access"},{"key":"16_CR7","doi-asserted-by":"publisher","unstructured":"Costa, J.C., Roxo, T., Sequeiros, J.B.F., Proen\u00e7a, H., In\u00e1cio, P.R.M.: Predicting CVSS metric via description interpretation. IEEE Access 10, 59125\u201359134 (2022). https:\/\/doi.org\/10.1109\/ACCESS.2022.3179692","DOI":"10.1109\/ACCESS.2022.3179692"},{"key":"16_CR8","unstructured":"Dabkowski, P., Gal, Y.: Real time image saliency for black box classifiers. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"16_CR9","doi-asserted-by":"crossref","unstructured":"Doshi, K., Yilmaz, Y.: Towards interpretable video anomaly detection. In: Proceedings of the IEEE\/CVF WACV, pp. 2655\u20132664 (2023)","DOI":"10.1109\/WACV56688.2023.00268"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Fong, R., Patrick, M., Vedaldi, A.: Understanding deep networks via extremal perturbations and smooth masks. In: Proceedings of the IEEE\/CVF ICCV, pp. 2950\u20132958 (2019)","DOI":"10.1109\/ICCV.2019.00304"},{"key":"16_CR11","doi-asserted-by":"crossref","unstructured":"Fong, R.C., Vedaldi, A.: Interpretable explanations of black boxes by meaningful perturbation. In: Proceedings of the IEEE ICCV, pp. 3429\u20133437 (2017)","DOI":"10.1109\/ICCV.2017.371"},{"issue":"8","key":"16_CR12","doi-asserted-by":"publisher","first-page":"5213","DOI":"10.1109\/TCSVT.2021.3137023","volume":"32","author":"J Fu","year":"2021","unstructured":"Fu, J., Gao, J., Xu, C.: Learning semantic-aware spatial-temporal attention for interpretable action recognition. IEEE Trans. Circuits Syst. Video Technol. 32(8), 5213\u20135224 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"16_CR13","doi-asserted-by":"crossref","unstructured":"Gao, W., et al.: TS-CAM: token semantic coupled attention map for weakly supervised object localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2886\u20132895 (2021)","DOI":"10.1109\/ICCV48922.2021.00288"},{"key":"16_CR14","unstructured":"Gildenblat, J., contributors: PyTorch library for cam methods. https:\/\/github.com\/jacobgil\/pytorch-grad-cam (2021)"},{"key":"16_CR15","doi-asserted-by":"crossref","unstructured":"Gu, J., Yang, Y., Tresp, V.: Understanding individual decisions of CNNs via contrastive backpropagation. In: Proceedings of the Computer Vision\u2013ACCV 2018: 14th Asian Conference on Computer Vision, Perth, Australia, December 2\u20136, 2018, Revised Selected Papers, Part III 14, pp. 119\u2013134. Springer (2019)","DOI":"10.1007\/978-3-030-20893-6_8"},{"issue":"2","key":"16_CR16","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1109\/TRPMS.2018.2890359","volume":"3","author":"Z Guo","year":"2019","unstructured":"Guo, Z., Li, X., Huang, H., Guo, N., Li, Q.: Deep learning-based image segmentation on multimodal medical imaging. IEEE Trans. Radiat. Plasma Med. Sci. 3(2), 162\u2013169 (2019)","journal-title":"IEEE Trans. Radiat. Plasma Med. Sci."},{"key":"16_CR17","doi-asserted-by":"crossref","unstructured":"Gupta, S., Lakhotia, S., Rawat, A., Tallamraju, R.: ViTOL: vision transformer for weakly supervised object localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4101\u20134110 (2022)","DOI":"10.1109\/CVPRW56347.2022.00455"},{"key":"16_CR18","unstructured":"Hiley, L., Preece, A., Hicks, Y., Chakraborty, S., Gurram, P., Tomsett, R.: Explaining motion relevance for activity recognition in video deep learning models. arXiv preprint arXiv:2003.14285 (2020)"},{"key":"16_CR19","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"16_CR20","doi-asserted-by":"crossref","unstructured":"Iwana, B.K., Kuroki, R., Uchida, S.: Explaining convolutional neural networks using softmax gradient layer-wise relevance propagation. In: Proceedings of the 2019 IEEE\/CVF ICCVW, pp. 4176\u20134185. IEEE (2019)","DOI":"10.1109\/ICCVW.2019.00513"},{"key":"16_CR21","doi-asserted-by":"publisher","first-page":"239","DOI":"10.1007\/s43154-020-00021-6","volume":"1","author":"K Kleeberger","year":"2020","unstructured":"Kleeberger, K., Bormann, R., Kraus, W., Huber, M.F.: A survey on learning-based robotic grasping. Curr. Robot. Rep. 1, 239\u2013249 (2020)","journal-title":"Curr. Robot. Rep."},{"key":"16_CR22","unstructured":"Krizhevsky, A., Hinton, G., et\u00a0al.: Learning multiple layers of features from tiny images. Master\u2019s thesis, University of Tront (2009)"},{"key":"16_CR23","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, W., Li, Z., Huang, Y., Sato, Y.: Towards visually explaining video understanding networks with perturbation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1120\u20131129 (2021)","DOI":"10.1109\/WACV48630.2021.00116"},{"key":"16_CR24","doi-asserted-by":"crossref","unstructured":"Liao, J., Duan, H., Feng, K., Zhao, W., Yang, Y., Chen, L.: A light weight model for active speaker detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 22932\u201322941 (2023)","DOI":"10.1109\/CVPR52729.2023.02196"},{"key":"16_CR25","doi-asserted-by":"crossref","unstructured":"Liu, Z., Luo, P., Wang, X., Tang, X.: Deep learning face attributes in the wild. In: Proceedings of International Conference on Computer Vision (ICCV) (2015)","DOI":"10.1109\/ICCV.2015.425"},{"key":"16_CR26","unstructured":"Lundberg, S.M., Lee, S.I.: A unified approach to interpreting model predictions. In: Guyon, I., Luxburg, U.V., Bengio, S., Wallach, H., Fergus, R., Vishwanathan, S., Garnett, R. (eds.) Advances in Neural Information Processing Systems 30, pp. 4765\u20134774. Curran Associates, Inc. (2017). http:\/\/papers.nips.cc\/paper\/7062-a-unified-approach-to-interpreting-model-predictions.pdf"},{"key":"16_CR27","unstructured":"Lundberg, S.M., Lee, S.I.: A unified approach to interpreting model predictions. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"16_CR28","doi-asserted-by":"crossref","unstructured":"Muhammad, M.B., Yeasin, M.: Eigen-CAM: class activation map using principal components. In: 2020 International Joint Conference on Neural Networks (IJCNN), pp.\u00a01\u20137. IEEE (2020)","DOI":"10.1109\/IJCNN48605.2020.9206626"},{"key":"16_CR29","doi-asserted-by":"crossref","unstructured":"Ning, E., Wang, C., Zhang, H., Ning, X., Tiwari, P.: Occluded person re-identification with deep learning: a survey and perspectives. Expert Syst. Appl. 122419 (2023)","DOI":"10.1016\/j.eswa.2023.122419"},{"key":"16_CR30","first-page":"24898","volume":"34","author":"B Pan","year":"2021","unstructured":"Pan, B., Panda, R., Jiang, Y., Wang, Z., Feris, R., Oliva, A.: IA-RED2: interpretability-aware redundancy reduction for vision transformers. Adv. Neural. Inf. Process. Syst. 34, 24898\u201324911 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR31","unstructured":"Petsiuk, V., Das, A., Saenko, K.: RISE: randomized input sampling for explanation of black-box models. In: Proceedings of the British Machine Vision Conference (BMVC) (2018)"},{"key":"16_CR32","first-page":"5052","volume":"35","author":"Y Qiang","year":"2022","unstructured":"Qiang, Y., Pan, D., Li, C., Li, X., Jang, R., Zhu, D.: AttCAT: explaining transformers via attentive class activation tokens. Adv. Neural. Inf. Process. Syst. 35, 5052\u20135064 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR33","doi-asserted-by":"crossref","unstructured":"Roth, J., et\u00a0al.: Ava active speaker: an audio-visual dataset for active speaker detection. In: 2020 IEEE ICASSP, pp. 4492\u20134496. IEEE (2020)","DOI":"10.1109\/ICASSP40776.2020.9053900"},{"key":"16_CR34","doi-asserted-by":"publisher","unstructured":"Roxo, T., Costa, J.C., In\u00e1cio, P.R.M., Proen\u00e7a, H.: WASD: a wilder active speaker detection dataset. IEEE Trans. Biometrics Behav. Identity Sci. 1\u20131 (2024). https:\/\/doi.org\/10.1109\/TBIOM.2024.3412821","DOI":"10.1109\/TBIOM.2024.3412821"},{"key":"16_CR35","doi-asserted-by":"crossref","unstructured":"Roxo, T., Costa, J.C., In\u00e1cio, P.R., Proen\u00e7a, H.: On exploring audio anomaly in speech. In: 2023 IEEE International Workshop on Information Forensics and Security (WIFS), pp.\u00a01\u20136. IEEE (2023)","DOI":"10.1109\/WIFS58808.2023.10374734"},{"key":"16_CR36","doi-asserted-by":"publisher","unstructured":"Roxo, T., Proen\u00e7a, H.: Is gender \u201cin-the-wild\u201d inference really a solved problem? IEEE Trans. Biometrics Behav. Identity Sci. 3(4), 573\u2013582 (2021). https:\/\/doi.org\/10.1109\/TBIOM.2021.3100926","DOI":"10.1109\/TBIOM.2021.3100926"},{"key":"16_CR37","doi-asserted-by":"publisher","first-page":"28122","DOI":"10.1109\/ACCESS.2022.3157857","volume":"10","author":"T Roxo","year":"2022","unstructured":"Roxo, T., Proen\u00e7a, H.: YinYang-Net: complementing face and body information for wild gender recognition. IEEE Access 10, 28122\u201328132 (2022). https:\/\/doi.org\/10.1109\/ACCESS.2022.3157857","journal-title":"IEEE Access"},{"issue":"4","key":"16_CR38","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3626961","volume":"13","author":"C Roy","year":"2023","unstructured":"Roy, C., et al.: Explainable activity recognition in videos using deep learning and tractable probabilistic models. ACM Trans. Interact. Intell. Syst. 13(4), 1\u201332 (2023)","journal-title":"ACM Trans. Interact. Intell. Syst."},{"key":"16_CR39","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., Batra, D.: Grad-CAM: visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE ICCV, pp. 618\u2013626 (2017)","DOI":"10.1109\/ICCV.2017.74"},{"key":"16_CR40","unstructured":"Shrikumar, A., Greenside, P., Kundaje, A.: Learning important features through propagating activation differences. In: Proceedings of the ICML, pp. 3145\u20133153. PMLR (2017)"},{"key":"16_CR41","unstructured":"Shrikumar, A., Greenside, P., Shcherbina, A., Kundaje, A.: Not just a black box: Learning important features through propagating activation differences. arXiv preprint arXiv:1605.01713 (2016)"},{"key":"16_CR42","unstructured":"Smilkov, D., Thorat, N., Kim, B., Vi\u00e9gas, F., Wattenberg, M.: SmoothGrad: removing noise by adding noise. arXiv preprint arXiv:1706.03825 (2017)"},{"key":"16_CR43","unstructured":"Srinivas, S., Fleuret, F.: Full-gradient representation for neural network visualization. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"16_CR44","doi-asserted-by":"crossref","unstructured":"Stergiou, A., Kapidis, G., Kalliatakis, G., Chrysoulas, C., Veltkamp, R., Poppe, R.: Saliency tubes: visual explanations for spatio-temporal convolutions. In: 2019 IEEE International Conference on Image Processing (ICIP), pp. 1830\u20131834. IEEE (2019)","DOI":"10.1109\/ICIP.2019.8803153"},{"key":"16_CR45","unstructured":"Sundararajan, M., Taly, A., Yan, Q.: Axiomatic attribution for deep networks. In: Proceedings of the ICML, pp. 3319\u20133328. PMLR (2017)"},{"key":"16_CR46","doi-asserted-by":"crossref","unstructured":"Tao, R., Pan, Z., Das, R.K., Qian, X., Shou, M.Z., Li, H.: Is someone speaking? Exploring long-term temporal features for audio-visual active speaker detection. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 3927\u20133935 (2021)","DOI":"10.1145\/3474085.3475587"},{"issue":"11","key":"16_CR47","doi-asserted-by":"publisher","first-page":"1958","DOI":"10.1109\/TPAMI.2008.128","volume":"30","author":"A Torralba","year":"2008","unstructured":"Torralba, A., Fergus, R., Freeman, W.T.: 80 million tiny images: a large data set for nonparametric object and scene recognition. IEEE Trans. Pattern Anal. Mach. Intell. 30(11), 1958\u20131970 (2008)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"16_CR48","doi-asserted-by":"crossref","unstructured":"Winter, M., Bailer, W., Thallinger, G.: Demystifying face-recognition with locally interpretable boosted features (LIBF). In: 2022 10th EUVIP, pp.\u00a01\u20136. IEEE (2022)","DOI":"10.1109\/EUVIP53989.2022.9922905"},{"key":"16_CR49","doi-asserted-by":"crossref","unstructured":"Yin, B., Tran, L., Li, H., Shen, X., Liu, X.: Towards interpretable face recognition. In: Proceedings of the IEEE\/CVF ICCV, pp. 9348\u20139357 (2019)","DOI":"10.1109\/ICCV.2019.00944"},{"key":"16_CR50","unstructured":"Yuan, T., Li, X., Xiong, H., Cao, H., Dou, D.: Explaining information flow inside vision transformers using markov chain. In: eXplainable AI Approaches for Debugging and Diagnosis. (2021)"},{"issue":"10","key":"16_CR51","doi-asserted-by":"publisher","first-page":"1084","DOI":"10.1007\/s11263-017-1059-x","volume":"126","author":"J Zhang","year":"2018","unstructured":"Zhang, J., Bargal, S.A., Lin, Z., Brandt, J., Shen, X., Sclaroff, S.: Top-down neural attention by excitation backprop. Int. J. Comput. Vision 126(10), 1084\u20131102 (2018)","journal-title":"Int. J. Comput. Vision"},{"issue":"9","key":"16_CR52","doi-asserted-by":"publisher","first-page":"2131","DOI":"10.1109\/TPAMI.2018.2858759","volume":"41","author":"B Zhou","year":"2018","unstructured":"Zhou, B., Bau, D., Oliva, A., Torralba, A.: Interpreting deep visual representations via network dissection. IEEE TPAMI 41(9), 2131\u20132145 (2018)","journal-title":"IEEE TPAMI"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-92089-9_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T12:49:28Z","timestamp":1748090968000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-92089-9_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031920882","9783031920899"],"references-count":52,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-92089-9_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}