{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T17:09:08Z","timestamp":1780765748337,"version":"3.54.1"},"reference-count":63,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T00:00:00Z","timestamp":1743465600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2023YFE0204200"],"award-info":[{"award-number":["2023YFE0204200"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U20A20387"],"award-info":[{"award-number":["U20A20387"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62376243"],"award-info":[{"award-number":["62376243"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Youth Foundation of China","doi-asserted-by":"publisher","award":["62307032"],"award-info":[{"award-number":["62307032"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Circuits Syst. Video Technol."],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1109\/tcsvt.2024.3403167","type":"journal-article","created":{"date-parts":[[2024,5,20]],"date-time":"2024-05-20T13:35:00Z","timestamp":1716212100000},"page":"3024-3038","source":"Crossref","is-referenced-by-count":5,"title":["Cross-Modality Image Interpretation via Concept Decomposition Vector of Visual-Language Models"],"prefix":"10.1109","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3270-0952","authenticated-orcid":false,"given":"Zhengqing","family":"Fang","sequence":"first","affiliation":[{"name":"Department of Ophthalmology, Sir Run Run Shaw Hospital, Zhejiang University School of Medicine, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhouhang","family":"Yuan","sequence":"additional","affiliation":[{"name":"Department of Ophthalmology, Sir Run Run Shaw Hospital, Zhejiang University School of Medicine, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ziyu","family":"Li","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0415-6937","authenticated-orcid":false,"given":"Jingyuan","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Education, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7024-9790","authenticated-orcid":false,"given":"Kun","family":"Kuang","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu-Feng","family":"Yao","sequence":"additional","affiliation":[{"name":"Department of Ophthalmology, Sir Run Run Shaw Hospital, Zhejiang University School of Medicine, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2139-8807","authenticated-orcid":false,"given":"Fei","family":"Wu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-019-0048-x"},{"issue":"1","key":"ref2","first-page":"1","article-title":"A theoretical computer science perspective on consciousness","volume":"8","author":"Blum","year":"2021","journal-title":"J. Artif. Intell. Consciousness"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.eng.2021.08.016"},{"key":"ref4","first-page":"5338","article-title":"Concept bottleneck models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Koh"},{"key":"ref5","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"139","author":"Radford"},{"key":"ref6","first-page":"1","article-title":"Post-hoc concept bottleneck models","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Yuksekgonul"},{"key":"ref7","first-page":"1","article-title":"Label-free concept bottleneck models","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Oikarinen"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01839"},{"key":"ref9","first-page":"51","article-title":"Pre-trained language models and their applications","volume":"25","author":"Wang","year":"2023","journal-title":"Engineering"},{"key":"ref10","first-page":"17612","article-title":"Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liang"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.00371"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-psych-122216-011829"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.eng.2022.04.021"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.eng.2022.10.017"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00518"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.74"},{"key":"ref17","first-page":"2668","article-title":"Interpretability beyond feature attribution: Quantitative testing with concept activation vectors (TCAV)","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Kim"},{"key":"ref18","first-page":"2376","article-title":"Counterfactual visual explanations","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Goyal"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N16-3020"},{"key":"ref20","article-title":"This looks like that: Deep learning for interpretable image recognition","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Chen"},{"key":"ref21","first-page":"1761","article-title":"Interpretable3D: An ad-hoc interpretable classifier for 3D point clouds","volume-title":"Proc. 38th AAAI Conf. Artif. Intell.","volume":"32","author":"Tuo"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00269"},{"key":"ref23","article-title":"Enabling collaborative clinical diagnosis of infectious keratitis by integrating expert knowledge and interpretable data-driven intelligence","author":"Fang","year":"2024","journal-title":"arXiv:2401.08695"},{"key":"ref24","first-page":"4320","article-title":"Neural insights for digital marketing content design","volume-title":"Proc. 29th ACM SIGKDD Conf. Knowl. Discovery Data Mining","author":"Kong"},{"key":"ref25","article-title":"Visual recognition with deep nearest centroids","author":"Wang","year":"2023","journal-title":"arXiv:2209.07383"},{"key":"ref26","first-page":"1","article-title":"This looks like those: Illuminating prototypical concepts using multiple visualizations","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Ma"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01002"},{"key":"ref28","first-page":"21400","article-title":"Concept embedding models: Beyond the accuracy-explainability trade-off","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Zarlenga"},{"key":"ref29","article-title":"Promises and pitfalls of black-box concept learning models","author":"Mahinpei","year":"2021","journal-title":"arXiv:2106.13314"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2024.3369699"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00271"},{"key":"ref32","first-page":"16784","article-title":"GLIDE: Towards photorealistic image generation and editing with text-guided diffusion models","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Nichol"},{"key":"ref33","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. Int. Conf. Machine Learn.","author":"Li"},{"key":"ref34","first-page":"36067","article-title":"GLIPv2: Unifying localization and vision-language understanding","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref35","article-title":"When are lemons purple? The concept association bias of CLIP","author":"Yamada","year":"2022","journal-title":"arXiv:2212.12043"},{"key":"ref36","first-page":"1","article-title":"Visual classification via description from large language models","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Menon"},{"key":"ref37","article-title":"A systematic survey of prompt engineering on vision-language foundation models","author":"Gu","year":"2023","journal-title":"arXiv:2307.12980"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1016\/j.metrad.2023.100047"},{"key":"ref39","article-title":"CLIP-adapter: Better vision-language models with feature adapters","author":"Gao","year":"2021","journal-title":"arXiv:2110.04544"},{"key":"ref40","article-title":"Tip-adapter: Training-free CLIP-adapter for better vision-language modeling","author":"Zhang","year":"2021","journal-title":"arXiv:2111.03930"},{"key":"ref41","article-title":"LoRa: Low-rank adaptation of large language models","author":"Hu","year":"2021","journal-title":"arXiv:2106.09685"},{"key":"ref42","article-title":"Visual-language prompt tuning with knowledge-guided context optimization","author":"Yao","year":"2023","journal-title":"arXiv:2303.13283"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-023-1385-0"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00361"},{"key":"ref45","article-title":"The segment anything model (SAM) for remote sensing applications: From zero to one shot","volume":"124","author":"Osco","year":"2023","journal-title":"Int. J. Appl. Earth Observ. Geoinf."},{"key":"ref46","article-title":"Segment anything model for medical image analysis: An experimental study","volume":"89","author":"Mazurowski","year":"2023","journal-title":"Med. Image Anal."},{"key":"ref47","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Brown"},{"key":"ref48","first-page":"5583","article-title":"ViLT: Vision-and-language transformer without convolution or region supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kim"},{"key":"ref49","volume-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.461"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1212.0402"},{"key":"ref52","volume-title":"The Caltech-UCSD birds-200\u20132011 dataset","author":"Wah","year":"2011"},{"key":"ref53","article-title":"Fine-grained visual classification of aircraft","author":"Maji","year":"2013","journal-title":"arXiv:1306.5151"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2019.2918242"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1038\/sdata.2018.161"},{"key":"ref56","volume-title":"Aptos 2019 Blindness Detection","author":"Karthik","year":"2019"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413557"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref59","first-page":"11205","article-title":"Leveraging sparse linear layers for debuggable deep networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wong"},{"key":"ref60","article-title":"Towards automatic concept-based explanations","volume-title":"Proc. 33rd Int. Conf. Neural Inf. Process. Syst.","author":"Ghorbani"},{"key":"ref61","article-title":"Developing a fidelity evaluation approach for interpretable machine learning","author":"Velmurugan","year":"2021","journal-title":"arXiv:2106.08492"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/747"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1145\/3583558"}],"container-title":["IEEE Transactions on Circuits and Systems for Video Technology"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/76\/10949577\/10535313.pdf?arnumber=10535313","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,5]],"date-time":"2026-01-05T18:40:55Z","timestamp":1767638455000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10535313\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4]]},"references-count":63,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tcsvt.2024.3403167","relation":{},"ISSN":["1051-8215","1558-2205"],"issn-type":[{"value":"1051-8215","type":"print"},{"value":"1558-2205","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4]]}}}