{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T23:42:09Z","timestamp":1786059729020,"version":"3.56.0"},"reference-count":232,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1109\/tpami.2026.3651319","type":"journal-article","created":{"date-parts":[[2026,1,12]],"date-time":"2026-01-12T22:00:30Z","timestamp":1768255230000},"page":"5672-5691","source":"Crossref","is-referenced-by-count":9,"title":["Advances in Multimodal Adaptation and Generalization: From Traditional Approaches to Foundation Models"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-8562-1946","authenticated-orcid":false,"given":"Hao","family":"Dong","sequence":"first","affiliation":[{"name":"ETH Z&#x00FC;rich, Z&#x00FC;rich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Moru","family":"Liu","sequence":"additional","affiliation":[{"name":"Technical University of Munich, Munich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8153-3903","authenticated-orcid":false,"given":"Kaiyang","family":"Zhou","sequence":"additional","affiliation":[{"name":"Hong Kong Baptist University, Hong Kong, SAR, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6870-240X","authenticated-orcid":false,"given":"Eleni","family":"Chatzi","sequence":"additional","affiliation":[{"name":"ETH Z&#x00FC;rich, Z&#x00FC;rich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Juho","family":"Kannala","sequence":"additional","affiliation":[{"name":"Aalto University, AALTO, Finland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1173-6972","authenticated-orcid":false,"given":"Cyrill","family":"Stachniss","sequence":"additional","affiliation":[{"name":"Center for Robotics, University of Bonn, Bonn, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9546-1488","authenticated-orcid":false,"given":"Olga","family":"Fink","sequence":"additional","affiliation":[{"name":"EPFL, Lausanne, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2018.05.083"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2022.3178128"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00503"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5757"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00361"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00087"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11596"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00020"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01262"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01471"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00270"},{"key":"ref12","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2021"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref14","first-page":"1","article-title":"Using language to extend to unseen domains","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dunlap","year":"2023"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01073"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3195549"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3369699"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240633"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3052083"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548313"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00966"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01336"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73202-7_16"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475660"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548009"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01342"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547990"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612320"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610316"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00015"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00120"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547987"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25400"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25448"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00702"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/j.isprsjprs.2021.04.012"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10341473"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19839-7_28"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01991"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/WACVW60836.2024.00070"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2902100"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00346"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.859"},{"key":"ref45","first-page":"1","article-title":"Test-time adaption against multi-modal reliability bias","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yang","year":"2024"},{"key":"ref46","first-page":"1","article-title":"Towards robust multimodal open-set test-time adaptation via adaptive entropy-aware optimization","volume-title":"Proc. ICLR","author":"Dong","year":"2025"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02524"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ijcnn64981.2025.11228216"},{"key":"ref49","first-page":"1","article-title":"Test-time adaptation for combating missing modalities in egocentric videos","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ramazanova","year":"2025"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01642"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01724"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73390-1_14"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01939"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i6.28398"},{"key":"ref55","first-page":"78674","article-title":"SIMMMDG: A simple and effective framework for multi-modal domain generalization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dong","year":"2023"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00024"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2133"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01068"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72933-1_6"},{"key":"ref60","first-page":"1","article-title":"Leveraging generative foundation models for domain generalization","volume-title":"Proc. Int. Conf. Mach. Learn. Workshop","author":"Hemati","year":"2024"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00300"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01264"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01707"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01439"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02209"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00946"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02218"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02238"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02258"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1527\/tjsai.38-6_b-mc2"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3327962"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3243"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02227"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/tetci.2025.3628755"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73650-6_2"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2025.3587430"},{"key":"ref77","article-title":"Clip the divergence: Language-guided unsupervised domain adaptation","author":"Zhu","year":"2024"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72775-7_4"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0770"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00267"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00297"},{"key":"ref82","first-page":"93964","article-title":"Unsupervised modality adaptation with text-to-image diffusion models for semantic segmentation","volume-title":"Proc. NeurIPS","author":"Xia","year":"2024"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02215-3"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20050-2_26"},{"key":"ref85","first-page":"31716","article-title":"CLIPood: Generalizing clip to out-of-distributions","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Shu","year":"2023"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01480"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72946-1_9"},{"key":"ref88","first-page":"4267","article-title":"CLIPCEIL: Domain generalization through clip via channel refinement and image-text alignment","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yu","year":"2024"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3245584"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02225"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00514"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01398"},{"key":"ref94","article-title":"Exploring visual prompts for adapting large-scale models","author":"Bahng","year":"2022"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/tcsvt.2026.3651774"},{"key":"ref96","first-page":"1","article-title":"Unleashing the power of visual prompting at the pixel level","author":"Wu","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref97","article-title":"Unified vision and language prompt learning","author":"Zang","year":"2022"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00556"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3291588"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72973-7_18"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01891-x"},{"key":"ref103","article-title":"SVL-Adapter: Self-supervised adapter for vision-language pretrained models","author":"Pantazis","year":"2022"},{"key":"ref104","article-title":"Improving zero-shot models with label distribution priors","author":"Kahana","year":"2022"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3311646"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01343"},{"key":"ref107","first-page":"493","article-title":"TIP-Adapter: Training-free clip-adapter for better vision-language modeling","volume-title":"Proc. ECCV","author":"Zhang","year":"2022"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00780"},{"key":"ref109","first-page":"1","article-title":"Masked unsupervised self-training for label-free image classification","volume-title":"Proc. ICLR","author":"Li","year":"2023"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00257"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25152"},{"key":"ref112","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02245"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02713"},{"key":"ref114","first-page":"1","article-title":"A hard-to-beat baseline for training-free clip-based adaptation","volume-title":"Proc. ICLR","author":"Wang","year":"2024"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01438"},{"key":"ref116","first-page":"1","article-title":"Visual classification via description from large language models","volume-title":"Proc. ICLR","author":"Menon","year":"2023"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01234"},{"key":"ref118","first-page":"14274","article-title":"Test-time prompt tuning for zero-shot generalization in vision-language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shu","year":"2022"},{"key":"ref119","first-page":"48015","article-title":"WATT: Weight average test-time adaption of clip","volume-title":"Proc. NeurIPS","author":"Osowiechi","year":"2024"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00255"},{"key":"ref121","doi-asserted-by":"publisher","DOI":"10.52202\/079017-4099"},{"key":"ref122","first-page":"65252","article-title":"Swapprompt: Test-time prompt adaptation for vision-language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ma","year":"2024"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_40"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02207"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01755"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58583-9_29"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4899-7502-7_79-1"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00705"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01237-3_9"},{"key":"ref130","first-page":"1","article-title":"Recall and refine: A simple but effective source-free open-set domain adaptation framework","author":"Nejjar","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref131","first-page":"16282","article-title":"Universal domain adaptation through self supervision","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Saito","year":"2020"},{"key":"ref132","article-title":"Deep domain confusion: Maximizing for domain invariance","author":"Tzeng","year":"2014"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00233"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00950"},{"key":"ref135","first-page":"1","article-title":"TENT: Fully test-time adaptation by entropy minimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Wang","year":"2021"},{"key":"ref136","first-page":"16888","article-title":"Efficient test-time model adaptation without forgetting","volume-title":"Proc. 39th Int. Conf. Mach. Learn.","author":"Niu","year":"2022"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00706"},{"key":"ref138","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.167"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00560"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"ref142","first-page":"25","article-title":"Self-supervised multimodal versatile networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Alayrac","year":"2020"},{"key":"ref143","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_48"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3429301"},{"key":"ref145","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown","year":"2020"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref147","first-page":"1","article-title":"DINOV2: Learning robust visual features without supervision","volume-title":"Trans. Mach. Learn. Res.","author":"Oquab","year":"2024"},{"key":"ref148","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1723"},{"key":"ref149","doi-asserted-by":"publisher","DOI":"10.1007\/s13042-024-02443-6"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-58347-1_10"},{"key":"ref151","first-page":"18661","article-title":"Supervised contrastive learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Khosla","year":"2020"},{"key":"ref152","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01431"},{"key":"ref153","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"ref154","article-title":"A2D2: Audi autonomous driving dataset","author":"Geyer","year":"2020"},{"key":"ref155","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00939"},{"key":"ref156","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612013"},{"key":"ref157","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02519"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2020.2972701"},{"key":"ref159","doi-asserted-by":"publisher","DOI":"10.1109\/JBHI.2022.3162118"},{"key":"ref160","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00711"},{"key":"ref161","first-page":"1","article-title":"Test-time adaptation for cross-modal retrieval with query shift","volume-title":"Proc. ICLR","author":"Li","year":"2025"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/650"},{"key":"ref163","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i16.33867"},{"key":"ref164","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2023.3336611"},{"key":"ref165","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-01998-9"},{"key":"ref166","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095723"},{"key":"ref167","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3519900"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-96-6960-8_1"},{"key":"ref169","article-title":"Denoising and alignment: Rethinking domain generalization for multimodal face anti-spoofing","author":"Ma","year":"2025"},{"key":"ref170","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00314"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00236"},{"key":"ref172","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72907-2_27"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01667"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72646-0_25"},{"key":"ref175","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01314"},{"key":"ref176","first-page":"1","article-title":"DynAlign: Unsupervised dynamic taxonomy alignment for cross-domain segmentation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Sun","year":"2025"},{"key":"ref177","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0749"},{"key":"ref178","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01134"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02205"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01107"},{"key":"ref181","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2024.104230"},{"key":"ref182","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"ref183","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01435"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.224"},{"key":"ref185","first-page":"1","article-title":"Prompt learning with optimal transport for vision-language models","volume-title":"Proc. ICLR","author":"Chen","year":"2023"},{"key":"ref186","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3525"},{"key":"ref187","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01394"},{"key":"ref188","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72655-2_4"},{"key":"ref189","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72661-3_19"},{"key":"ref190","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00361"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2024.103310"},{"key":"ref192","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73661-2_11"},{"key":"ref193","first-page":"39224","article-title":"Rethinking misalignment in vision-language model adaptation from a causal perspective","volume-title":"Proc. NeurIPS","author":"Zhang","year":"2024"},{"key":"ref194","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01852"},{"key":"ref195","article-title":"Vt-clip: Enhancing vision-language models with visual-guided texts","author":"Zhang","year":"2021"},{"key":"ref196","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01049"},{"key":"ref197","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01424"},{"key":"ref198","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00166"},{"key":"ref199","first-page":"39224","article-title":"Lora: Low-rank adaptation of large language models","volume-title":"Proc. ICLR","author":"Hu","year":"2021"},{"key":"ref200","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2024.3468884"},{"key":"ref201","first-page":"1","article-title":"In search of forgotten domain generalization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Mayilvahanan","year":"2025"},{"key":"ref202","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00772"},{"key":"ref203","article-title":"UCF101: A dataset of 101 human actions classes from videos in the wild","author":"Soomro","year":"2012"},{"key":"ref204","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"ref205","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-15552-9_29"},{"key":"ref206","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.352"},{"key":"ref207","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_7"},{"key":"ref208","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.350"},{"key":"ref209","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr42600.2020.00271"},{"key":"ref210","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.534"},{"key":"ref211","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02704"},{"key":"ref212","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.208"},{"key":"ref213","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.572"},{"key":"ref214","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.591"},{"key":"ref215","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00149"},{"key":"ref216","first-page":"5637","article-title":"Wilds: A benchmark of in-the-wild distribution shifts","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Koh","year":"2021"},{"key":"ref217","article-title":"A short note about kinetics-600","author":"Carreira","year":"2018"},{"key":"ref218","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00252"},{"key":"ref219","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01059"},{"key":"ref220","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-009-5152-4"},{"key":"ref221","first-page":"23519","article-title":"Towards a theoretical framework of out-of-distribution generalization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Ye","year":"2021"},{"key":"ref222","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2501"},{"key":"ref223","first-page":"129250","article-title":"MultiOOD: Scaling out-of-distribution detection for multiple modalities","volume-title":"Proc. NeurIPS","author":"Dong","year":"2024"},{"key":"ref224","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01130"},{"key":"ref225","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3264"},{"key":"ref226","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00638"},{"key":"ref227","first-page":"1","article-title":"Rethinking the value of network pruning","volume-title":"Proc. ICLR","author":"Liu","year":"2019"},{"key":"ref228","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"ref229","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.06083"},{"key":"ref230","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3403"},{"key":"ref231","article-title":"Explainable artificial intelligence: Understanding, visualizing and interpreting deep learning models","author":"Samek","year":"2017"},{"key":"ref232","article-title":"Visionary-R1: Mitigating shortcuts in visual reasoning with reinforcement learning","author":"Xia","year":"2025"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11474534\/11342305.pdf?arnumber=11342305","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T19:56:30Z","timestamp":1775505390000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11342305\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":232,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2026.3651319","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5]]}}}