{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,30]],"date-time":"2026-01-30T12:49:57Z","timestamp":1769777397264,"version":"3.49.0"},"reference-count":84,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276192"],"award-info":[{"award-number":["62276192"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Image Process."],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/tip.2026.3654367","type":"journal-article","created":{"date-parts":[[2026,1,21]],"date-time":"2026-01-21T21:12:53Z","timestamp":1769029973000},"page":"872-887","source":"Crossref","is-referenced-by-count":0,"title":["SigMa: Semantic Similarity-Guided Semi-Dense Feature Matching"],"prefix":"10.1109","volume":"35","author":[{"given":"Xiang","family":"Fang","sequence":"first","affiliation":[{"name":"Electronic Information School, Wuhan University, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0986-4924","authenticated-orcid":false,"given":"Zizhuo","family":"Li","sequence":"additional","affiliation":[{"name":"Electronic Information School, Wuhan University, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3264-3265","authenticated-orcid":false,"given":"Jiayi","family":"Ma","sequence":"additional","affiliation":[{"name":"Electronic Information School and the School of Robotics, Wuhan University, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1017\/cbo9780511811685"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.5120\/17374-7818"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2007.1052"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2008.190"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2014.2307478"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1117-z"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3334515"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.isprsjprs.2024.11.004"},{"key":"ref9","article-title":"Selecting and pruning: A differentiable causal sequentialized state-space model for two-view correspondence learning","author":"Fang","year":"2025","journal-title":"arXiv:2503.17938"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-020-01359-2"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00881"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02047"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-26313-2_16"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19824-3_2"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25341"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2024.3473301"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01391"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref19","first-page":"5156","article-title":"Transformers are RNNs: Fast autoregressive transformers with linear attention","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"1","author":"Katharopoulos"},{"key":"ref20","article-title":"QuadTree attention for vision transformers","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Tang"},{"key":"ref21","article-title":"Efficientvit: Enhanced linear attention for high-resolution low-computation visual recognition","author":"Cai","year":"2022","journal-title":"arXiv:2205.14756"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5539963"},{"key":"ref23","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab","year":"2023","journal-title":"arXiv:2304.07193"},{"key":"ref24","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref25","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1023\/b:visi.0000029664.99615.94"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2007.09.014"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126544"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2018.00060"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46466-4_28"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2023.3271000"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/3DV62453.2024.00035"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00499"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00343"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00624"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01616"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/358669.358692"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.257"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01044"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.221"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00282"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00594"},{"key":"ref43","article-title":"DeCo: Decoupling token compression from semantic abstraction in multimodal large language models","author":"Yao","year":"2024","journal-title":"arXiv:2405.20985"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00525"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/iccv51070.2023.00885"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01704"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01871"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3369699"},{"key":"ref49","first-page":"1363","article-title":"Emergent correspondence from image diffusion","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Tang"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-96-0911-6_4"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00504"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01878"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01911"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681021"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01352"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02529"},{"key":"ref61","first-page":"38504","article-title":"Leveraging vision-centric multi-modal expertise for 3D object detection","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Huang"},{"key":"ref62","article-title":"Qwen3 technical report","volume-title":"arXiv:2505.09388","author":"Yang","year":"2025"},{"key":"ref63","first-page":"7480","article-title":"Scaling vision transformers to 22 billion parameters","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dehghani"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"ref65","first-page":"15816","article-title":"Learnable Fourier features for multi-dimensional spatial positional encoding","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Li"},{"key":"ref66","article-title":"The llama 3 herd of models","author":"Grattafiori","year":"2024","journal-title":"arXiv:2407.21783"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00218"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681069"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i10.33095"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.110094"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2024.3370781"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.410"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2870930"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00897"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00752"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i2.27915"},{"key":"ref78","volume-title":"Image Matching Challenge 2022","author":"Howard","year":"2022"},{"key":"ref79","first-page":"16344","article-title":"FlashAttention: Fast and memory-efficient exact attention with IO-awareness","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dao"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.458"},{"key":"ref81","article-title":"FastDINOv2: Frequency based curriculum learning improves robustness and training speed","author":"Zhang","year":"2025","journal-title":"arXiv:2507.03779"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2025.3592065"},{"key":"ref83","first-page":"9113","article-title":"Not all views are created equal: Analyzing viewpoint instabilities in vision foundation models","volume-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis.","author":"Michalkiewicz"},{"key":"ref84","article-title":"Towards robust semantic correspondence: A benchmark and insights","author":"Chong","year":"2025","journal-title":"arXiv:2508.00272"}],"container-title":["IEEE Transactions on Image Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/83\/11355710\/11360609.pdf?arnumber=11360609","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T21:27:17Z","timestamp":1769722037000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11360609\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":84,"URL":"https:\/\/doi.org\/10.1109\/tip.2026.3654367","relation":{},"ISSN":["1057-7149","1941-0042"],"issn-type":[{"value":"1057-7149","type":"print"},{"value":"1941-0042","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}