{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T12:29:29Z","timestamp":1777984169840,"version":"3.51.4"},"reference-count":72,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Hong Kong Innovation and Technology Commission"},{"name":"Institute of Digital Medicine, City University of Hong Kong","award":["9229503"],"award-info":[{"award-number":["9229503"]}]},{"name":"Institute of Digital Medicine, City University of Hong Kong","award":["9610460"],"award-info":[{"award-number":["9610460"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["12561095"],"award-info":[{"award-number":["12561095"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Special Posts of Guizhou University","award":["[2025]06"],"award-info":[{"award-number":["[2025]06"]}]},{"name":"Guizhou Provincial Major Project of Basic Research Program"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/tpami.2025.3646452","type":"journal-article","created":{"date-parts":[[2025,12,19]],"date-time":"2025-12-19T18:58:30Z","timestamp":1766170710000},"page":"4792-4809","source":"Crossref","is-referenced-by-count":2,"title":["The CUR Decomposition of Self-Attention Matrices in Vision Transformers"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3405-742X","authenticated-orcid":false,"given":"Chong","family":"Wu","sequence":"first","affiliation":[{"name":"Department of Electrical Engineering, City University of Hong Kong, Hong Kong SAR, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9956-062X","authenticated-orcid":false,"given":"Maolin","family":"Che","sequence":"additional","affiliation":[{"name":"School of Mathematics and Statistics and State Key Laboratory of Public Big Data, Guizhou University, Guiyang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9661-3095","authenticated-orcid":false,"given":"Hong","family":"Yan","sequence":"additional","affiliation":[{"name":"Department of Electrical Engineering, City University of Hong Kong, Hong Kong SAR, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21386"},{"key":"ref3","article-title":"GreaseLM: Graph REASoning enhanced language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang","year":"2022"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19962"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i16.17664"},{"key":"ref6","first-page":"21297","article-title":"SOFT: Softmax-free Transformer with linear complexity","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Lu","year":"2021"},{"key":"ref7","first-page":"5156","article-title":"Transformers are RNNs: Fast autoregressive transformers with linear attention","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","volume":"119","author":"Katharopoulos","year":"2020"},{"key":"ref8","article-title":"Rethinking attention with performers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Choromanski","year":"2021"},{"key":"ref9","article-title":"cosFormer: Rethinking softmax in attention","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Qin","year":"2022"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1137\/18M1183480"},{"key":"ref11","article-title":"Linformer: Self-attention with linear complexity","author":"Wang","year":"2020"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2840"},{"key":"ref13","article-title":"Explicit sparse transformer: Concentrated attention through explicit selection","author":"Zhao","year":"2019"},{"key":"ref14","article-title":"LongFormer: The long-document transformer","author":"Beltagy","year":"2020"},{"key":"ref15","article-title":"Generating long sequences with sparse transformers","author":"Child","year":"2019"},{"key":"ref16","first-page":"9438","article-title":"Sparse sinkhorn attention","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","volume":"119","author":"Tay","year":"2020"},{"key":"ref17","first-page":"17283","article-title":"Big Bird: Transformers for longer sequences","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Zaheer","year":"2020"},{"key":"ref18","article-title":"Reformer: The efficient transformer","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kitaev","year":"2020"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1189"},{"key":"ref20","article-title":"FlashAttention-2: Faster attention with better parallelism and work partitioning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dao","year":"2024"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2193"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1126"},{"key":"ref23","article-title":"DuSA: Fast and accurate dual-stage sparse attention mechanism accelerating both training and inference","volume-title":"Proc. 39th Annu. Conf. Neural Inf. Process. Syst.","author":"Wu","year":"2025"},{"key":"ref24","article-title":"Random feature attention","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Peng","year":"2021"},{"key":"ref25","article-title":"Replacing softmax with ReLU in vision transformers","author":"Wortsman","year":"2023"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-25082-8_3"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01587"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00548"},{"key":"ref29","first-page":"3531","article-title":"Efficient attention: Attention with linear complexities","volume-title":"Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis.","author":"Shen","year":"2021"},{"key":"ref30","article-title":"Interactive multi-head self-attention with linear complexity","author":"Kang","year":"2024"},{"key":"ref31","article-title":"Is attention better than matrix decomposition?","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Geng","year":"2021"},{"key":"ref32","article-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy","year":"2021"},{"key":"ref33","article-title":"Recent advances in vision transformer: A survey and outlook of recent work","author":"Islam","year":"2022"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00599"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20252"},{"key":"ref36","first-page":"23818","article-title":"Efficient training of visual Transformers with small datasets","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Liu Sangineto","year":"2021"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref38","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"Proc. 38th Int. Conf. Mach. Learn.","volume":"139","author":"Touvron","year":"2021"},{"key":"ref39","first-page":"12633","article-title":"Global context vision transformers","volume-title":"Proc. 40th Int. Conf. Mach. Learn.","volume":"202","author":"Hatamizadeh","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1016\/S0024-3795(96)00301-1"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/PL00005410"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/s006070070031"},{"issue":"72","key":"ref43","first-page":"2153","article-title":"On the nystr\u00f6m method for approximating a gram matrix for improved kernel-based learning","volume":"6","author":"Drineas","year":"2005","journal-title":"J. Mach. Learn. Res."},{"key":"ref44","first-page":"5672","article-title":"On fast leverage score sampling and optimal learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Rudi","year":"2018"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1137\/110852310"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1137\/07070471X"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1006\/jfan.1998.3384"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1137\/140978430"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.0803205106"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"ref51","article-title":"Learning multiple layers of features from tiny images","author":"Krizhevsky","year":"2009"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754825"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1155\/2014\/563787"},{"key":"ref55","article-title":"Training compute-optimal large language models","author":"Hoffmann","year":"2022"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01157"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01414"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2017.544"},{"key":"ref60","article-title":"Fixing weight decay regularization in Adam","author":"Loshchilov","year":"2017"},{"key":"ref61","first-page":"5785","article-title":"FastViT: A fast hybrid vision transformer using structural reparameterization","volume-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis.","author":"Vasu","year":"2023"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02035-5"},{"key":"ref63","article-title":"How to train vision transformer on small-scale datasets?","volume-title":"Proc. 33rd Brit. Mach. Vis. Conf.","author":"Gani","year":"2022"},{"key":"ref64","first-page":"103031","article-title":"VMamba: Visual state space model","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"37","author":"Liu","year":"2024"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01167"},{"key":"ref66","article-title":"Long range arena : A benchmark for efficient transformers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Tay","year":"2021"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00020"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1580"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-4827"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1016\/j.laa.2024.01.001"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_26"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11424231\/11304748.pdf?arnumber=11304748","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T01:34:15Z","timestamp":1773106455000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11304748\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":72,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2025.3646452","relation":{"has-preprint":[{"id-type":"doi","id":"10.36227\/techrxiv.171392846.60982484\/v3","asserted-by":"object"},{"id-type":"doi","id":"10.36227\/techrxiv.171392846.60982484\/v4","asserted-by":"object"},{"id-type":"doi","id":"10.36227\/techrxiv.171392846.60982484\/v2","asserted-by":"object"}]},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}