{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T18:19:58Z","timestamp":1783361998643,"version":"3.54.6"},"reference-count":84,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62306022"],"award-info":[{"award-number":["62306022"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62236010"],"award-info":[{"award-number":["62236010"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62576009"],"award-info":[{"award-number":["62576009"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62576017"],"award-info":[{"award-number":["62576017"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62561002"],"award-info":[{"award-number":["62561002"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004772","name":"Natural Science Foundation of Ningxia Province","doi-asserted-by":"publisher","award":["2025AAC020006"],"award-info":[{"award-number":["2025AAC020006"]}],"id":[{"id":"10.13039\/501100004772","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013047","name":"Cultivating Plan Program for the Leader in Science and Technology of Yunnan Province","doi-asserted-by":"publisher","award":["2025GKLRLX26"],"award-info":[{"award-number":["2025GKLRLX26"]}],"id":[{"id":"10.13039\/501100013047","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Science Review"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.cosrev.2026.101003","type":"journal-article","created":{"date-parts":[[2026,5,27]],"date-time":"2026-05-27T13:12:54Z","timestamp":1779887574000},"page":"101003","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Video vision transformer: to discover the \"four secrets\" of video patches"],"prefix":"10.1016","volume":"62","author":[{"given":"Yuxia","family":"Niu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lifang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huiling","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huiyu","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cosrev.2026.101003_bib0001","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"1","key":"10.1016\/j.cosrev.2026.101003_bib0002","doi-asserted-by":"crossref","first-page":"154","DOI":"10.1109\/TPAMI.2018.2876404","article-title":"Neural machine translation with deep attention","volume":"42","author":"Zhang","year":"2018","journal-title":"IEEe Trans. Pattern. Anal. Mach. Intell."},{"issue":"140","key":"10.1016\/j.cosrev.2026.101003_bib0003","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.cosrev.2026.101003_bib0004","series-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies","first-page":"4171","article-title":"Bert: pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.cosrev.2026.101003_bib0005","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018","journal-title":"OpenAI blog."},{"issue":"8","key":"10.1016\/j.cosrev.2026.101003_bib0006","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog."},{"key":"10.1016\/j.cosrev.2026.101003_bib0007","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"8081","key":"10.1016\/j.cosrev.2026.101003_bib0008","doi-asserted-by":"crossref","first-page":"633","DOI":"10.1038\/s41586-025-09422-z","article-title":"DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning","volume":"645","author":"Guo","year":"2025","journal-title":"Nature"},{"key":"10.1016\/j.cosrev.2026.101003_bib0009","unstructured":"Comanici G., Bieber E., Schaekermann M., et al. Gemini 2.5: pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities. arXiv preprint arXiv:2507.06261, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0010","unstructured":"Lenz B., Lieber O., Arazi A., et al. Jamba: hybrid transformer-mamba language models\/\/the thirteenth international conference on learning representations. 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0011","unstructured":"Dosovitskiy A., Beyer L., Kolesnikov A., et al. An image is worth 16x16 words: transformers for image recognition at scale. arxiv preprint arxiv:2010.11929, 2020."},{"key":"10.1016\/j.cosrev.2026.101003_bib0012","article-title":"Identity-mapping ResFormer: a computer-aided diagnosis model for pneumonia X-ray images","author":"Zhou","year":"2025","journal-title":"IEEE Trans. Instrum. Meas."},{"issue":"6","key":"10.1016\/j.cosrev.2026.101003_bib0013","doi-asserted-by":"crossref","first-page":"180","DOI":"10.1007\/s10462-025-11177-y","article-title":"MambaYOLACT: you only look at mamba prediction head for head-neck lymph nodes","volume":"58","author":"Zhou","year":"2025","journal-title":"Artif. Intell. Rev."},{"key":"10.1016\/j.cosrev.2026.101003_bib0014","doi-asserted-by":"crossref","DOI":"10.1016\/j.asoc.2025.113410","article-title":"Model-Data Co-driven U-net segmentation network for multimodal lung tumor images","volume":"180","author":"Zhou","year":"2025","journal-title":"Appl. Soft. Comput."},{"key":"10.1016\/j.cosrev.2026.101003_bib0015","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8126","article-title":"Transformer tracking","author":"Chen","year":"2021"},{"key":"10.1016\/j.cosrev.2026.101003_bib0016","first-page":"14745","article-title":"Transgan: two pure transformers can make one strong gan, and that can scale up","volume":"34","author":"Jiang","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cosrev.2026.101003_bib0017","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111028","article-title":"TBConvL-Net: a hybrid deep learning architecture for robust medical image segmentation","volume":"158","author":"Iqbal","year":"2025","journal-title":"Pattern. Recognit."},{"key":"10.1016\/j.cosrev.2026.101003_bib0018","article-title":"ChangeViT: unleashing plain vision transformers for change detection in remote sensing images","author":"Zhu","year":"2025","journal-title":"Pattern. Recognit."},{"key":"10.1016\/j.cosrev.2026.101003_bib0019","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6836","article-title":"Vivit: a video vision transformer","author":"Arnab","year":"2021"},{"key":"10.1016\/j.cosrev.2026.101003_bib0020","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1884","article-title":"Towards long-form video understanding","author":"Wu","year":"2021"},{"key":"10.1016\/j.cosrev.2026.101003_bib0021","unstructured":"Ma X., Wang Y., Jia G., et al. Latte: latent diffusion transformer for video generation. arxiv preprint arxiv:2401.03048, 2024."},{"key":"10.1016\/j.cosrev.2026.101003_bib0022","doi-asserted-by":"crossref","first-page":"5427","DOI":"10.1109\/TIP.2022.3195321","article-title":"End-to-end temporal action detection with transformer","volume":"31","author":"Liu","year":"2022","journal-title":"IEEe Trans. Image Process."},{"key":"10.1016\/j.cosrev.2026.101003_bib0023","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"14668","article-title":"Masked feature prediction for self-supervised visual pre-training","author":"Wei","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0024","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"24330","article-title":"Player-centric multimodal prompt generation for large language model based identity-aware basketball video captioning","author":"Xi","year":"2025"},{"key":"10.1016\/j.cosrev.2026.101003_bib0025","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102248","article-title":"Vision transformer: to discover the \u201cfour secrets\u201d of image patches","volume":"105","author":"Zhou","year":"2024","journal-title":"Inf. Fusion."},{"key":"10.1016\/j.cosrev.2026.101003_bib0026","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3333","article-title":"Multiview transformers for video recognition","author":"Yan","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0027","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"2214","article-title":"Rethinking video vits: sparse video tubes for joint image and video learning","author":"Piergiovanni","year":"2023"},{"key":"10.1016\/j.cosrev.2026.101003_bib0028","series-title":"European Conference on Computer Vision","first-page":"549","article-title":"Cavit: contextual alignment vision transformer for video object re-identification","author":"Wu","year":"2022"},{"issue":"12","key":"10.1016\/j.cosrev.2026.101003_bib0029","doi-asserted-by":"crossref","first-page":"8776","DOI":"10.1109\/TII.2022.3151766","article-title":"Multidirection and multiscale pyramid in transformer for video-based pedestrian retrieval","volume":"18","author":"Zang","year":"2022","journal-title":"IEEE Trans. Ind. Inform."},{"key":"10.1016\/j.cosrev.2026.101003_bib0030","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"14040","article-title":"Fuseformer: fusing fine-grained information in transformers for video inpainting","author":"Liu","year":"2021"},{"issue":"3","key":"10.1016\/j.cosrev.2026.101003_bib0031","doi-asserted-by":"crossref","first-page":"2351","DOI":"10.1109\/LRA.2024.3354627","article-title":"Learning sequence descriptor based on spatio-temporal attention for visual place recognition","volume":"9","author":"Zhao","year":"2024","journal-title":"IEEe Robot. Autom. Lett."},{"key":"10.1016\/j.cosrev.2026.101003_bib0032","series-title":"European conference on computer vision","first-page":"237","article-title":"Videomamba: state space model for efficient video understanding","author":"Li","year":"2024"},{"key":"10.1016\/j.cosrev.2026.101003_bib0033","unstructured":"Assran M., Bardes A., Fan D., et al. V-jepa 2: self-supervised video models enable understanding, prediction and planning. arXiv preprint arXiv:2506.09985, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0034","doi-asserted-by":"crossref","DOI":"10.1016\/j.asoc.2026.114632","article-title":"Transformer optimization algorithm for selecting tokens based on genetic algorithm","volume":"190","author":"Zhou","year":"2026","journal-title":"Appl. Soft. Comput."},{"key":"10.1016\/j.cosrev.2026.101003_bib0035","series-title":"European Conference on Computer Vision","first-page":"69","article-title":"Efficient video transformers with spatial-temporal token selection","author":"Wang","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0036","series-title":"ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"4305","article-title":"Haltingvt: adaptive token halting transformer for efficient video recognition","author":"Wu","year":"2024"},{"key":"10.1016\/j.cosrev.2026.101003_bib0037","series-title":"Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval","first-page":"970","article-title":"Centerclip: token clustering for efficient text-video retrieval","author":"Zhao","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0038","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"10388","article-title":"Efficient video action detection with token dropout and context refinement","author":"Chen","year":"2023"},{"key":"10.1016\/j.cosrev.2026.101003_bib0039","doi-asserted-by":"crossref","first-page":"28127","DOI":"10.52202\/079017-0882","article-title":"Don't look twice: faster video transformers with run-length tokenization","volume":"37","author":"Choudhury","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cosrev.2026.101003_bib0040","doi-asserted-by":"crossref","first-page":"6833","DOI":"10.52202\/079017-0219","article-title":"One token to seg them all: language instructed reasoning segmentation in videos","volume":"37","author":"Bai","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cosrev.2026.101003_bib0041","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"16945","article-title":"Prune spatio-temporal tokens by semantic-aware temporal accumulation","author":"Ding","year":"2023"},{"key":"10.1016\/j.cosrev.2026.101003_bib0042","doi-asserted-by":"crossref","unstructured":"Ren S., Chen S., Li S., et al. TESTA: temporal-spatial token aggregation for long-form video-language understanding. arxiv preprint arxiv:2310.19060, 2023.","DOI":"10.18653\/v1\/2023.findings-emnlp.66"},{"key":"10.1016\/j.cosrev.2026.101003_bib0043","unstructured":"Shen L., Hao T., He T., et al. Tempme: video temporal token merging for efficient text-video retrieval. arxiv preprint arxiv:2409.01156, 2024."},{"issue":"4","key":"10.1016\/j.cosrev.2026.101003_bib0044","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3633781","article-title":"Efficient video transformers via spatial-temporal token merging for action recognition","volume":"20","author":"Feng","year":"2024","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.cosrev.2026.101003_bib0045","unstructured":"Zhang Y., Zhao Z., Chen Z., et al. Beyond training: dynamic token merging for zero-shot video understanding. arxiv preprint arxiv:2411.14401, 2024."},{"key":"10.1016\/j.cosrev.2026.101003_bib0046","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7486","article-title":"Vidtome: video token merging for zero-shot video editing","author":"Li","year":"2024"},{"key":"10.1016\/j.cosrev.2026.101003_bib0047","unstructured":"Feng Y., Shi Y., Liu F., et al. Motion guided token compression for efficient masked video modeling. arxiv preprint arxiv:2402.18577, 2024."},{"issue":"4","key":"10.1016\/j.cosrev.2026.101003_bib0048","doi-asserted-by":"crossref","first-page":"3030","DOI":"10.1109\/TCSVT.2023.3307700","article-title":"Tranphys: spatiotemporal masked transformer steered remote photoplethysmography estimation","volume":"34","author":"Shao","year":"2023","journal-title":"IEEe Trans. Circuits. Syst. Video Technol."},{"key":"10.1016\/j.cosrev.2026.101003_bib0049","unstructured":"Zhang Y., Lu Y., Wang T., et al. Flexselect: flexible token selection for efficient long video understanding. arXiv preprint arXiv:2506.00993, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0050","article-title":"H2OT: hierarchical hourglass tokenizer for efficient video pose transformers","author":"Li","year":"2025","journal-title":"IEEe Trans. Pattern. Anal. Mach. Intell."},{"key":"10.1016\/j.cosrev.2026.101003_bib0051","article-title":"SELongVLM: empowering long video language models with self-corrective clip selection","author":"Zhang","year":"2026","journal-title":"IEEe Trans. Pattern. Anal. Mach. Intell."},{"key":"10.1016\/j.cosrev.2026.101003_bib0052","series-title":"European Conference on Computer Vision","first-page":"549","article-title":"Cavit: contextual alignment vision transformer for video object re-identification","author":"Wu","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0053","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109905","article-title":"Relative-position embedding based spatially and temporally decoupled Transformer for action recognition","volume":"145","author":"Ma","year":"2024","journal-title":"Pattern. Recognit."},{"key":"10.1016\/j.cosrev.2026.101003_bib0054","unstructured":"Li K., Wang Y., Gao P., et al. Uniformer: unified transformer for efficient spatiotemporal representation learning. arxiv preprint arxiv:2201.04676, 2022."},{"key":"10.1016\/j.cosrev.2026.101003_bib0055","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8741","article-title":"End-to-end video instance segmentation with transformers","author":"Wang","year":"2021"},{"key":"10.1016\/j.cosrev.2026.101003_bib0056","unstructured":"Wei X., Liu X., Zang Y., et al. VideoRoPE: what makes for good video rotary position embedding? arxiv preprint arxiv:2502.05173, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0057","article-title":"Dynamic scale position embedding for cross-modal representation learning","author":"Shin","year":"2025","journal-title":"Neural Netw."},{"key":"10.1016\/j.cosrev.2026.101003_bib0058","series-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing","first-page":"14471","article-title":"Vrope: rotary position embedding for video large language models","author":"Liu","year":"2025"},{"issue":"3","key":"10.1016\/j.cosrev.2026.101003_bib0059","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"2","author":"Bertasius","year":"2021","journal-title":"ICML"},{"key":"10.1016\/j.cosrev.2026.101003_bib0060","first-page":"19594","article-title":"Space-time mixing attention for video transformer","volume":"34","author":"Bulat","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cosrev.2026.101003_bib0061","first-page":"12493","article-title":"Keep your eye on the ball: trajectory attention in video transformers","volume":"34","author":"Patrick","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cosrev.2026.101003_bib0062","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3202","article-title":"Video swin transformer","author":"Liu","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0063","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"20030","article-title":"Direcformer: a directed attention in transformer approach to robust action recognition","author":"Truong","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0064","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3192","article-title":"Stand-alone inter-frame attention in video models","author":"Long","year":"2022"},{"key":"10.1016\/j.cosrev.2026.101003_bib0065","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"10265","article-title":"Cross-modal learning with 3D deformable attention for action recognition","author":"Kim","year":"2023"},{"key":"10.1016\/j.cosrev.2026.101003_bib0066","series-title":"ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"1","article-title":"Logo-former: local-global spatio-temporal transformer for dynamic facial expression recognition","author":"Ma","year":"2023"},{"key":"10.1016\/j.cosrev.2026.101003_bib0067","series-title":"2023 IEEE International Conference on Multimedia and Expo (ICME)","first-page":"1409","article-title":"Trajectory alignment based multi-scaled temporal attention for efficient video transformer","author":"Zhang","year":"2023"},{"key":"10.1016\/j.cosrev.2026.101003_bib0068","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"10351","article-title":"How much temporal long-term context is needed for action segmentation?","author":"Bahrami","year":"2023"},{"key":"10.1016\/j.cosrev.2026.101003_bib0069","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"7036","article-title":"Triplet attention transformer for spatiotemporal predictive learning","author":"Nie","year":"2024"},{"key":"10.1016\/j.cosrev.2026.101003_bib0070","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.127256","article-title":"k-NN attention-based video vision transformer for action recognition","volume":"574","author":"Sun","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.cosrev.2026.101003_bib0071","unstructured":"Gao Y., Huang J., Sun X., et al. Matten: video generation with mamba-attention. arxiv preprint arxiv:2405.03025, 2024."},{"key":"10.1016\/j.cosrev.2026.101003_bib0072","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18847","article-title":"CSTA: cNN-based spatiotemporal attention for video summarization","author":"Son","year":"2024"},{"key":"10.1016\/j.cosrev.2026.101003_bib0073","doi-asserted-by":"crossref","first-page":"7512","DOI":"10.1109\/TCSVT.2024.3372944","article-title":"Aggregating global and local representations via hybrid transformer for video deraining","volume":"34","author":"Mao","year":"2024","journal-title":"IEEe Trans. Circuits. Syst. Video Technol."},{"key":"10.1016\/j.cosrev.2026.101003_bib0074","unstructured":"Lan L., Jiang L., Yu T., et al. FullTransNet: full transformer with local-global attention for video summarization. arxiv preprint arxiv:2501.00882, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0075","unstructured":"Ding H., Li D., Su R., et al. Efficient-vDiT: efficient video diffusion transformers with attention tile. arxiv preprint arxiv:2502.06155, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0076","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111509","article-title":"Mask-aware 3D axial transformer for video inpainting","volume":"164","author":"Sun","year":"2025","journal-title":"Pattern. Recognit."},{"key":"10.1016\/j.cosrev.2026.101003_bib0077","unstructured":"Xi H., Yang S., Zhao Y., et al. Sparse VideoGen: accelerating video diffusion transformers with spatial-temporal sparsity. arxiv preprint arxiv:2502.01776, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0078","unstructured":"Wu J., Hou L., Yang H., et al. Vmoba: mixture-of-block attention for video diffusion models. arXiv preprint arXiv:2506.23858, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0079","unstructured":"Sun W., Tu R.C., Ding Y., et al. Vorta: efficient video diffusion via routing sparse attention. arXiv preprint arXiv:2505.18809, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0080","unstructured":"Liang C., Chen H., Hou L., et al. VMonarch: efficient video diffusion transformers with structured attention. arXiv preprint arXiv:2601.22275, 2026."},{"key":"10.1016\/j.cosrev.2026.101003_bib0081","unstructured":"Agarwal K., Chen Z., Luo C., et al. MonarchRT: efficient attention for real-time video generation. arXiv preprint arXiv:2602.12271, 2026."},{"key":"10.1016\/j.cosrev.2026.101003_bib0082","unstructured":"Feng W., Yang C., Qin H., et al. QuantSparse: comprehensively compressing video diffusion transformer with model quantization and attention sparsification. arXiv preprint arXiv:2509.23681, 2025."},{"key":"10.1016\/j.cosrev.2026.101003_bib0083","article-title":"DA-SWTS: dual-attention and temporal sampling make long video understanding efficient","author":"Sun","year":"2025","journal-title":"Inf. Sci."},{"key":"10.1016\/j.cosrev.2026.101003_bib0084","unstructured":"Fang T., Zhang H., Xie R., et al. SALAD: achieve high-sparsity attention via efficient linear attention tuning for video diffusion transformer. arXiv preprint arXiv:2601.16515, 2026."}],"container-title":["Computer Science Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1574013726001115?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1574013726001115?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T17:56:17Z","timestamp":1783360577000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1574013726001115"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":84,"alternative-id":["S1574013726001115"],"URL":"https:\/\/doi.org\/10.1016\/j.cosrev.2026.101003","relation":{},"ISSN":["1574-0137"],"issn-type":[{"value":"1574-0137","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Video vision transformer: to discover the \"four secrets\" of video patches","name":"articletitle","label":"Article Title"},{"value":"Computer Science Review","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cosrev.2026.101003","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101003"}}