{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T13:01:54Z","timestamp":1784638914457,"version":"3.55.0"},"reference-count":42,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100013061","name":"Jilin Scientific and Technological Development Program","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013061","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.cviu.2026.104879","type":"journal-article","created":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T07:02:41Z","timestamp":1784530961000},"page":"104879","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Co-designing architecture and feature guidance for efficient video understanding"],"prefix":"10.1016","volume":"271","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4904-6818","authenticated-orcid":false,"given":"Shilin","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6466-8876","authenticated-orcid":false,"given":"Xingwang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaohui","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1979-4420","authenticated-orcid":false,"given":"Kun","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104879_b1","series-title":"ICML 2021","first-page":"813","article-title":"Is space-time attention all you need for video understanding?","author":"Bertasius","year":"2021"},{"key":"10.1016\/j.cviu.2026.104879_b2","series-title":"CVPR 2017","first-page":"4724","article-title":"Quo vadis, action recognition? A new model and the kinetics dataset","author":"Carreira","year":"2017"},{"key":"10.1016\/j.cviu.2026.104879_b3","article-title":"STAN: Spatio-temporal analysis network for efficient video action recognition","author":"Chen","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.cviu.2026.104879_b4","series-title":"2020 25th International Conference on Pattern Recognition","first-page":"4183","article-title":"RWF-2000: an open large scale video database for violence detection","author":"Cheng","year":"2021"},{"key":"10.1016\/j.cviu.2026.104879_b5","doi-asserted-by":"crossref","first-page":"701","DOI":"10.1109\/TIP.2024.3522809","article-title":"Deformable convolution-enhanced hierarchical transformer with spectral-spatial cluster attention for hyperspectral image classification","volume":"34","author":"Fang","year":"2025","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104879_b6","series-title":"ICCV 2017","first-page":"5843","article-title":"The \u201csomething something\u201d video database for learning and evaluating visual common sense","author":"Goyal","year":"2017"},{"key":"10.1016\/j.cviu.2026.104879_b7","series-title":"CVPR 2021","first-page":"13713","article-title":"Coordinate attention for efficient mobile network design","author":"Hou","year":"2021"},{"key":"10.1016\/j.cviu.2026.104879_b8","series-title":"ECCV 2022, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXV","first-page":"259","article-title":"TDAM: Top-down attention module for contextually guided feature selection in CNNs","volume":"vol. 13685","author":"Jaiswal","year":"2022"},{"key":"10.1016\/j.cviu.2026.104879_b9","doi-asserted-by":"crossref","DOI":"10.1016\/j.neunet.2024.106321","article-title":"Ensuring spatial scalability with temporal-wise spatial attentive pooling for temporal action detection","volume":"176","author":"Kim","year":"2024","journal-title":"Neural Netw."},{"key":"10.1016\/j.cviu.2026.104879_b10","series-title":"ECCV 2020, Glasgow, UK, August 23-28, 2020, Proceedings, Part XVI","first-page":"345","article-title":"MotionSqueeze: Neural motion feature learning for video understanding","volume":"vol. 12361","author":"Kwon","year":"2020"},{"issue":"9","key":"10.1016\/j.cviu.2026.104879_b11","doi-asserted-by":"crossref","first-page":"5174","DOI":"10.1109\/TCSVT.2023.3250646","article-title":"Spatio-temporal adaptive network with bidirectional temporal difference for action recognition","volume":"33","author":"Li","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104879_b12","series-title":"ICLR 2022","article-title":"UniFormer: Unified transformer for efficient spatial-temporal representation learning","author":"Li","year":"2022"},{"key":"10.1016\/j.cviu.2026.104879_b13","series-title":"ICCV 2023","first-page":"1632","article-title":"UniFormerV2: Unlocking the potential of image ViTs for video understanding","author":"Li","year":"2023"},{"key":"10.1016\/j.cviu.2026.104879_b14","series-title":"CVPR 2022","first-page":"4794","article-title":"MViTv2: Improved multiscale vision transformers for classification and detection","author":"Li","year":"2022"},{"key":"10.1016\/j.cviu.2026.104879_b15","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.104150","article-title":"M-adapter: Multi-level image-to-video adaptation for video action recognition","volume":"249","author":"Li","year":"2024","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104879_b16","series-title":"ECCV 2024, Milan, Italy, September 29-October 4, 2024, Proceedings, Part LXXXIII","first-page":"425","article-title":"ZeroI2V: Zero-cost adaptation of pre-trained transformers from image to video","volume":"vol. 15141","author":"Li","year":"2024"},{"issue":"5","key":"10.1016\/j.cviu.2026.104879_b17","first-page":"2760","article-title":"TSM: Temporal shift module for efficient and scalable video understanding on edge devices","volume":"44","author":"Lin","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104879_b18","series-title":"Computer Vision - ECCV 2022 - 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part XXXV","first-page":"388","article-title":"Frozen CLIP models are efficient video learners","volume":"vol. 13695","author":"Lin","year":"2022"},{"key":"10.1016\/j.cviu.2026.104879_b19","series-title":"CVPR 2022","first-page":"3192","article-title":"Video swin transformer","author":"Liu","year":"2022"},{"key":"10.1016\/j.cviu.2026.104879_b20","doi-asserted-by":"crossref","first-page":"4104","DOI":"10.1109\/TIP.2022.3180585","article-title":"Motion-driven visual tempo learning for video-based action recognition","volume":"31","author":"Liu","year":"2022","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104879_b21","series-title":"ICASSP 2023","first-page":"1","article-title":"Efficient multi-scale attention module with cross-spatial learning","author":"Ouyang","year":"2023"},{"key":"10.1016\/j.cviu.2026.104879_b22","series-title":"ST-adapter: Parameter-efficient image-to-video transfer learning for action recognition","author":"Pan","year":"2022"},{"key":"10.1016\/j.cviu.2026.104879_b23","doi-asserted-by":"crossref","first-page":"218","DOI":"10.1109\/TMM.2023.3263288","article-title":"MAR: Masked autoencoders for efficient action recognition","volume":"26","author":"Qing","year":"2024","journal-title":"IEEE Trans. Multim."},{"key":"10.1016\/j.cviu.2026.104879_b24","series-title":"UCF101: a dataset of 101 human actions classes from videos in the wild","author":"Soomro","year":"2012"},{"key":"10.1016\/j.cviu.2026.104879_b25","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104456","article-title":"UniMultNet: Action recognition method based on multi-scale feature fusion and video-text constraint guidance","volume":"260","author":"Tian","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104879_b26","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.104229","article-title":"Multi-scale adaptive skeleton transformer for action recognition","volume":"250","author":"Wang","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104879_b27","series-title":"CVPR 2021","first-page":"1895","article-title":"TDN: Temporal difference networks for efficient action recognition","author":"Wang","year":"2021"},{"key":"10.1016\/j.cviu.2026.104879_b28","series-title":"CVPR 2020","first-page":"11531","article-title":"ECA-net: Efficient channel attention for deep convolutional neural networks","author":"Wang","year":"2020"},{"issue":"D1","key":"10.1016\/j.cviu.2026.104879_b29","doi-asserted-by":"crossref","first-page":"622","DOI":"10.1093\/nar\/gkab1062","article-title":"HMDB 5.0: the human metabolome database for 2022","volume":"50","author":"Wishart","year":"2022","journal-title":"Nucleic Acids Res."},{"key":"10.1016\/j.cviu.2026.104879_b30","series-title":"ECCV 2018, Munich, Germany, September 8-14, 2018, Proceedings, Part VII","first-page":"3","article-title":"CBAM: Convolutional block attention module","volume":"vol. 11211","author":"Woo","year":"2018"},{"key":"10.1016\/j.cviu.2026.104879_b31","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104434","article-title":"SPKDB-net: A salient-part pose keypoints-based dual-branch network for repetitive action counting","volume":"259","author":"Wu","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"issue":"12","key":"10.1016\/j.cviu.2026.104879_b32","doi-asserted-by":"crossref","first-page":"13338","DOI":"10.1109\/TCSVT.2024.3445151","article-title":"Target-aware camera placement for large-scale video surveillance","volume":"34","author":"Wu","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"2","key":"10.1016\/j.cviu.2026.104879_b33","doi-asserted-by":"crossref","first-page":"995","DOI":"10.1109\/TCSVT.2023.3288878","article-title":"Online unsupervised video object segmentation via contrastive motion clustering","volume":"34","author":"Xi","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104879_b34","series-title":"ICASSP 2023","first-page":"1","article-title":"One-shot medical action recognition with a cross-attention mechanism and dynamic time warping","author":"Xie","year":"2023"},{"key":"10.1016\/j.cviu.2026.104879_b35","doi-asserted-by":"crossref","unstructured":"Xiong, Yuwen, Li, Zhiqi, Chen, Yuntao, Wang, Feng, Zhu, Xizhou, Luo, Jiapeng, Wang, Wenhai, Lu, Tong, Li, Hongsheng, Qiao, Yu, Lu, Lewei, Zhou, Jie, Dai, Jifeng, 2024. Efficient Deformable ConvNets: Rethinking Dynamic and Sparse Operator for Vision Applications. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 5652\u20135661.","DOI":"10.1109\/CVPR52733.2024.00540"},{"key":"10.1016\/j.cviu.2026.104879_b36","series-title":"CVPR 2022","first-page":"3323","article-title":"Multiview transformers for video recognition","author":"Yan","year":"2022"},{"key":"10.1016\/j.cviu.2026.104879_b37","unstructured":"Yang, Brandon, Bender, Gabriel, Le, Quoc V., Ngiam, Jiquan, 2019. CondConv: Conditionally Parameterized Convolutions for Efficient Inference. In: NeurIPS 2019. December 8-14, 2019, Vancouver, BC, Canada, pp. 1305\u20131316."},{"key":"10.1016\/j.cviu.2026.104879_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104379","article-title":"Anomaly-aware self-supervised feature learning for weakly supervised video anomaly detection","volume":"257","author":"Yang","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104879_b39","series-title":"ICLR 2023","article-title":"AIM: adapting image models for efficient video action recognition","author":"Yang","year":"2023"},{"key":"10.1016\/j.cviu.2026.104879_b40","doi-asserted-by":"crossref","first-page":"3242","DOI":"10.1109\/TIP.2024.3391692","article-title":"Multi-label action anticipation for real-world videos with scene understanding","volume":"33","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.cviu.2026.104879_b41","series-title":"ICASSP 2021","first-page":"2235","article-title":"SA-net: Shuffle attention for deep convolutional neural networks","author":"Zhang","year":"2021"},{"key":"10.1016\/j.cviu.2026.104879_b42","doi-asserted-by":"crossref","unstructured":"Zhao, Zhiyu, Huang, Bingkun, Xing, Sen, Wu, Gangshan, Qiao, Yu, Wang, Limin, 2024. Asymmetric Masked Distillation for Pre-Training Small Foundation Models. In: CVPR. pp. 18516\u201318526.","DOI":"10.1109\/CVPR52733.2024.01752"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226002468?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226002468?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T12:10:39Z","timestamp":1784635839000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226002468"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":42,"alternative-id":["S1077314226002468"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104879","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Co-designing architecture and feature guidance for efficient video understanding","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104879","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104879"}}