{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T11:55:30Z","timestamp":1777290930667,"version":"3.51.4"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Image and Vision Computing"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.imavis.2026.105976","type":"journal-article","created":{"date-parts":[[2026,4,6]],"date-time":"2026-04-06T01:10:59Z","timestamp":1775437859000},"page":"105976","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Action-aware anchor-based frame selection strategy for action recognition"],"prefix":"10.1016","volume":"170","author":[{"given":"Zhiming","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxin","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0400-9366","authenticated-orcid":false,"given":"Fan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ge","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6090-0110","authenticated-orcid":false,"given":"Zhuo","family":"Su","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.imavis.2026.105976_b1","doi-asserted-by":"crossref","unstructured":"J. Yue-Hei Ng, M. Hausknecht, S. Vijayanarasimhan, O. Vinyals, R. Monga, G. Toderici, Beyond short snippets: Deep networks for video classification, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2015, pp. 4694\u20134702.","DOI":"10.1109\/CVPR.2015.7299101"},{"issue":"3","key":"10.1016\/j.imavis.2026.105976_b2","doi-asserted-by":"crossref","first-page":"2259","DOI":"10.1007\/s10462-020-09904-8","article-title":"A survey on video-based human action recognition: recent updates, datasets, challenges, and applications","volume":"54","author":"Pareek","year":"2021","journal-title":"Artif. Intell. Rev."},{"issue":"5","key":"10.1016\/j.imavis.2026.105976_b3","doi-asserted-by":"crossref","first-page":"1366","DOI":"10.1007\/s11263-022-01594-9","article-title":"Human action recognition and prediction: A survey","volume":"130","author":"Kong","year":"2022","journal-title":"Int. J. Comput. Vis."},{"issue":"9","key":"10.1016\/j.imavis.2026.105976_b4","first-page":"2761","article-title":"Comparative analysis of keyframe extraction techniques for video summarization","volume":"14","author":"Parikh","year":"2021","journal-title":"Recent. Patents Comput. Sci."},{"key":"10.1016\/j.imavis.2026.105976_b5","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2019.102678","article-title":"Quality-guided key frames selection from video stream based on object detection","volume":"65","author":"Chen","year":"2019","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.imavis.2026.105976_b6","doi-asserted-by":"crossref","unstructured":"Z. Wu, C. Xiong, C.-Y. Ma, R. Socher, L.S. Davis, Adaframe: Adaptive frame selection for fast video recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 1278\u20131287.","DOI":"10.1109\/CVPR.2019.00137"},{"key":"10.1016\/j.imavis.2026.105976_b7","doi-asserted-by":"crossref","first-page":"76","DOI":"10.1016\/j.imavis.2016.06.002","article-title":"Effective and efficient human action recognition using dynamic frame skipping and trajectory rejection","volume":"58","author":"Seo","year":"2017","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.105976_b8","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2023.103959","article-title":"Action recognition method based on lightweight network and rough-fine keyframe extraction","volume":"97","author":"Pan","year":"2023","journal-title":"J. Vis. Commun. Image Represent."},{"issue":"7","key":"10.1016\/j.imavis.2026.105976_b9","doi-asserted-by":"crossref","first-page":"6197","DOI":"10.1109\/TMC.2025.3541580","article-title":"Robust motion-guided frame sampler with interpretive evaluation for video action recognition","volume":"24","author":"Bai","year":"2025","journal-title":"IEEE Trans. Mob. Comput."},{"key":"10.1016\/j.imavis.2026.105976_b10","series-title":"2022 World Automation Congress","first-page":"494","article-title":"Research on video keyframe extraction method based on action analysis","author":"Wang","year":"2022"},{"issue":"1","key":"10.1016\/j.imavis.2026.105976_b11","doi-asserted-by":"crossref","first-page":"475","DOI":"10.1007\/s13042-024-02235-y","article-title":"Action recognition method based on a novel keyframe extraction method and enhanced 3D convolutional neural network","volume":"16","author":"Tian","year":"2025","journal-title":"Int. J. Mach. Learn. Cybern."},{"issue":"1","key":"10.1016\/j.imavis.2026.105976_b12","doi-asserted-by":"crossref","first-page":"114","DOI":"10.1016\/j.jvcir.2011.08.005","article-title":"Key frame extraction based on visual attention model","volume":"23","author":"Lai","year":"2012","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.imavis.2026.105976_b13","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2022.103740","article-title":"Action density based frame sampling for human action recognition in videos","volume":"90","author":"Lin","year":"2023","journal-title":"J. Vis. Commun. Image Represent."},{"issue":"9","key":"10.1016\/j.imavis.2026.105976_b14","doi-asserted-by":"crossref","first-page":"11958","DOI":"10.1007\/s11227-024-05893-5","article-title":"Multi-stream network with key frame sampling for human action recognition","volume":"80","author":"Xia","year":"2024","journal-title":"J. Supercomput."},{"key":"10.1016\/j.imavis.2026.105976_b15","doi-asserted-by":"crossref","unstructured":"D.-A. Huang, V. Ramanathan, D. Mahajan, L. Torresani, M. Paluri, L. Fei-Fei, J.C. Niebles, What makes a video a video: Analyzing temporal information in video understanding models and datasets, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 7366\u20137375.","DOI":"10.1109\/CVPR.2018.00769"},{"key":"10.1016\/j.imavis.2026.105976_b16","series-title":"UCF101: A dataset of 101 human actions classes from videos in the wild","author":"Soomro","year":"2012"},{"key":"10.1016\/j.imavis.2026.105976_b17","series-title":"2011 International Conference on Computer Vision","first-page":"2556","article-title":"HMDB: a large video database for human motion recognition","author":"Kuehne","year":"2011"},{"key":"10.1016\/j.imavis.2026.105976_b18","article-title":"Two-stream convolutional networks for action recognition in videos","volume":"27","author":"Simonyan","year":"2014","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.imavis.2026.105976_b19","doi-asserted-by":"crossref","unstructured":"D. Tran, L. Bourdev, R. Fergus, L. Torresani, M. Paluri, Learning spatiotemporal features with 3d convolutional networks, in: Proceedings of the IEEE International Conference on Computer Vision, 2015, pp. 4489\u20134497.","DOI":"10.1109\/ICCV.2015.510"},{"key":"10.1016\/j.imavis.2026.105976_b20","doi-asserted-by":"crossref","unstructured":"A. Arnab, M. Dehghani, G. Heigold, C. Sun, M. Lu\u010di\u0107, C. Schmid, Vivit: A video vision transformer, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 6836\u20136846.","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"10.1016\/j.imavis.2026.105976_b21","series-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"10.1016\/j.imavis.2026.105976_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2021.104329","article-title":"EAR: Efficient action recognition with local-global temporal aggregation","volume":"116","author":"Zhang","year":"2021","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.105976_b23","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2023.104740","article-title":"Fsformer: Fast-slow transformer for video action recognition","volume":"137","author":"Li","year":"2023","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.imavis.2026.105976_b24","series-title":"27th European Conference on Artificial Intelligence","first-page":"113","article-title":"An animation-based augmentation approach for action recognition from discontinuous video","author":"Song","year":"2024"},{"key":"10.1016\/j.imavis.2026.105976_b25","doi-asserted-by":"crossref","unstructured":"J. Hwang, J.-H. Kim, J.-H. Choi, J.-S. Lee, Just one moment: Structural vulnerability of deep action recognition against one frame attack, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 7668\u20137676.","DOI":"10.1109\/ICCV48922.2021.00757"},{"key":"10.1016\/j.imavis.2026.105976_b26","first-page":"674","article-title":"An iterative image registration technique with an application to stereo vision","volume":"vol. 2","author":"Lucas","year":"1981"},{"issue":"1\u20133","key":"10.1016\/j.imavis.2026.105976_b27","doi-asserted-by":"crossref","first-page":"185","DOI":"10.1016\/0004-3702(81)90024-2","article-title":"Determining optical flow","volume":"17","author":"Horn","year":"1981","journal-title":"Artificial Intelligence"},{"key":"10.1016\/j.imavis.2026.105976_b28","series-title":"Scandinavian Conference on Image Analysis","first-page":"363","article-title":"Two-frame motion estimation based on polynomial expansion","author":"Farneb\u00e4ck","year":"2003"},{"key":"10.1016\/j.imavis.2026.105976_b29","series-title":"Pattern Recognition: 29th DAGM Symposium, Heidelberg, Germany, September 12-14, 2007. Proceedings 29","first-page":"214","article-title":"A duality based approach for realtime tv-l 1 optical flow","author":"Zach","year":"2007"},{"key":"10.1016\/j.imavis.2026.105976_b30","doi-asserted-by":"crossref","unstructured":"A. Dosovitskiy, P. Fischer, E. Ilg, P. Hausser, C. Hazirbas, V. Golkov, P. Van Der Smagt, D. Cremers, T. Brox, Flownet: Learning optical flow with convolutional networks, in: Proceedings of the IEEE International Conference on Computer Vision, 2015, pp. 2758\u20132766.","DOI":"10.1109\/ICCV.2015.316"},{"key":"10.1016\/j.imavis.2026.105976_b31","doi-asserted-by":"crossref","unstructured":"E. Ilg, N. Mayer, T. Saikia, M. Keuper, A. Dosovitskiy, T. Brox, Flownet 2.0: Evolution of optical flow estimation with deep networks, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 2462\u20132470.","DOI":"10.1109\/CVPR.2017.179"},{"key":"10.1016\/j.imavis.2026.105976_b32","doi-asserted-by":"crossref","unstructured":"D. Sun, X. Yang, M.-Y. Liu, J. Kautz, Pwc-net: Cnns for optical flow using pyramid, warping, and cost volume, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 8934\u20138943.","DOI":"10.1109\/CVPR.2018.00931"},{"key":"10.1016\/j.imavis.2026.105976_b33","series-title":"European Conference on Computer Vision","first-page":"402","article-title":"Raft: Recurrent all-pairs field transforms for optical flow","author":"Teed","year":"2020"},{"key":"10.1016\/j.imavis.2026.105976_b34","doi-asserted-by":"crossref","unstructured":"L. Hu, R. Zhao, Z. Ding, L. Ma, B. Shi, R. Xiong, T. Huang, Optical flow estimation for spiking camera, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 17844\u201317853.","DOI":"10.1109\/CVPR52688.2022.01732"},{"key":"10.1016\/j.imavis.2026.105976_b35","first-page":"48070","article-title":"Unsupervised optical flow estimation with dynamic timing representation for spike camera","volume":"36","author":"Xia","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.imavis.2026.105976_b36","first-page":"7905","article-title":"Learning optical flow from continuous spike streams","volume":"35","author":"Zhao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.imavis.2026.105976_b37","first-page":"7496","article-title":"Optical flow for spike camera with hierarchical spatial-temporal spike fusion","volume":"vol. 38","author":"Zhao","year":"2024"},{"key":"10.1016\/j.imavis.2026.105976_b38","first-page":"525","article-title":"Spatio-temporal recurrent networks for event-based optical flow estimation","volume":"vol. 36","author":"Ding","year":"2022"},{"key":"10.1016\/j.imavis.2026.105976_b39","doi-asserted-by":"crossref","unstructured":"H. Wang, C. Schmid, Action recognition with improved trajectories, in: Proceedings of the IEEE International Conference on Computer Vision, 2013, pp. 3551\u20133558.","DOI":"10.1109\/ICCV.2013.441"},{"key":"10.1016\/j.imavis.2026.105976_b40","series-title":"2011 International Conference on Computer Vision","first-page":"2564","article-title":"ORB: An efficient alternative to SIFT or SURF","author":"Rublee","year":"2011"},{"issue":"6","key":"10.1016\/j.imavis.2026.105976_b41","doi-asserted-by":"crossref","first-page":"381","DOI":"10.1145\/358669.358692","article-title":"Random sample consensus: A paradigm for model fitting with applications to image analysis and automated cartography","volume":"24","author":"Fischler","year":"1981","journal-title":"Commun. ACM"},{"key":"10.1016\/j.imavis.2026.105976_b42","doi-asserted-by":"crossref","unstructured":"D. Tran, H. Wang, L. Torresani, J. Ray, Y. LeCun, M. Paluri, A closer look at spatiotemporal convolutions for action recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 6450\u20136459.","DOI":"10.1109\/CVPR.2018.00675"},{"key":"10.1016\/j.imavis.2026.105976_b43","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"10.1016\/j.imavis.2026.105976_b44","doi-asserted-by":"crossref","unstructured":"J. Redmon, S. Divvala, R. Girshick, A. Farhadi, You only look once: Unified, real-time object detection, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 779\u2013788.","DOI":"10.1109\/CVPR.2016.91"},{"key":"10.1016\/j.imavis.2026.105976_b45","doi-asserted-by":"crossref","unstructured":"H. Zhao, J. Liu, W. Wang, Research on Human Behavior Recognition Based on Video Key Frame, in: The 2nd International Conference on Computing and Data Science, 2021, pp. 1\u20135.","DOI":"10.1145\/3448734.3450778"},{"key":"10.1016\/j.imavis.2026.105976_b46","first-page":"1451","article-title":"Smart frame selection for action recognition","volume":"vol. 35","author":"Gowda","year":"2021"},{"key":"10.1016\/j.imavis.2026.105976_b47","doi-asserted-by":"crossref","unstructured":"Y. Zhi, Z. Tong, L. Wang, G. Wu, Mgsampler: An explainable sampling strategy for video action recognition, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 1513\u20131522.","DOI":"10.1109\/ICCV48922.2021.00154"},{"key":"10.1016\/j.imavis.2026.105976_b48","first-page":"8247","article-title":"Attention-aware sampling via deep reinforcement learning for action recognition","volume":"vol. 33","author":"Dong","year":"2019"}],"container-title":["Image and Vision Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626000831?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0262885626000831?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T11:21:05Z","timestamp":1777288865000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0262885626000831"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":48,"alternative-id":["S0262885626000831"],"URL":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105976","relation":{},"ISSN":["0262-8856"],"issn-type":[{"value":"0262-8856","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Action-aware anchor-based frame selection strategy for action recognition","name":"articletitle","label":"Article Title"},{"value":"Image and Vision Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.imavis.2026.105976","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"105976"}}