{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T13:02:39Z","timestamp":1784552559689,"version":"3.55.0"},"reference-count":58,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T00:00:00Z","timestamp":1774828800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"DOI":"10.13039\/100031060","name":"European High Performance Computing Joint Undertaking","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100031060","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.cviu.2026.104749","type":"journal-article","created":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T08:03:48Z","timestamp":1775030628000},"page":"104749","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["ProfVLM: A lightweight video-language model for multi-view proficiency estimation"],"prefix":"10.1016","volume":"268","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0963-9543","authenticated-orcid":false,"given":"Edoardo","family":"Bianchi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1260-4640","authenticated-orcid":false,"given":"Jacopo","family":"Staiano","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2773-4421","authenticated-orcid":false,"given":"Antonio","family":"Liotta","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104749_b1","series-title":"SmolLM2: When smol goes big \u2013 data-centric training of a small language model","author":"Allal","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b2","series-title":"Is space-time attention all you need for video understanding?","author":"Bertasius","year":"2021"},{"key":"10.1016\/j.cviu.2026.104749_b3","series-title":"Latest Advancements in Mechanical Engineering","first-page":"257","article-title":"Egocentric video-based human action recognition in industrial environments","author":"Bianchi","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b4","doi-asserted-by":"crossref","unstructured":"Bianchi,\u00a0E., Lanz,\u00a0O., 2025. Gate-Shift-Pose: Enhancing Action Recognition in Sports with Skeleton Information. In: Proceedings of the Winter Conference on Applications of Computer Vision (WACV) Workshops. pp. 1257\u20131264.","DOI":"10.1109\/WACVW65960.2025.00139"},{"key":"10.1016\/j.cviu.2026.104749_b5","series-title":"2025 IEEE International Workshop on Sport, Technology and Research","first-page":"1","article-title":"PATS: Proficiency-aware temporal sampling for multi-view sports skill assessment","author":"Bianchi","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b6","series-title":"SkillFormer: Unified multi-view video understanding for proficiency estimation","author":"Bianchi","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b7","series-title":"egoPPG: Heart rate estimation from eye-tracking cameras in egocentric systems to benefit downstream vision tasks","author":"Braun","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b8","series-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"4724","article-title":"Quo vadis, action recognition? A new model and the kinetics dataset","author":"Carreira","year":"2017"},{"key":"10.1016\/j.cviu.2026.104749_b9","series-title":"Vicuna: an open-source chatbot impressing GPT-4 with 90%* ChatGPT quality","author":"Chiang","year":"2023"},{"key":"10.1016\/j.cviu.2026.104749_b10","series-title":"2015 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"1110","article-title":"Hierarchical recurrent neural network for skeleton based action recognition","author":"Du","year":"2015"},{"key":"10.1016\/j.cviu.2026.104749_b11","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"2959","article-title":"Revisiting skeleton-based action recognition","author":"Duan","year":"2022"},{"key":"10.1016\/j.cviu.2026.104749_b12","series-title":"2024 IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"8496","article-title":"Tracking skiers from the top to the bottom","author":"Dunnhofer","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b13","series-title":"Project aria: A new tool for egocentric multi-modal AI research","author":"Engel","year":"2023"},{"key":"10.1016\/j.cviu.2026.104749_b14","series-title":"Machine Learning in Sports: Open Approach for Next Play Analytics","first-page":"21","article-title":"Computer vision for sports analytics","author":"Fujii","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b15","series-title":"Computer Vision","first-page":"296","article-title":"The (Computer) vision of sports: Recent trends in research and commercial systems for sport analytics","author":"Gade","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b16","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15180","article-title":"ImageBind one embedding space to bind them all","author":"Girdhar","year":"2023"},{"key":"10.1016\/j.cviu.2026.104749_b17","doi-asserted-by":"crossref","unstructured":"Grauman,\u00a0K., Westbury,\u00a0A., Torresani,\u00a0L., Kitani,\u00a0K., Malik,\u00a0J., Afouras,\u00a0T., Ashutosh,\u00a0K., Baiyya,\u00a0V., Bansal,\u00a0S., Boote,\u00a0B., Byrne,\u00a0E., Chavis,\u00a0Z., Chen,\u00a0J., Cheng,\u00a0F., Chu,\u00a0F.J., Crane,\u00a0S., Dasgupta,\u00a0A., Dong,\u00a0J., Escobar,\u00a0M., Forigua,\u00a0C., Gebreselasie,\u00a0A., Haresh,\u00a0S., Huang,\u00a0J., Islam,\u00a0M.M., Jain,\u00a0S., Khirodkar,\u00a0R., Kukreja,\u00a0D., Liang,\u00a0K.J., Liu,\u00a0J.W., Majumder,\u00a0S., Mao,\u00a0Y., Martin,\u00a0M., Mavroudi,\u00a0E., Nagarajan,\u00a0T., Ragusa,\u00a0F., Ramakrishnan,\u00a0S.K., Seminara,\u00a0L., Somayazulu,\u00a0A., Song,\u00a0Y., Su,\u00a0S., Xue,\u00a0Z., Zhang,\u00a0E., Zhang,\u00a0J., Castillo,\u00a0A., Chen,\u00a0C., Fu,\u00a0X., Furuta,\u00a0R., Gonzalez,\u00a0C., Gupta,\u00a0P., Hu,\u00a0J., Huang,\u00a0Y., Huang,\u00a0Y., Khoo,\u00a0W., Kumar,\u00a0A., Kuo,\u00a0R., Lakhavani,\u00a0S., Liu,\u00a0M., Luo,\u00a0M., Luo,\u00a0Z., Meredith,\u00a0B., Miller,\u00a0A., Oguntola,\u00a0O., Pan,\u00a0X., Peng,\u00a0P., Pramanick,\u00a0S., Ramazanova,\u00a0M., Ryan,\u00a0F., Shan,\u00a0W., Somasundaram,\u00a0K., Song,\u00a0C., Southerland,\u00a0A., Tateno,\u00a0M., Wang,\u00a0H., Wang,\u00a0Y., Yagi,\u00a0T., Yan,\u00a0M., Yang,\u00a0X., Yu,\u00a0Z., Zha,\u00a0S.C., Zhao,\u00a0C., Zhao,\u00a0Z., Zhu,\u00a0Z., Zhuo,\u00a0J., Arbelaez,\u00a0P., Bertasius,\u00a0G., Damen,\u00a0D., Engel,\u00a0J., Farinella,\u00a0G.M., Furnari,\u00a0A., Ghanem,\u00a0B., Hoffman,\u00a0J., Jawahar,\u00a0C., Newcombe,\u00a0R., Park,\u00a0H.S., Rehg,\u00a0J.M., Sato,\u00a0Y., Savva,\u00a0M., Shi,\u00a0J., Shou,\u00a0M.Z., Wray,\u00a0M., 2024. Ego-Exo4D: Understanding Skilled Human Activity from First- and Third-Person Perspectives. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 19383\u201319400.","DOI":"10.1109\/CVPR52733.2024.01834"},{"key":"10.1016\/j.cviu.2026.104749_b18","series-title":"Image Analysis","first-page":"295","article-title":"Towards an AI-powered video assistant referee system (VARS) for association football","author":"Held","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b19","series-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2021"},{"key":"10.1016\/j.cviu.2026.104749_b20","doi-asserted-by":"crossref","unstructured":"Huang,\u00a0Y., Chen,\u00a0G., Xu,\u00a0J., Zhang,\u00a0M., Yang,\u00a0L., Pei,\u00a0B., Zhang,\u00a0H., Lu,\u00a0D., Wang,\u00a0Y., Wang,\u00a0L., Qiao,\u00a0Y., 2024. EgoExoLearn: A Dataset for Bridging Asynchronous Ego- and Exo-centric View of Procedural Activities in Real World. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52733.2024.02084"},{"key":"10.1016\/j.cviu.2026.104749_b21","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.104275","article-title":"SSL-Rehab: Assessment of physical rehabilitation exercises through self-supervised learning of 3D skeleton representations","volume":"251","author":"Kourbane","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104749_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.104228","article-title":"Action assessment in rehabilitation: Leveraging machine learning and vision-based analysis","volume":"251","author":"Kryeem","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104749_b23","series-title":"Proceedings of the Second Workshop on Statistical Machine Translation","first-page":"228","article-title":"Meteor: an automatic metric for MT evaluation with high levels of correlation with human judgments","author":"Lavie","year":"2007"},{"key":"10.1016\/j.cviu.2026.104749_b24","first-page":"1","article-title":"H2OT: Hierarchical hourglass tokenizer for efficient video pose transformers","author":"Li","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104749_b25","series-title":"Computer Vision \u2013 ECCV 2024: 18th European Conference, Milan, Italy, September 29\u2013October 4, 2024, Proceedings, Part XLVI","first-page":"323","article-title":"LLaMA-VID: An image is worth 2 tokens in large language models","author":"Li","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b26","series-title":"Text Summarization Branches Out","first-page":"74","article-title":"ROUGE: A package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.cviu.2026.104749_b27","series-title":"Video-LLaVA: Learning united visual representation by alignment before projection","author":"Lin","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b28","series-title":"TacticExpert: Spatial-temporal graph language model for basketball tactics","author":"Lingrui","year":"2025"},{"issue":"2","key":"10.1016\/j.cviu.2026.104749_b29","doi-asserted-by":"crossref","first-page":"708","DOI":"10.1109\/TPAMI.2024.3479776","article-title":"VALOR: Vision-audio-language omni-perception pretraining model and dataset","volume":"47","author":"Liu","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104749_b30","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2024.104186","article-title":"Bidirectional temporal and frame-segment attention for sparse action segmentation of figure skating","volume":"249","author":"Liu","year":"2024","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104749_b31","series-title":"NeurIPS","article-title":"Visual instruction tuning","author":"Liu","year":"2023"},{"key":"10.1016\/j.cviu.2026.104749_b32","series-title":"Video swin transformer","author":"Liu","year":"2021"},{"key":"10.1016\/j.cviu.2026.104749_b33","series-title":"RoBERTa: A robustly optimized BERT pretraining approach","author":"Liu","year":"2019"},{"key":"10.1016\/j.cviu.2026.104749_b34","series-title":"Video-ChatGPT: Towards detailed video understanding via large vision and language models","author":"Maaz","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b35","series-title":"SmolVLM: Redefining small and efficient multimodal models","author":"Marafioti","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b36","series-title":"2024 IEEE International Workshop on Sport, Technology and Research","first-page":"120","article-title":"Ski pose estimation","author":"Martinelli","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b37","series-title":"Design and Architecture for Signal and Image Processing","first-page":"81","article-title":"KD-AHOSVD: Neural network compression via knowledge distillation and tensor decomposition","author":"Meneghetti","year":"2025"},{"key":"10.1016\/j.cviu.2026.104749_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104410","article-title":"Spatio-temporal graph neural network based child action recognition using data-efficient methods: A systematic analysis","volume":"259","author":"Mohottala","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104749_b39","doi-asserted-by":"crossref","unstructured":"Pan,\u00a0Y., Zhang,\u00a0C., Bertasius,\u00a0G., 2025. BASKET: A Large-Scale Video Dataset for Fine-Grained Skill Estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. CVPR, pp. 28952\u201328962.","DOI":"10.1109\/CVPR52734.2025.02696"},{"key":"10.1016\/j.cviu.2026.104749_b40","series-title":"2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"304","article-title":"What and how well you performed? A multitask learning approach to action quality assessment","author":"Parmar","year":"2019"},{"key":"10.1016\/j.cviu.2026.104749_b41","doi-asserted-by":"crossref","DOI":"10.1016\/j.is.2020.101562","article-title":"Sports analytics \u2014 Evaluation of basketball players and team performance","volume":"93","author":"Sarlis","year":"2020","journal-title":"Inf. Syst."},{"issue":"28","key":"10.1016\/j.cviu.2026.104749_b42","doi-asserted-by":"crossref","first-page":"33475","DOI":"10.1007\/s11042-024-20576-2","article-title":"Application of human activity\/action recognition: a review","volume":"84","author":"sedaghati","year":"2025","journal-title":"Multimedia Tools Appl."},{"key":"10.1016\/j.cviu.2026.104749_b43","series-title":"PandaGPT: One model to instruction-follow them all","author":"Su","year":"2023"},{"key":"10.1016\/j.cviu.2026.104749_b44","series-title":"2019 IEEE\/CVF International Conference on Computer Vision","first-page":"7463","article-title":"VideoBERT: A Joint Model for Video and Language Representation Learning","author":"Sun","year":"2019"},{"key":"10.1016\/j.cviu.2026.104749_b45","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104456","article-title":"UniMultNet: Action recognition method based on multi-scale feature fusion and video-text constraint guidance","volume":"260","author":"Tian","year":"2025","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.cviu.2026.104749_b46","series-title":"Learning spatiotemporal features with 3D convolutional networks","author":"Tran","year":"2015"},{"key":"10.1016\/j.cviu.2026.104749_b47","series-title":"VideoAgent: Long-form video understanding with large language model as agent","author":"Wang","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b48","series-title":"Skating-mixer: Long-term sport audio-visual modeling with MLPs","author":"Xia","year":"2022"},{"key":"10.1016\/j.cviu.2026.104749_b49","series-title":"2023 IEEE\/CVF International Conference on Computer Vision","first-page":"11941","article-title":"Sigmoid loss for language image pre-training","author":"Zhai","year":"2023"},{"key":"10.1016\/j.cviu.2026.104749_b50","series-title":"2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18430","article-title":"Narrative action evaluation with prompt-guided multimodal interaction","author":"Zhang","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b51","series-title":"BERTScore: Evaluating text generation with BERT","author":"Zhang","year":"2020"},{"key":"10.1016\/j.cviu.2026.104749_b52","series-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing: System Demonstrations","first-page":"543","article-title":"Video-LLaMA: An instruction-tuned audio-visual language model for video understanding","author":"Zhang","year":"2023"},{"issue":"1","key":"10.1016\/j.cviu.2026.104749_b53","doi-asserted-by":"crossref","first-page":"29173","DOI":"10.1038\/s41598-025-14985-y","article-title":"Learning spatio-temporal context for basketball action pose estimation with a multi-stream network","volume":"15","author":"Zhang","year":"2025","journal-title":"Sci. Rep."},{"issue":"23","key":"10.1016\/j.cviu.2026.104749_b54","doi-asserted-by":"crossref","DOI":"10.3390\/electronics13234733","article-title":"A review of state-of-the-art methodologies and applications in action recognition","volume":"13","author":"Zhao","year":"2024","journal-title":"Electronics"},{"key":"10.1016\/j.cviu.2026.104749_b55","series-title":"A comprehensive survey of action quality assessment: Method and benchmark","author":"Zhou","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b56","series-title":"MiniGPT-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"},{"key":"10.1016\/j.cviu.2026.104749_b57","series-title":"Video-STaR: Self-training enables video instruction tuning with any supervision","author":"Zohar","year":"2024"},{"key":"10.1016\/j.cviu.2026.104749_b58","series-title":"Apollo: An exploration of video understanding in large multimodal models","author":"Zohar","year":"2024"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001165?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001165?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T22:32:33Z","timestamp":1778797953000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001165"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":58,"alternative-id":["S1077314226001165"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104749","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"ProfVLM: A lightweight video-language model for multi-view proficiency estimation","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104749","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier Inc.","name":"copyright","label":"Copyright"}],"article-number":"104749"}}