{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:22:00Z","timestamp":1778048520873,"version":"3.51.4"},"reference-count":72,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00432","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"4438-4450","source":"Crossref","is-referenced-by-count":0,"title":["Ego-EXTRA: video-language Egocentric Dataset for EXpert-TRAinee assistance"],"prefix":"10.1109","author":[{"given":"Francesco","family":"Ragusa","sequence":"first","affiliation":[{"name":"University of Catania,Department of Mathematics and Computer Science,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Michele","family":"Mazzamuto","sequence":"additional","affiliation":[{"name":"University of Catania,Department of Mathematics and Computer Science,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rosario","family":"Forte","sequence":"additional","affiliation":[{"name":"University of Catania,Department of Mathematics and Computer Science,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Irene","family":"D\u2019Ambra","sequence":"additional","affiliation":[{"name":"University of Catania,Department of Mathematics and Computer Science,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"James","family":"Fort","sequence":"additional","affiliation":[{"name":"Meta Reality Labs Research,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jakob","family":"Engel","sequence":"additional","affiliation":[{"name":"Meta Reality Labs Research,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Antonino","family":"Furnari","sequence":"additional","affiliation":[{"name":"University of Catania,Department of Mathematics and Computer Science,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giovanni Maria","family":"Farinella","sequence":"additional","affiliation":[{"name":"University of Catania,Department of Mathematics and Computer Science,Italy"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Aria mps"},{"key":"ref2","volume-title":"ehow"},{"key":"ref3","volume-title":"Llama 3.1 instruct"},{"key":"ref4","volume-title":"Llama 3.3 instruct turbo"},{"key":"ref5","volume-title":"wikihow"},{"key":"ref6","volume-title":"youtube"},{"key":"ref7","article-title":"Video-mined task graphs for keystep recognition in instructional videos","volume":"36","author":"Ashutosh","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref8","author":"Ataallah","year":"2024","journal-title":"Minigpt4-video: Advancing multimodal llms for video understanding with interleaved visual-textual tokens"},{"key":"ref9","article-title":"Qwen2.5-vl technical report","author":"Bai","year":"2025"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681618"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-025-02676-0"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-025-02676-0"},{"key":"ref13","article-title":"VidEgoThink: assessing egocentric video understanding capabilities for embodied AI","author":"Cheng","year":"2024"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01355"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2142"},{"key":"ref16","first-page":"720","article-title":"Scaling egocentric vision: The epic-kitchens dataset","volume-title":"Proceedings of the European conference on computer vision (ECCV)","author":"Damen"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01531-2"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01853"},{"key":"ref19","author":"DeepSeek-AI","year":"2025","journal-title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning"},{"key":"ref20","article-title":"Mamba fusion: Learning actions through questioning","author":"Dong","year":"2024"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00634"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00805"},{"key":"ref23","author":"Dubey","year":"2024","journal-title":"The llama 3 herd of models"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00536"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01749"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00170"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01325"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01834"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-long.579"},{"key":"ref31","author":"Hasegawa","year":"2024","journal-title":"Promqa: Question answering dataset for multimodal procedural activity understanding"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72655-2_25"},{"key":"ref33","first-page":"0","article-title":"Epictent: An egocentric video dataset for camping tent assembly","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops","author":"Jang"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58574-7_46"},{"key":"ref35","author":"Jia","year":"2022","journal-title":"Egotaskqa: Understanding human tasks in egocentric videos"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2012.2200554"},{"issue":"3","key":"ref37","first-page":"119","article-title":"Wizard of oz (woz): a yellow brick journey","volume":"13","author":"F. ( \u201cJeff\u201d) Kelley","year":"2018","journal-title":"J. Usability Studies"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01263"},{"key":"ref39","author":"Li","year":"2024","journal-title":"Llava-onevision: Easy visual task transfer"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2004"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00778"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01758"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1163"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.3025105"},{"key":"ref47","author":"Peddi","year":"2024","journal-title":"CaptainCook4D: A Dataset for Understanding Errors in Procedural Activities"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02095-7"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103764"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/wacv57701.2024.00449"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00431"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1895"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02042"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58517-4_10"},{"key":"ref55","article-title":"Project aria: A new tool for egocentric multi-modal ai research","author":"Somasundaram","year":"2023"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1688"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.833"},{"key":"ref58","article-title":"Chameleon: Mixed-modal early-fusion foundation models","author":"Team","year":"2024"},{"key":"ref59","author":"Team","year":"2024","journal-title":"Qwen2.5: A party of foundation models"},{"key":"ref60","volume-title":"Tobii pro fusion bar","author":"Tobii"},{"key":"ref61","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref62","author":"Touvron","year":"2023","journal-title":"Llama 2: Open foundation and fine-tuned chat models"},{"key":"ref63","article-title":"Attention is all you need","author":"Vaswani","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref64","article-title":"Qwen2-VL: enhancing vision-language models perception of the world at any resolution","author":"Wang","year":"2024"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01854"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_28"},{"key":"ref67","article-title":"MM-Ego: towards building egocentric multimodal LLMs","author":"Ye","year":"2024"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_29"},{"key":"ref69","author":"Zhang","year":"2024","journal-title":"Video instruction tuning with synthetic data"},{"key":"ref70","article-title":"Antgpt: Can large language models help long-term action anticipation from videos?","author":"Zhao","year":"2023"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01033"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00319"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492599.pdf?arnumber=11492599","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:05:33Z","timestamp":1778047533000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492599\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":72,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00432","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}