{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,30]],"date-time":"2026-05-30T09:03:26Z","timestamp":1780131806230,"version":"3.54.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,7]],"date-time":"2026-06-07T00:00:00Z","timestamp":1780790400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"DARPA PTG program","award":["PTG program"],"award-info":[{"award-number":["PTG program"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,8]]},"DOI":"10.1145\/3811427.3811456","type":"proceedings-article","created":{"date-parts":[[2026,5,30]],"date-time":"2026-05-30T08:29:11Z","timestamp":1780129751000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["InsightAR: A Tool for Multi-modal Summarization and Interactive Analysis of AR-based Egocentric Task Videos"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9244-173X","authenticated-orcid":false,"given":"Guande","family":"Wu","sequence":"first","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0200-721X","authenticated-orcid":false,"given":"Dishita","family":"Turakhia","sequence":"additional","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6455-917X","authenticated-orcid":false,"given":"Eden","family":"Wu","sequence":"additional","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6881-3006","authenticated-orcid":false,"given":"Sonia Castelo","family":"Quispe","sequence":"additional","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3341-7059","authenticated-orcid":false,"given":"Jo\u00e3o","family":"Rulff","sequence":"additional","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7565-3052","authenticated-orcid":false,"given":"Erin","family":"McGowan","sequence":"additional","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3471-6130","authenticated-orcid":false,"given":"Jianben","family":"He","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9480-1145","authenticated-orcid":false,"given":"Yawei","family":"Wang","sequence":"additional","affiliation":[{"name":"AWS AI, Santa Clara, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5517-3035","authenticated-orcid":false,"given":"Jing","family":"Qian","sequence":"additional","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2452-2295","authenticated-orcid":false,"given":"Claudio","family":"Silva","sequence":"additional","affiliation":[{"name":"New York University, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,7]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Gilles\u00a0E Gignac and Eva\u00a0T Szodorai. 2016. Effect size guidelines for individual differences researchers. Personality and individual differences 102 (2016) 74\u201378.","DOI":"10.1016\/j.paid.2016.06.069"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01428"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","unstructured":"Anubhav Jangra Sourajit Mukherjee Adam Jatowt Sriparna Saha and Mohammad Hasanuzzaman. 2023. A Survey on Multi-modal Summarization. ACM Comput. Surv. 55 13s (2023) 296:1\u2013296:36. 10.1145\/3584700","DOI":"10.1145\/3584700"},{"key":"e_1_3_3_1_5_2","unstructured":"Will Kay Jo\u00e3o Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev Mustafa Suleyman and Andrew Zisserman. 2017. The Kinetics Human Action Video Dataset. CoRR abs\/1705.06950 (2017). arXiv:https:\/\/arXiv.org\/abs\/1705.06950http:\/\/arxiv.org\/abs\/1705.06950"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"crossref","unstructured":"Rachael Lear Sophia Ellis Tiffany Ollivierre-Harris Susannah Long and Erik\u00a0K Mayer. 2023. Video Recording Patients for Direct Care Purposes: Systematic Review and Narrative Synthesis of International Empirical Studies and UK Professional Guidance. Journal of Medical Internet Research 25 (2023) e46478.","DOI":"10.2196\/46478"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01765"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","unstructured":"Rosario Leonardi Francesco Ragusa Antonino Furnari and Giovanni\u00a0Maria Farinella. 2023. Exploiting Multimodal Synthetic Data for Egocentric Human-Object Interaction Detection in an Industrial Scenario. CoRR abs\/2306.12152 (2023). arXiv:https:\/\/arXiv.org\/abs\/2306.1215210.48550\/ARXIV.2306.12152","DOI":"10.48550\/ARXIV.2306.12152"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706598.3714188"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/D17-1114"},{"key":"e_1_3_3_1_11_2","series-title":"Proceedings of Machine Learning Research","first-page":"12888","volume-title":"International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA","volume":"162","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven C.\u00a0H. Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In International Conference on Machine Learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA(Proceedings of Machine Learning Research, Vol.\u00a0162), Kamalika Chaudhuri, Stefanie Jegelka, Le\u00a0Song, Csaba Szepesv\u00e1ri, Gang Niu, and Sivan Sabato (Eds.). PMLR, 12888\u201312900. https:\/\/proceedings.mlr.press\/v162\/li22n.html"},{"key":"e_1_3_3_1_12_2","first-page":"9760","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics, COLING 2025, Abu Dhabi, UAE, January 19-24, 2025","author":"Li Xinzhe","year":"2025","unstructured":"Xinzhe Li. 2025. A Review of Prominent Paradigms for LLM-Based Agents: Tool Use, Planning (Including RAG), and Feedback Learning. In Proceedings of the 31st International Conference on Computational Linguistics, COLING 2025, Abu Dhabi, UAE, January 19-24, 2025, Owen Rambow, Leo Wanner, Marianna Apidianaki, Hend Al-Khalifa, Barbara\u00a0Di Eugenio, and Steven Schockaert (Eds.). Association for Computational Linguistics, 9760\u20139779. https:\/\/aclanthology.org\/2025.coling-main.652\/"},{"key":"e_1_3_3_1_13_2","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022","author":"Lin Kevin\u00a0Qinghong","year":"2022","unstructured":"Kevin\u00a0Qinghong Lin, Jinpeng Wang, Mattia Soldan, Michael Wray, Rui Yan, Eric\u00a0Zhongcong Xu, Difei Gao, Rong-Cheng Tu, Wenzhe Zhao, Weijie Kong, Chengfei Cai, Hongfa Wang, Dima Damen, Bernard Ghanem, Wei Liu, and Mike\u00a0Zheng Shou. 2022. Egocentric Video-Language Pretraining. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022, Sanmi Koyejo, S.\u00a0Mohamed, A.\u00a0Agarwal, Danielle Belgrave, K.\u00a0Cho, and A.\u00a0Oh (Eds.)."},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.318"},{"key":"e_1_3_3_1_15_2","unstructured":"MD Mair Chris Elsey Paul\u00a0V Smith and Patrick\u00a0G Watson. 2018. War on video: Combat footage vernacular video analysis and military culture from within. Ethnographic Studies15 (2018) 83\u2013105."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Sebeom Park Shokhrukh Bokijonov and Yosoon Choi. 2021. Review of microsoft hololens applications over the past five years. Applied sciences 11 16 (2021) 7259.","DOI":"10.3390\/app11167259"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","unstructured":"Ludan Ruan and Qin Jin. 2022. Survey: Transformer based video-language pre-training. AI Open 3 (2022) 1\u201313. 10.1016\/J.AIOPEN.2022.01.001","DOI":"10.1016\/J.AIOPEN.2022.01.001"},{"key":"e_1_3_3_1_18_2","unstructured":"Carl\u00a0Ren\u00e9 Sauer and Peter Burggr\u00e4f. 2024. Hybrid intelligence\u2013systematic approach and framework to determine the level of Human-AI collaboration for production management use cases. Production Engineering (2024) 1\u201317."},{"key":"e_1_3_3_1_19_2","unstructured":"Tim\u00a0J. Schoonbeek Tim Houben Hans Onvlee Peter H.\u00a0N. de With and Fons van\u00a0der Sommen. 2023. IndustReal: A Dataset for Procedure Step Recognition Handling Execution Errors in Egocentric Videos in an Industrial-Like Setting. arxiv:https:\/\/arXiv.org\/abs\/2310.17323\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2310.17323"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00431"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","unstructured":"Xiang Suo Weidi Tang and Zhen Li. 2024. Motion Capture Technology in Sports Scenarios: A Survey. Sensors 24 9 (2024) 2947. 10.3390\/S24092947","DOI":"10.3390\/S24092947"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Dishita Turakhia Mark Parent Tovi Grossman Michael Glueck and Ben Lafreniere. 2025. Investigating Augmented Reality for Adaptive Motor-Skill Training. (2025).","DOI":"10.1145\/3769872.3769891"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1145\/3334480.3381069"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00417"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72989-8_4"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","unstructured":"Muhammad Wasim Imran Ahmed Jamil Ahmad Muhammad Nawaz Eatedal Alabdulkreem and Yazeed Ghadi. 2022. A video summarization framework based on activity attention modeling using deep features for smart campus surveillance system. PeerJ Comput. Sci. 8 (2022) e911. 10.7717\/PEERJ-CS.911","DOI":"10.7717\/PEERJ-CS.911"},{"key":"e_1_3_3_1_27_2","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed\u00a0H. Chi, Quoc\u00a0V. Le, and Denny Zhou. 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022, Sanmi Koyejo, S.\u00a0Mohamed, A.\u00a0Agarwal, Danielle Belgrave, K.\u00a0Cho, and A.\u00a0Oh (Eds.). http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/9d5609613524ecf4f15af0f7b31abca4-Abstract-Conference.html"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01025"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.5244\/C.35.415"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","unstructured":"Keunwoo\u00a0Peter Yu Zheyuan Zhang Fengyuan Hu and Joyce Chai. 2023. Efficient In-Context Learning in Vision-Language Models for Egocentric Videos. CoRR abs\/2311.17041 (2023). arXiv:https:\/\/arXiv.org\/abs\/2311.1704110.48550\/ARXIV.2311.17041","DOI":"10.48550\/ARXIV.2311.17041"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.11238"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3675094.3678992"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","unstructured":"Jieqiong Zhao Morteza Karimzadeh Luke\u00a0S. Snyder Chittayong Surakitbanharn Zhenyu\u00a0Cheryl Qian and David\u00a0S. Ebert. 2020. MetricsVis: A Visual Analytics System for Evaluating Employee Performance in Public Safety Agencies. IEEE Trans. Vis. Comput. Graph. 26 1 (2020) 1193\u20131203. 10.1109\/TVCG.2019.2934603","DOI":"10.1109\/TVCG.2019.2934603"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3654777.3676366"}],"event":{"name":"AVI '26: International Conference on Advanced Visual Interfaces","location":"Venice Italy","acronym":"AVI '26"},"container-title":["Proceedings of the 2026 International Conference on Advanced Visual Interfaces"],"original-title":[],"deposited":{"date-parts":[[2026,5,30]],"date-time":"2026-05-30T08:29:58Z","timestamp":1780129798000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3811427.3811456"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,7]]},"references-count":33,"alternative-id":["10.1145\/3811427.3811456","10.1145\/3811427"],"URL":"https:\/\/doi.org\/10.1145\/3811427.3811456","relation":{},"subject":[],"published":{"date-parts":[[2026,6,7]]},"assertion":[{"value":"2026-06-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}