{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T01:39:47Z","timestamp":1773193187023,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62036012, U23A20387, 62322212"],"award-info":[{"award-number":["62036012, U23A20387, 62322212"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Pengcheng Laboratory Research Project","award":["PCL2023A08"],"award-info":[{"award-number":["PCL2023A08"]}]},{"name":"Postdoctoral Fellowship Program of CPSF","award":["GZC20251036"],"award-info":[{"award-number":["GZC20251036"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754749","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:51Z","timestamp":1761377211000},"page":"2762-2770","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["EgoPrompt: Prompt Learning for Egocentric Action Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-1866-9429","authenticated-orcid":false,"given":"Huaihai","family":"Lyu","sequence":"first","affiliation":[{"name":"MAIS, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7970-7698","authenticated-orcid":false,"given":"Chaofan","family":"Chen","sequence":"additional","affiliation":[{"name":"MAIS, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4898-6918","authenticated-orcid":false,"given":"Yuheng","family":"Ji","sequence":"additional","affiliation":[{"name":"MAIS, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8343-9665","authenticated-orcid":false,"given":"Changsheng","family":"Xu","sequence":"additional","affiliation":[{"name":"MAIS, Institute of Automation, Chinese Academy of Sciences, Beijing, China and Peng Cheng Laboratory, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models. arXiv preprint arXiv:2407.07895","author":"Li Feng","year":"2024","unstructured":"Feng Li, Renrui Zhang, Hao Zhang, Yuanhan Zhang, Bo Li, Wei Li, Zejun Ma, and Chunyuan Li. 2024. Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models. arXiv preprint arXiv:2407.07895 (2024)."},{"key":"e_1_3_2_1_2_1","volume-title":"Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326","author":"Li Bo","year":"2024","unstructured":"Bo Li, Yuanhan Zhang, Dong Guo, Renrui Zhang, Feng Li, Hao Zhang, Kaichen Zhang, Yanwei Li, Ziwei Liu, and Chunyuan Li. 2024. Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:2408.03326 (2024)."},{"key":"e_1_3_2_1_3_1","first-page":"33485","article-title":"Egodistill: Egocentric head motion distillation for efficient video understanding","volume":"36","author":"Tan Shuhan","year":"2023","unstructured":"Shuhan Tan, Tushar Nagarajan, and Kristen Grauman. 2023. Egodistill: Egocentric head motion distillation for efficient video understanding. Advances in Neural Information Processing Systems 36 (2023), 33485--33498.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00641"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01278"},{"key":"e_1_3_2_1_6_1","volume-title":"Opening the vocabulary of egocentric actions. Advances in Neural Information Processing Systems 36","author":"Chatterjee Dibyadip","year":"2024","unstructured":"Dibyadip Chatterjee, Fadime Sener, Shugao Ma, and Angela Yao. 2024. Opening the vocabulary of egocentric actions. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612484"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_38"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_1_11_1","volume-title":"Antonino Furnari, Evangelos Kazakos, Jian Ma, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, et al.","author":"Damen Dima","year":"2022","unstructured":"Dima Damen, Hazel Doughty, Giovanni Maria Farinella, Antonino Furnari, Evangelos Kazakos, Jian Ma, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, et al. 2022. Rescaling egocentric vision: Collection, pipeline and challenges for epic-kitchens-100. International Journal of Computer Vision (2022), 1--23."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02084"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i15.33729"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02212"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Muhammad Uzair Khattak Hanoona Rasheed Muhammad Maaz Salman Khan and Fahad Shahbaz Khan. 2023. MaPLe: Multi-modal Prompt Learning. arXiv:2210.03117 [cs.CV]","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"e_1_3_2_1_17_1","unstructured":"Zangwei Zheng Xiangyu Yue Kai Wang and Yang You. 2022. Prompt Vision Transformer for Domain Generalization. arXiv:2208.08914 [cs.CV]"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01891-x"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653--1"},{"key":"e_1_3_2_1_20_1","volume-title":"Chen Change Loy, and Ziwei Liu","author":"Zhou Kaiyang","year":"2022","unstructured":"Kaiyang Zhou, Jingkang Yang, Chen Change Loy, and Ziwei Liu. 2022. Conditional Prompt Learning for Vision-Language Models. arXiv:2203.05557 [cs.CV]"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2025.3557780"},{"key":"e_1_3_2_1_22_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3239197"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01394"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00653"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00024"},{"key":"e_1_3_2_1_27_1","volume-title":"Hardik Shah, Mike Zheng Shou, Rama Chellappa, and Pengchuan Zhang.","author":"Pramanick Shraman","year":"2023","unstructured":"Shraman Pramanick, Yale Song, Sayan Nag, Kevin Qinghong Lin, Hardik Shah, Mike Zheng Shou, Rama Chellappa, and Pengchuan Zhang. 2023. EgoVLPv2: Egocentric Video-Language Pre-training with Fusion in the Backbone. arXiv:2307.05463 [cs.CV] https:\/\/arxiv.org\/abs\/2307.05463"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Sijie Cheng Zhicheng Guo Jingwen Wu Kechen Fang Peng Li Huaping Liu and Yang Liu. 2024. EgoThink: Evaluating First-Person Perspective Thinking Capability of Vision-Language Models. arXiv:2311.15596 [cs.CV] https:\/\/arxiv.org\/abs\/2311.15596","DOI":"10.1109\/CVPR52733.2024.01355"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02490"},{"key":"e_1_3_2_1_30_1","volume-title":"Mattia Soldan, Michael Wray, Rui Yan, Eric Zhongcong Xu, Difei Gao, Rongcheng Tu, Wenzhe Zhao, Weijie Kong, Chengfei Cai, Hongfa Wang, Dima Damen","author":"Lin Kevin Qinghong","year":"2022","unstructured":"Kevin Qinghong Lin, Alex Jinpeng Wang, Mattia Soldan, Michael Wray, Rui Yan, Eric Zhongcong Xu, Difei Gao, Rongcheng Tu, Wenzhe Zhao, Weijie Kong, Chengfei Cai, Hongfa Wang, Dima Damen, Bernard Ghanem, Wei Liu, and Mike Zheng Shou. 2022. Egocentric Video-Language Pretraining. arXiv:2206.01670 [cs.CV] https:\/\/arxiv.org\/abs\/2206.01670"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00168"},{"key":"e_1_3_2_1_32_1","unstructured":"BAAI RoboBrain Team Mingyu Cao Huajie Tan Yuheng Ji Minglan Lin Zhiyu Li Zhou Cao Pengwei Wang Enshen Zhou Yi Han Yingbo Tang Xiangqi Xu Wei Guo Yaoxu Lyu Yijie Xu Jiayu Shi Mengfei Du Cheng Chi Mengdi Zhao Xiaoshuai Hao Junkai Zhao Xiaojie Zhang Shanyu Rong Huaihai Lyu Zhengliang Cai Yankai Fu Ning Chen Bolun Zhang Lingfeng Zhang Shuyi Zhang Dong Liu Xi Feng Songjing Wang Xiaodan Liu Yance Jiao Mengsi Lyu Zhuo Chen Chenrui He Yulong Ao Xue Sun Zheqi He Jingshu Zheng Xi Yang Donghai Shi Kunchang Xie Bochao Zhang Shaokai Nie Chunlei Men Yonghua Lin Zhongyuan Wang Tiejun Huang and Shanghang Zhang. 2025. RoboBrain 2.0 Technical Report. arXiv:2507.02029 [cs.RO] https:\/\/arxiv.org\/abs\/2507.02029"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00484"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02206"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00653"},{"key":"e_1_3_2_1_38_1","volume-title":"Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754749","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:06:06Z","timestamp":1765339566000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754749"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":38,"alternative-id":["10.1145\/3746027.3754749","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754749","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}