{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:17:19Z","timestamp":1783315039451,"version":"3.54.6"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T00:00:00Z","timestamp":1773100800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T00:00:00Z","timestamp":1773100800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62036012, 62322212, 62502518, 62532003, U25A20536, U23A20387"],"award-info":[{"award-number":["62036012, 62322212, 62502518, 62532003, U25A20536, U23A20387"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Postdoctoral Fellowship Program of CPSF","award":["GZC20251036"],"award-info":[{"award-number":["GZC20251036"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s00530-026-02282-1","type":"journal-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T14:04:19Z","timestamp":1773151459000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["EgoFusion: unified semantic and scale-aware prompt fusion for egocentric action recognition"],"prefix":"10.1007","volume":"32","author":[{"given":"Hechenrui","family":"Fan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huaihai","family":"Lyu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chaofan","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,10]]},"reference":[{"key":"2282_CR1","doi-asserted-by":"crossref","unstructured":"Kareer, S., Patel, D., Punamiya, R., Mathur, P., Cheng, S., Wang, C., Hoffman, J., Xu, D.: Egomimic: Scaling imitation learning via egocentric video. arXiv preprint arXiv:2410.24221 (2024)","DOI":"10.1109\/ICRA55743.2025.11127989"},{"key":"2282_CR2","doi-asserted-by":"crossref","unstructured":"Xu, S., Ling, H.Y., Wang, Y.-X., Gui, L.-Y.: Intermimic: Towards universal whole-body control for physics-based human-object interactions. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 12266\u201312277 (2025)","DOI":"10.1109\/CVPR52734.2025.01145"},{"key":"2282_CR3","doi-asserted-by":"crossref","unstructured":"Li, Y., Ye, Z., Rehg, J.M.: Delving into egocentric actions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 287\u2013295 (2015)","DOI":"10.1109\/CVPR.2015.7298625"},{"key":"2282_CR4","doi-asserted-by":"crossref","unstructured":"Kazakos, E., Nagrani, A., Zisserman, A., Damen, D.: Epic-fusion: Audio-visual temporal binding for egocentric action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5492\u20135501 (2019)","DOI":"10.1109\/ICCV.2019.00559"},{"key":"2282_CR5","doi-asserted-by":"crossref","unstructured":"Fathi, A., Farhadi, A., Rehg, J.M.: Understanding egocentric activities. In: 2011 International Conference on Computer Vision, pp. 407\u2013414 (2011). IEEE","DOI":"10.1109\/ICCV.2011.6126269"},{"key":"2282_CR6","doi-asserted-by":"crossref","unstructured":"Wang, H., Singh, M.K., Torresani, L.: Ego-only: Egocentric action detection without exocentric transferring. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5250\u20135261 (2023)","DOI":"10.1109\/ICCV51070.2023.00484"},{"key":"2282_CR7","doi-asserted-by":"crossref","unstructured":"Radevski, G., Grujicic, D., Blaschko, M., Moens, M.-F., Tuytelaars, T.: Multimodal distillation for egocentric action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5213\u20135224 (2023)","DOI":"10.1109\/ICCV51070.2023.00481"},{"key":"2282_CR8","doi-asserted-by":"crossref","unstructured":"Huang, C., Tian, Y., Kumar, A., Xu, C.: Egocentric audio-visual object localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22910\u201322921 (2023)","DOI":"10.1109\/CVPR52729.2023.02194"},{"key":"2282_CR9","doi-asserted-by":"crossref","unstructured":"Lyu, H., Chen, C., Ji, Y., Xu, C.: Egoprompt: Prompt learning for egocentric action recognition. In: Proceedings of the 33rd ACM International Conference on Multimedia, pp. 2762\u20132770 (2025)","DOI":"10.1145\/3746027.3754749"},{"key":"2282_CR10","doi-asserted-by":"publisher","first-page":"33174","DOI":"10.52202\/075280-1440","volume":"36","author":"D Chatterjee","year":"2023","unstructured":"Chatterjee, D., Sener, F., Ma, S., Yao, A.: Opening the vocabulary of egocentric actions. Adv. Neural. Inf. Process. Syst. 36, 33174\u201333187 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2282_CR11","doi-asserted-by":"crossref","unstructured":"Kukleva, A., Sener, F., Remelli, E., Tekin, B., Sauser, E., Schiele, B., Ma, S.: X-mic: Cross-modal instance conditioning for egocentric action generalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26364\u201326373 (2024)","DOI":"10.1109\/CVPR52733.2024.02490"},{"issue":"6","key":"2282_CR12","doi-asserted-by":"publisher","first-page":"6605","DOI":"10.1109\/TPAMI.2020.3015894","volume":"45","author":"X Wang","year":"2020","unstructured":"Wang, X., Zhu, L., Wu, Y., Yang, Y.: Symbiotic attention for egocentric action recognition with object-centric alignment. IEEE Trans. Pattern Anal. Mach. Intell. 45(6), 6605\u20136617 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2282_CR13","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhu, L., Wang, H., Yang, Y.: Interactive prototype learning for egocentric action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8168\u20138177 (2021)","DOI":"10.1109\/ICCV48922.2021.00806"},{"key":"2282_CR14","doi-asserted-by":"crossref","unstructured":"Pramanick, S., Song, Y., Nag, S., Lin, K.Q., Shah, H., Shou, M.Z., Chellappa, R., Zhang, P.: Egovlpv2: Egocentric video-language pre-training with fusion in the backbone. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5285\u20135297 (2023)","DOI":"10.1109\/ICCV51070.2023.00487"},{"key":"2282_CR15","doi-asserted-by":"crossref","unstructured":"Chen, C., Ashutosh, K., Girdhar, R., Harwath, D., Grauman, K.: Soundingactions: Learning how actions sound from narrated egocentric videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 27252\u201327262 (2024)","DOI":"10.1109\/CVPR52733.2024.02573"},{"key":"2282_CR16","doi-asserted-by":"crossref","unstructured":"Pasca, R.-G., Gavryushin, A., Hamza, M., Kuo, Y.-L., Mo, K., Van Gool, L., Hilliges, O., Wang, X.: Summarize the past to predict the future: Natural language descriptions of context boost multimodal object interaction. In: CVPR 2024 (2024)","DOI":"10.1109\/CVPR52733.2024.01731"},{"key":"2282_CR17","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2117\u20132125 (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"2282_CR18","doi-asserted-by":"crossref","unstructured":"Damen, D., Doughty, H., Farinella, G.M., Furnari, A., Kazakos, E., Ma, J., Moltisanti, D., Munro, J., Perrett, T., Price, W., et al.: Rescaling egocentric vision: Collection, pipeline and challenges for epic-kitchens-100. International Journal of Computer Vision, 1\u201323 (2022)","DOI":"10.1007\/s11263-021-01531-2"},{"key":"2282_CR19","unstructured":"Grauman, K., Westbury, A., Byrne, E., Chavis, Z., Furnari, A., Girdhar, R., Hamburger, J., Jiang, H., Liu, M., Liu, X., et al.: Ego4d: Around the world in 3,000 hours of egocentric video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18995\u201319012 (2022)"},{"key":"2282_CR20","doi-asserted-by":"crossref","unstructured":"Gong, X., Mohan, S., Dhingra, N., Bazin, J.-C., Li, Y., Wang, Z., Ranjan, R.: Mmg-ego4d: Multimodal generalization in egocentric action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6481\u20136491 (2023)","DOI":"10.1109\/CVPR52729.2023.00627"},{"key":"2282_CR21","doi-asserted-by":"publisher","first-page":"33485","DOI":"10.52202\/075280-1455","volume":"36","author":"S Tan","year":"2023","unstructured":"Tan, S., Nagarajan, T., Grauman, K.: Egodistill: egocentric head motion distillation for efficient video understanding. Adv. Neural. Inf. Process. Syst. 36, 33485\u201333498 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"9","key":"2282_CR22","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vision 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"2282_CR23","doi-asserted-by":"crossref","unstructured":"Khattak, M.U., Rasheed, H., Maaz, M., Khan, S., Khan, F.S.: Maple: Multi-modal prompt learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19113\u201319122 (2023)","DOI":"10.1109\/CVPR52729.2023.01832"},{"key":"2282_CR24","doi-asserted-by":"crossref","unstructured":"Jia, M., Tang, L., Chen, B.-C., Cardie, C., Belongie, S., Hariharan, B., Lim, S.-N.: Visual prompt tuning. In: European Conference on Computer Vision, pp. 709\u2013727 (2022). Springer","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"2282_CR25","doi-asserted-by":"crossref","unstructured":"Lyu, H., Yao, H., Xu, C.: Multiple local prompts distillation for domain generalization. IEEE Transactions on Multimedia (2025)","DOI":"10.1109\/TMM.2025.3607719"},{"key":"2282_CR26","doi-asserted-by":"crossref","unstructured":"Yao, H., Zhang, R., Xu, C.: Visual-language prompt tuning with knowledge-guided context optimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6757\u20136767 (2023)","DOI":"10.1109\/CVPR52729.2023.00653"},{"key":"2282_CR27","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021). PmLR"},{"key":"2282_CR28","doi-asserted-by":"crossref","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Conditional prompt learning for vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16816\u201316825 (2022)","DOI":"10.1109\/CVPR52688.2022.01631"},{"key":"2282_CR29","unstructured":"Roy, S., Etemad, A.: Consistency-guided prompt learning for vision-language models. arXiv preprint arXiv:2306.01195 (2023)"},{"key":"2282_CR30","doi-asserted-by":"crossref","unstructured":"Chen, H., Wu, Z., Han, X., Jia, M., Jiang, Y.-G.: Promptfusion: Decoupling stability and plasticity for continual learning. In: European Conference on Computer Vision, pp. 196\u2013212 (2024). Springer","DOI":"10.1007\/978-3-031-73021-4_12"},{"key":"2282_CR31","doi-asserted-by":"crossref","unstructured":"Xu, B., Zheng, S., Jin, Q.: Pov: Prompt-oriented view-agnostic learning for egocentric hand-object interaction in the multi-view world. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 2807\u20132816 (2023)","DOI":"10.1145\/3581783.3612484"},{"key":"2282_CR32","first-page":"18661","volume":"33","author":"P Khosla","year":"2020","unstructured":"Khosla, P., Teterwak, P., Wang, C., Sarna, A., Tian, Y., Isola, P., Maschinot, A., Liu, C., Krishnan, D.: Supervised contrastive learning. Adv. Neural. Inf. Process. Syst. 33, 18661\u201318673 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"6","key":"2282_CR33","doi-asserted-by":"publisher","first-page":"471","DOI":"10.1007\/s00530-025-02061-4","volume":"31","author":"X Li","year":"2025","unstructured":"Li, X., Zhang, W., Pang, M., Zhu, J.: Mdikd: multi-dimensional integration knowledge distillation. Multimedia Syst. 31(6), 471 (2025)","journal-title":"Multimedia Syst."},{"issue":"6","key":"2282_CR34","doi-asserted-by":"publisher","first-page":"472","DOI":"10.1007\/s00530-025-02065-0","volume":"31","author":"T Liu","year":"2025","unstructured":"Liu, T., An, G., Yang, Z., Ren, X., Ruan, Q.: Eira: an explicit-implicit representation alignment for multimodal relation extraction. Multimedia Syst. 31(6), 472 (2025)","journal-title":"Multimedia Syst."},{"issue":"2","key":"2282_CR35","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","volume":"132","author":"P Gao","year":"2024","unstructured":"Gao, P., Geng, S., Zhang, R., Ma, T., Fang, R., Zhang, Y., Li, H., Qiao, Y.: Clip-adapter: better vision-language models with feature adapters. Int. J. Comput. Vision 132(2), 581\u2013595 (2024)","journal-title":"Int. J. Comput. Vision"},{"key":"2282_CR36","doi-asserted-by":"crossref","unstructured":"Ju, C., Han, T., Zheng, K., Zhang, Y., Xie, W.: Prompting visual-language models for efficient video understanding. In: European Conference on Computer Vision, pp. 105\u2013124 (2022). Springer","DOI":"10.1007\/978-3-031-19833-5_7"},{"key":"2282_CR37","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Misra, I., Kr\u00e4henb\u00fchl, P., Girdhar, R.: Learning video representations from large language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6586\u20136597 (2023)","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"2282_CR38","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"2282_CR39","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F.L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et al.: Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02282-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02282-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02282-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T04:59:57Z","timestamp":1783313997000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02282-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,10]]},"references-count":39,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2282"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02282-1","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-8283592\/v1","asserted-by":"object"}]},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,10]]},"assertion":[{"value":"5 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"217"}}