{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:51:17Z","timestamp":1765309877484,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","funder":[{"name":"the National Key Research and Development Program of China","award":["2022ZD0160402"],"award-info":[{"award-number":["2022ZD0160402"]}]},{"name":"the National Natural Science Foundation of China","award":["U21A20514"],"award-info":[{"award-number":["U21A20514"]}]},{"name":"the Major Science and Technology Plan Project on the Future Industry Fields of Xiamen City","award":["3502Z20241027"],"award-info":[{"award-number":["3502Z20241027"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755250","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:38Z","timestamp":1761377198000},"page":"402-411","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["TFPA: Text Features Guided Dynamic Parameter Adjustment for Few Shot Action Recognition"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-8877-5474","authenticated-orcid":false,"given":"Hanyu","family":"Guo","sequence":"first","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Ministry of Education of China, Xiamen University, Xiamen, China and Fujian Key Laboratory of Sensing and Computing for Smart City, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5183-5056","authenticated-orcid":false,"given":"Suzhou","family":"Que","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Ministry of Education of China, Xiamen University, Xiamen, China and Fujian Key Laboratory of Sensing and Computing for Smart City, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8734-1021","authenticated-orcid":false,"given":"Junlong","family":"Gao","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Ministry of Education of China, Xiamen University, Xiamen, China and Fujian Key Laboratory of Sensing and Computing for Smart City, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6913-9786","authenticated-orcid":false,"given":"Hanzi","family":"Wang","sequence":"additional","affiliation":[{"name":"Key Laboratory of Multimedia Trusted Perception and Efficient Computing, Ministry of Education of China, Xiamen University, Xiamen, China and Fujian Key Laboratory of Sensing and Computing for Smart City, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681081"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_5_1","volume-title":"International Conference on Learning Representations. 1-12","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An image is worth 16x16 words: Transformers for image recognition at scale. In International Conference on Learning Representations. 1-12."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2025.3533573"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351015"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413502"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.622"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2605-2615","author":"Goyal Raghav","year":"2023","unstructured":"Raghav Goyal, Samira Ebrahimi Kahou, Vincent Michalski, Joanna Materzynska, Susanne Westphal, Heuna Kim, Valentin Haenel, Ingo Fruend, Peter Yianilos, Moritz Mueller-Freitag, Florian Hoppe, Christian Thurau, Ingo Bax, and Roland Memisevic. 2023. Not all features matter: Enhancing few-shot CLIP with adaptive prior refinement. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2605-2615."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2025.126411"},{"key":"e_1_3_2_1_12_1","volume-title":"Video-to-task learning via motion-guided attention for few-shot action recognition. arXiv:2411.11335","author":"Guo Hanyu","year":"2024","unstructured":"Hanyu Guo, Wanchuan Yu, Suzhou Que, Kaiwen Du, Yan Yan, and Hanzi Wang. 2024a. Video-to-task learning via motion-guided attention for few-shot action recognition. arXiv:2411.11335 (2024)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447209"},{"key":"e_1_3_2_1_14_1","volume-title":"European Conference on Computer Vision. 182-199","author":"Hatano Masashi","year":"2024","unstructured":"Masashi Hatano, Ryo Hachiuma, Ryo Fujii, and Hideo Saito. 2024. Multimodal cross-domain few-shot learning for egocentric action recognition. In European Conference on Computer Vision. 182-199."},{"key":"e_1_3_2_1_15_1","volume-title":"International Conference on Learning Representations. 1-12","author":"He Junxian","year":"2022","unstructured":"Junxian He, Chunting Zhou, Xuezhe Ma, Taylor Berg-Kirkpatrick, and Graham Neubig. 2022. Towards a unified view of parameter-efficient transfer learning. In International Conference on Learning Representations. 1-12."},{"key":"e_1_3_2_1_16_1","volume-title":"International Conference on Machine Learning. 2790-2799","author":"Houlsby Neil","year":"2019","unstructured":"Neil Houlsby, Andrei Giurgiu, Stanislaw Jastrzebski, Bruna Morrone, Quentin De Laroussilhe, Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly. 2019. Parameter-efficient transfer learning for NLP. In International Conference on Machine Learning. 2790-2799."},{"key":"e_1_3_2_1_17_1","volume-title":"International Conference on Learning Representations. 1-13","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-rank adaptation of large language models. In International Conference on Learning Representations. 1-13."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681062"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-024-02017-7"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.59"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2556-2563","author":"Kuehne H.","key":"e_1_3_2_1_22_1","unstructured":"H. Kuehne, H. Jhuang, E. Garrote, T. Poggio, and T. Serre. 2011. HMDB: A large video database for human motion recognition. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 2556-2563."},{"key":"e_1_3_2_1_23_1","volume-title":"European Conference on Computer Vision. 474 - 493","author":"Kumar Pulkit","year":"2024","unstructured":"Pulkit Kumar, Namitha Padmanabhan, Luke Luo, Sai Saketh Rambhatla, and Abhinav Shrivastava. 2024. Trajectory-aligned space-time tokens for few-shot action recognition. In European Conference on Computer Vision. 474 - 493."},{"key":"e_1_3_2_1_24_1","volume-title":"Hyun Seok Seong, and Jae-Pil Heo","author":"Lee SuBeen","year":"2025","unstructured":"SuBeen Lee, WonJun Moon, Hyun Seok Seong, and Jae-Pil Heo. 2025. Temporal Alignment-Free Video Matching for Few-shot Action Recognition. arXiv:2504.05956 (2025)."},{"key":"e_1_3_2_1_25_1","volume-title":"Learning causal domain-invariant temporal dynamics for few-shot action recognition. arXiv:2402.12706","author":"Li Yuke","year":"2024","unstructured":"Yuke Li, Guangyi Chen, Ben Abramowitz, Stefano Anzellott, and Donglai Wei. 2024. Learning causal domain-invariant temporal dynamics for few-shot action recognition. arXiv:2402.12706 (2024)."},{"key":"e_1_3_2_1_26_1","volume-title":"Hill","author":"Markham Georgia","year":"2024","unstructured":"Georgia Markham, Mehala Balamurali, and Andrew J. Hill. 2024. Understanding the cross-domain capabilities of video-based few-shot action recognition models. arXiv:2406.01073 (2024)."},{"key":"e_1_3_2_1_27_1","first-page":"26462","article-title":"St-adapter: Parameter-efficient image-to-video transfer learning","volume":"35","author":"Pan Junting","year":"2022","unstructured":"Junting Pan, Ziyi Lin, Xiatian Zhu, Jing Shao, and Hongsheng Li. 2022. St-adapter: Parameter-efficient image-to-video transfer learning. Advances in Neural Information Processing Systems, Vol. 35 (2022), 26462-26477.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00054"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.590"},{"key":"e_1_3_2_1_30_1","volume-title":"International Conference on Machine Learning. 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. 8748-8763."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3103677"},{"key":"e_1_3_2_1_32_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Roshan Zamir, and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_1_33_1","first-page":"12991","article-title":"Lst: Ladder side-tuning for parameter and memory efficient transfer learning","volume":"35","author":"Sung Yi-Lin","year":"2022","unstructured":"Yi-Lin Sung, Jaemin Cho, and Mohit Bansal. 2022. Lst: Ladder side-tuning for parameter and memory efficient transfer learning. Advances in Neural Information Processing Systems, Vol. 35 (2022), 12991-13005.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00633"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01933"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2020.2978855"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2023.3331841"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3384875"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2024.3354104"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3262670"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01917-4"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01727"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01932"},{"key":"e_1_3_2_1_47_1","volume-title":"TAMT: Temporal-aware model tuning for cross-domain few-shot action recognition. arXiv:2411.19041","author":"Wang Yilong","year":"2024","unstructured":"Yilong Wang, Zilin Gao, Qilong Wang, Zhaofeng Chen, Peihua Li, and Qinghua Hu. 2024a. TAMT: Temporal-aware model tuning for cross-domain few-shot action recognition. arXiv:2411.19041 (2024)."},{"key":"e_1_3_2_1_48_1","volume-title":"Medical sam adapter: Adapting segment anything model for medical image segmentation. arXiv:2304.12620","author":"Wu Junde","year":"2023","unstructured":"Junde Wu, Wei Ji, Yuanpei Liu, Huazhu Fu, Min Xu, Yanwu Xu, and Yueming Jin. 2023. Medical sam adapter: Adapting segment anything model for medical image segmentation. arXiv:2304.12620 (2023)."},{"key":"e_1_3_2_1_49_1","volume-title":"International Conference on Learning Representations. 1-14","author":"Yang Taojiannan","year":"2023","unstructured":"Taojiannan Yang, Yi Zhu, Yusheng Xie, Aston Zhang, Chen Chen, and Mu Li. 2023. Aim: Adapting image models for efficient video understanding. In International Conference on Learning Representations. 1-14."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612380"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612380"},{"key":"e_1_3_2_1_52_1","volume-title":"International Joint Conference on Artificial Intelligence. 5425-5433","author":"Zhang Bin","year":"2024","unstructured":"Bin Zhang, Yuanjie Dang, Peng Chen, Ronghua Liang, Nan Gao, Ruohong Huan, and Xiaofei He. 2024. Task-agnostic self-distillation for few-shot action recognition. In International Joint Conference on Artificial Intelligence. 5425-5433."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_31"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612192"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_18"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_46"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755250","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:47:30Z","timestamp":1765309650000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755250"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":56,"alternative-id":["10.1145\/3746027.3755250","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755250","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}