{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T16:21:16Z","timestamp":1771258876561,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":50,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819561223","type":"print"},{"value":"9789819561230","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-6123-0_51","type":"book-chapter","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T15:43:55Z","timestamp":1771256635000},"page":"550-561","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Temporally-Aware Multi-task Representation Learning for\u00a0Compositional Action Recognition"],"prefix":"10.1007","author":[{"given":"Peng","family":"Huang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenxuan","family":"Ge","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"He","family":"Yan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Henghao","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiangbo","family":"Shu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,17]]},"reference":[{"key":"51_CR1","doi-asserted-by":"crossref","unstructured":"Materzynska, J., Xiao, T., Herzig, R., et al.: Something-else: compositional action recognition with spatial-temporal interaction networks. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1049\u20131059 (2020)","DOI":"10.1109\/CVPR42600.2020.00113"},{"key":"51_CR2","doi-asserted-by":"crossref","unstructured":"Yan, R., Huang, P., Shu, X., et al.: Look less think more: rethinking compositional action recognition. In: ACM International Conference on Multimedia (ACM MM), pp. 3666\u20133675 (2022)","DOI":"10.1145\/3503161.3547862"},{"key":"51_CR3","doi-asserted-by":"publisher","first-page":"297","DOI":"10.1109\/TIP.2023.3341297","volume":"33","author":"P Huang","year":"2023","unstructured":"Huang, P., Yan, R., Shu, X., et al.: Semantic-disentangled transformer with noun-verb embedding for compositional action recognition. IEEE Trans. Image Process. 33, 297\u2013309 (2023)","journal-title":"IEEE Trans. Image Process."},{"key":"51_CR4","doi-asserted-by":"crossref","unstructured":"Huang, P., Shu, X., Yan, R., et al.: Appearance-agnostic representation learning for compositional action recognition. IEEE Trans. Circuits Syst. Video Technol. (2024)","DOI":"10.1109\/TCSVT.2024.3384392"},{"issue":"8","key":"51_CR5","doi-asserted-by":"publisher","first-page":"5281","DOI":"10.1109\/TCSVT.2022.3142771","volume":"32","author":"X Shu","year":"2022","unstructured":"Shu, X., Yang, J., Yan, R., et al.: Expansion-squeeze-excitation fusion network for elderly activity recognition. IEEE Trans. Circuits Syst. Video Technol. 32(8), 5281\u20135292 (2022)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"8","key":"51_CR6","doi-asserted-by":"publisher","first-page":"11035","DOI":"10.1109\/TNNLS.2023.3247103","volume":"35","author":"B Xu","year":"2023","unstructured":"Xu, B., Shu, X., Zhang, J., et al.: Spatiotemporal decouple-and-squeeze contrastive learning for semisupervised skeleton-based action recognition. IEEE Trans. Neural Netw. Learn. Syst. 35(8), 11035\u201311048 (2023)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"11","key":"51_CR7","doi-asserted-by":"publisher","first-page":"10718","DOI":"10.1109\/TCSVT.2024.3416732","volume":"34","author":"X Zhu","year":"2024","unstructured":"Zhu, X., Shu, X., Tang, J., et al.: Motion-aware mask feature reconstruction for skeleton-based action recognition. IEEE Trans. Circuits Syst. Video Technol. 34(11), 10718\u201310731 (2024)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"51_CR8","unstructured":"Zhao, H., Lin, K., Yan, R., et al.: DiffusionVMR: diffusion model for joint video moment retrieval and highlight detection. IEEE Trans. Neural Netw. Learn. Syst. 1\u201314 (2024)"},{"key":"51_CR9","doi-asserted-by":"crossref","unstructured":"Zhao, H., Ji, G., Yan, R., et al.: VideoExpert: augmented LLM for temporal-sensitive video understanding. arXiv preprint arXiv:2504.07519 (2025)","DOI":"10.1109\/TCSVT.2026.3653742"},{"key":"51_CR10","doi-asserted-by":"crossref","unstructured":"Li, R., Feng, Z., Xu, T., et al.: C2C: component-to-composition learning for zero-shot compositional action recognition. In: European Conference on Computer Vision (ECCV), pp. 369\u2013388 (2024)","DOI":"10.1007\/978-3-031-72920-1_21"},{"key":"51_CR11","unstructured":"Patrick, M., Campbell, D., Asano, Y., et al.: Keeping your eye on the ball: trajectory attention in video transformers. In: Advances in Neural Information Processing Systems (NeurIPS), pp. 12493\u201312506 (2021)"},{"key":"51_CR12","unstructured":"Li, K., Wang, Y., He, Y., et al.: UniformerV2: Spatiotemporal learning by arming image ViTs with video Uniformer. arXiv preprint arXiv:2211.09552 (2022)"},{"key":"51_CR13","doi-asserted-by":"crossref","unstructured":"Li, Y., Wu, C.-Y., Fan, H., et al.: MViTv2: improved multiscale vision transformers for classification and detection. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"51_CR14","doi-asserted-by":"crossref","unstructured":"Qu, H., Yan, R., Shu, X., et al.: MVP-shot: multi-velocity progressive-alignment framework for few-shot action recognition. IEEE Trans. Multimedia (2025)","DOI":"10.1109\/TMM.2025.3586118"},{"issue":"4","key":"51_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3655025","volume":"21","author":"Z Tu","year":"2025","unstructured":"Tu, Z., Shu, X., Huang, P., et al.: Leveraging frame- and feature-level progressive augmentation for semi-supervised action recognition. ACM Trans. Multimedia Comput. Commun. Appl. 21(4), 1\u201321 (2025)","journal-title":"ACM Trans. Multimedia Comput. Commun. Appl."},{"issue":"3","key":"51_CR16","doi-asserted-by":"publisher","first-page":"534","DOI":"10.26599\/BDMA.2024.9020076","volume":"8","author":"R Wei","year":"2025","unstructured":"Wei, R., Yan, R., Qu, H., et al.: SVMFN-FSAR: semantic-guided video multimodal fusion network for few-shot action recognition. Big Data Min. Anal. 8(3), 534\u2013550 (2025)","journal-title":"Big Data Min. Anal."},{"key":"51_CR17","doi-asserted-by":"crossref","unstructured":"Yan, R., Wang, J., Qu, H., et al.: TEST-V: TEst-time Support-set Tuning for Zero-shot Video Classification. arXiv preprint arXiv:2502.00426 (2025)","DOI":"10.24963\/ijcai.2025\/239"},{"key":"51_CR18","doi-asserted-by":"crossref","unstructured":"Wang, H., Schmid, C.: Action recognition with improved trajectories. In: International Conference on Computer Vision (ICCV), pp. 3551\u20133558 (2013)","DOI":"10.1109\/ICCV.2013.441"},{"key":"51_CR19","doi-asserted-by":"crossref","unstructured":"Peng, X., Zou, C., Qiao, Y., et al.: Action recognition with stacked fisher vectors. In: European Conference on Computer Vision (ECCV), pp. 581\u2013595 (2014)","DOI":"10.1007\/978-3-319-10602-1_38"},{"issue":"8","key":"51_CR20","doi-asserted-by":"publisher","first-page":"10317","DOI":"10.1109\/TPAMI.2023.3261659","volume":"45","author":"R Yan","year":"2023","unstructured":"Yan, R., Xie, L., Shu, X., et al.: Progressive instance-aware feature learning for compositional action recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45(8), 10317\u201310330 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"6","key":"51_CR21","doi-asserted-by":"publisher","first-page":"6955","DOI":"10.1109\/TPAMI.2020.3034233","volume":"45","author":"R Yan","year":"2020","unstructured":"Yan, R., Xie, L., Tang, J., et al.: HiGCIN: hierarchical graph-based cross inference network for group activity recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45(6), 6955\u20136968 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"15","key":"51_CR22","doi-asserted-by":"publisher","first-page":"4860","DOI":"10.3390\/s24154860","volume":"24","author":"Q Chen","year":"2024","unstructured":"Chen, Q., Liu, Y., Huang, P., et al.: Linguistic-driven partial semantic relevance learning for skeleton-based action recognition. Sensors 24(15), 4860 (2024)","journal-title":"Sensors"},{"issue":"3","key":"51_CR23","doi-asserted-by":"publisher","first-page":"606","DOI":"10.26599\/BDMA.2024.9020087","volume":"8","author":"J Wang","year":"2025","unstructured":"Wang, J., Guo, J., Wang, R., et al.: Parameter disentanglement for diverse representations. Big Data Min. Anal. 8(3), 606\u2013623 (2025)","journal-title":"Big Data Min. Anal."},{"key":"51_CR24","doi-asserted-by":"crossref","unstructured":"Wang, L., Qiao, Y., Tang, X.: Action recognition with trajectory-pooled deep-convolutional descriptors. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4305\u20134314 (2015)","DOI":"10.1109\/CVPR.2015.7299059"},{"key":"51_CR25","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., et al.: Temporal segment networks: towards good practices for deep action recognition. In: European Conference on Computer Vision (ECCV), pp. 20\u201336 (2016)","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"51_CR26","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the Kinetics dataset. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"51_CR27","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C.: X3D: expanding architectures for efficient video recognition. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 203\u2013213 (2020)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"51_CR28","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"51_CR29","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: International Conference on Machine Learning (ICML), pp. 813\u2013824 (2021)"},{"key":"51_CR30","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., et al.: Multiscale vision transformers. In: International Conference on Computer Vision (ICCV), pp. 6824\u20136835 (2021)","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"51_CR31","unstructured":"Tong, Z., Song, Y., Wang, J., et al.: VideoMAE: masked autoencoders are data-efficient learners for self-supervised video pre-training. In: Advances in Neural Information Processing Systems (NeurIPS), pp. 10078\u201310093 (2022)"},{"key":"51_CR32","unstructured":"Yan, R., Xie, L., Shu, X., et al.: Interactive fusion of multi-level features for compositional activity recognition. arXiv preprint arXiv:2012.05689 (2020)"},{"key":"51_CR33","doi-asserted-by":"crossref","unstructured":"Radevski, G., Moens, M.F., Tuytelaars, T.: Revisiting spatio-temporal layouts for compositional action recognition. In: British Machine Vision Conference (BMVC), p. 110 (2021)","DOI":"10.5244\/C.35.278"},{"key":"51_CR34","doi-asserted-by":"crossref","unstructured":"Herzig, R., Ben-Avraham, E., Mangalam, K., et al.: Object-region video transformers. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3148\u20133159 (2022)","DOI":"10.1109\/CVPR52688.2022.00315"},{"key":"51_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, C., Gupta, A., Zisserman, A.: Is an object-centric video representation beneficial for transfer? In: Asian Conference on Computer Vision (ACCV), pp. 1976\u20131994 (2022)","DOI":"10.1007\/978-3-031-26316-3_23"},{"key":"51_CR36","doi-asserted-by":"crossref","unstructured":"Huang, P., Qu, H., Shu, X.: Revisiting few-shot compositional action recognition with knowledge calibration. IEEE Signal Process. Lett. (2025)","DOI":"10.1109\/LSP.2025.3542702"},{"key":"51_CR37","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., et al.: Segment anything. In: International Conference on Computer Vision (ICCV), pp. 3992\u20134003 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"51_CR38","doi-asserted-by":"crossref","unstructured":"Goyal, R., Kahou, S.E., Michalski, V., et al.: The \u201csomething something\u201d video database for learning and evaluating visual common sense. In: International Conference on Computer Vision (ICCV), pp. 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"51_CR39","unstructured":"Kay, W., Carreira, J., Simonyan, K., et al.: The Kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"51_CR40","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., et al.: SlowFast networks for video recognition. In: International Conference on Computer Vision (ICCV), pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"51_CR41","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., Han, S., et al.: TSM: temporal shift module for efficient video understanding. In: International Conference on Computer Vision (ICCV), pp. 7083\u20137093 (2019)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"51_CR42","doi-asserted-by":"crossref","unstructured":"Sun, P., Wu, B., Li, X., et al.: Counterfactual debiasing inference for compositional action recognition. In: ACM International Conference on Multimedia (ACM MM), pp. 3220\u20133228 (2021)","DOI":"10.1145\/3474085.3475472"},{"key":"51_CR43","doi-asserted-by":"crossref","unstructured":"Kim, T.S., Jones, J., Hager, G.D.: Motion-guided attention fusion to recognize interactions from videos. In: International Conference on Computer Vision (ICCV), pp. 13076\u201313086 (2021)","DOI":"10.1109\/ICCV48922.2021.01283"},{"key":"51_CR44","doi-asserted-by":"crossref","unstructured":"Zhou, X., Arnab, A., Sun, C., et al.: How can objects help action recognition? In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2353\u20132362 (2023)","DOI":"10.1109\/CVPR52729.2023.00233"},{"key":"51_CR45","unstructured":"Radford, A., Kim, J.W., Hallacy, C., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning (ICML), pp. 8748\u20138763 (2021)"},{"issue":"8","key":"51_CR46","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., et al.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"51_CR47","unstructured":"Sanh, V., Debut, L., Chaumond, J., et al.: DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108 (2019)"},{"key":"51_CR48","unstructured":"He, P., Liu, X., Gao, J., et al.: DeBERTa: decoding-enhanced BERT with disentangled attention. arXiv preprint arXiv:2006.03654 (2020)"},{"key":"51_CR49","unstructured":"Lan, Z., Chen, M., Goodman, S., et al.: ALBERT: a lite BERT for self-supervised learning of language representations. arXiv preprint arXiv:1909.11942 (2019)"},{"key":"51_CR50","unstructured":"Liu, Y., Ott, M., Goyal, N., et al.: RoBERTa: a robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"}],"container-title":["Lecture Notes in Computer Science","Biometric Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-6123-0_51","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T15:44:05Z","timestamp":1771256645000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-6123-0_51"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819561223","9789819561230"],"references-count":50,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-6123-0_51","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"17 February 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CCBR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Biometric Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Nanchang","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 November 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccbr2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ccbr99.cn\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}