{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T15:29:13Z","timestamp":1781018953073,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":39,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819785100","type":"print"},{"value":"9789819785117","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8511-7_29","type":"book-chapter","created":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T05:11:18Z","timestamp":1730524278000},"page":"409-423","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Language-Skeleton Pre-training to Collaborate with Self-Supervised Human Action Recognition"],"prefix":"10.1007","author":[{"given":"Yi","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruyi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wentian","family":"Xin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiguang","family":"Miao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuzhi","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiahao","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,3]]},"reference":[{"key":"29_CR1","unstructured":"Oord, A., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. In: arXiv:1807.03748 (2018)"},{"key":"29_CR2","doi-asserted-by":"crossref","unstructured":"Xin, W., Liu, Y., Liu, R., Miao, Q., et al.: Auto-Learning-GCN: an ingenious framework for skeleton-based action recognition. In: PRCV, pp. 29\u201342 (2023)","DOI":"10.1007\/978-981-99-8429-9_3"},{"key":"29_CR3","doi-asserted-by":"crossref","unstructured":"Liu, R., Liu, Y., Xin, W., Miao, Q., et al.: Action Jitter Killer: joint noise optimization cascade for skeleton-based action recognition. In: IEEE TIM (2024)","DOI":"10.1109\/TIM.2024.3370958"},{"key":"29_CR4","doi-asserted-by":"crossref","unstructured":"Xin, W., Miao, Q., Liu, Y., Liu, R., et al.: Skeleton MixFormer: multivariate topology representation for skeleton-based action recognition. In: ACMMM, pp. 2211\u20132220 (2023)","DOI":"10.1145\/3581783.3611900"},{"key":"29_CR5","doi-asserted-by":"crossref","unstructured":"Xin, W., Lin, H., Liu, R., Liu, Y., Miao, Q: Is really correlation information represented well in self-attention for skeleton-based action recognition? In: ICME, pp. 780\u2013785 (2023)","DOI":"10.1109\/ICME55011.2023.00139"},{"key":"29_CR6","doi-asserted-by":"crossref","unstructured":"Xin, W., Liu, R., Liu, Y. et al.: Transformer for skeleton-based action recognition: a review of recent advances. In: Neurocomputing (2023)","DOI":"10.1016\/j.neucom.2023.03.001"},{"key":"29_CR7","doi-asserted-by":"crossref","unstructured":"Tevet, G., Gordon, B., Hertz, A. et al.: Motionclip: exposing human motion generation to clip space. In: ECCV, pp. 358\u2013374 (2022)","DOI":"10.1007\/978-3-031-20047-2_21"},{"key":"29_CR8","unstructured":"Wang, M., Xing, J., Liu, Y.: Actionclip: a new paradigm for video action recognition. In: arXiv:2109.08472 (2021)"},{"key":"29_CR9","unstructured":"Zhang, Y., Jiang, H., Miura, Y. et al.: Contrastive learning of medical visual representations from paired images and text. In: PMLR, pp. 2\u201325 (2022)"},{"key":"29_CR10","doi-asserted-by":"crossref","unstructured":"Sariyildiz, M.B., Perez, J., Larlus, D.: Learning visual representations with caption annotations. In: ECCV, pp. 153\u2013170 (2020)","DOI":"10.1007\/978-3-030-58598-3_10"},{"key":"29_CR11","doi-asserted-by":"crossref","unstructured":"Xiang, W., Li, C., Zhou, Y. et al.: Generative action description prompts for skeleton-based action recognition. In: CVPR, pp. 10276\u201310285 (2023)","DOI":"10.1109\/ICCV51070.2023.00943"},{"key":"29_CR12","doi-asserted-by":"crossref","unstructured":"Sato, F., Hachiuma, R., Sekii, T.: Prompt-guided zero-shot anomaly action recognition using pretrained deep skeleton features. In: CVPR, pp. 6471\u20136480 (2023)","DOI":"10.1109\/CVPR52729.2023.00626"},{"key":"29_CR13","doi-asserted-by":"crossref","unstructured":"Hyeon-Woo, N., Yu-Ji, K., Heo, B., Han, D., et al.: Scratching visual transformer\u2019s back with uniform attention. In: ICCV, pp. 5807\u20135818 (2023)","DOI":"10.1109\/ICCV51070.2023.00534"},{"key":"29_CR14","unstructured":"Jia, C., Yang, Y. et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: PMLR, pp. 4904\u20134916 (2021)"},{"key":"29_CR15","unstructured":"Radford, A. et al.: Learning transferable visual models from natural language supervision. In: PMLR, pp. 8748\u20138763 (2021)"},{"key":"29_CR16","doi-asserted-by":"crossref","unstructured":"Li, K., Zhang, Y., Li, K. et al.: Visual semantic reasoning for image-text matching. In: ICCV, pp. 4654\u20134662 (2019)","DOI":"10.1109\/ICCV.2019.00475"},{"key":"29_CR17","doi-asserted-by":"crossref","unstructured":"Chen, Y.C., Li, L., Yu, L. et al.: Uniter: universal image-text representation learning. In: ECCV, pp. 104\u2013120 (2020)","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"29_CR18","doi-asserted-by":"crossref","unstructured":"Song, S., Lan, C., Xing, J. et al.: Skeleton-indexed deep multi-modal feature learning for high performance human action recognition. In: ICME, pp. 1\u20136 (2018)","DOI":"10.1109\/ICME.2018.8486486"},{"key":"29_CR19","doi-asserted-by":"crossref","unstructured":"Zhu, X., Zhu, Y., Wang, H., et al.: Skeleton sequence and RGB frame based multi-modality feature fusion network for action recognition. In: IEEE TOMM, pp. 1\u201324 (2022)","DOI":"10.1145\/3491228"},{"key":"29_CR20","doi-asserted-by":"crossref","unstructured":"Li, L., Wang, M., Ni, B., et al.: 3d human action representation learning via cross-view consistency pursuit. In: CVPR, pp. 4741\u20134750 (2021)","DOI":"10.1109\/CVPR46437.2021.00471"},{"key":"29_CR21","doi-asserted-by":"crossref","unstructured":"Guo, T., Liu, H., Z. et al: Contrastive learning from extremely augmented skeleton sequences for self-supervised action recognition. In: AAAI, pp. 762\u2013770 (2022)","DOI":"10.1609\/aaai.v36i1.19957"},{"key":"29_CR22","unstructured":"Chen, X., Fan, H., Girshick, R., et al.: Improved baselines with momentum contrastive learning. In: arXiv:2003.04297 (2020)"},{"key":"29_CR23","doi-asserted-by":"crossref","unstructured":"Shahroudy, A., Liu, J., et al.: Ntu rgb+ d: a large scale dataset for 3d human activity analysis. In: CVPR, pp. 1010\u20131019 (2016)","DOI":"10.1109\/CVPR.2016.115"},{"key":"29_CR24","doi-asserted-by":"crossref","unstructured":"Liu, J., Shahroudy, A., et al.: Ntu rgb+ d 120: A large-scale benchmark for 3d human activity understanding. In: IEEE TPAMI 42(10), pp. 2684\u20132701 (2019)","DOI":"10.1109\/TPAMI.2019.2916873"},{"key":"29_CR25","doi-asserted-by":"crossref","unstructured":"Liu, J., Song, S. et al.: A benchmark dataset and comparison study for multi-modal human action analytics. In: IEEE TOMM 16(2), pp. 1\u201324 (2020)","DOI":"10.1145\/3365212"},{"key":"29_CR26","doi-asserted-by":"crossref","unstructured":"Li, T., Liu, J., Zhang, W., et al.: Uav-human: a large benchmark for human behavior understanding with unmanned aerial vehicles. In: CVPR, pp. 16266\u201316275 (2021)","DOI":"10.1109\/CVPR46437.2021.01600"},{"key":"29_CR27","doi-asserted-by":"crossref","unstructured":"Zheng, N., Wen, J., Liu, R. et al.: Unsupervised representation learning with long-term dynamics for skeleton based action recognition. In: AAAI, Vol. 32, No. 1 (2018)","DOI":"10.1609\/aaai.v32i1.11853"},{"key":"29_CR28","unstructured":"Lin, L., Song, S., Yang, W., et al.: Multi-task self-supervised learning for skeleton based action recognition. In: ACMMM, pp. 2490\u20132498 (2020)"},{"key":"29_CR29","doi-asserted-by":"crossref","unstructured":"Su, K., Liu, X., et al.: Predict & cluster:Unsupervised skeleton based action recognition. In: CVPR, pp. 9631\u20139640 (2020)","DOI":"10.1109\/CVPR42600.2020.00965"},{"key":"29_CR30","unstructured":"Chen, Z., Liu, H., Guo, T., Chen, Z., Song, P. et al.: Contrastive learning from spatio-temporal mixed skeleton sequences for self-supervised skeleton-based action recognition. In: arXiv:2207.03065 (2022)"},{"key":"29_CR31","doi-asserted-by":"crossref","unstructured":"Rao, H., Xu, S., et al.: Augmented skeleton based contrastive action learning with momentum lstm for unsupervised action recognition. In: Information Sciences, pp. 90\u2013109 (2021)","DOI":"10.1016\/j.ins.2021.04.023"},{"key":"29_CR32","doi-asserted-by":"crossref","unstructured":"Thoker, F.M., Doughty, H. et al.: Skeleton-contrastive 3d action representation learning. In: ACMMM, pp. 1655\u20131663 (2021)","DOI":"10.1145\/3474085.3475307"},{"key":"29_CR33","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: AAAI, Vol. 32, No. 1 (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"29_CR34","doi-asserted-by":"crossref","unstructured":"Lin, L., Zhang, J., Liu, J.: Actionlet-dependent contrastive learning for unsupervised skeleton-based action recognition. In: CVPR, pp. 2363\u20132372 (2023)","DOI":"10.1109\/CVPR52729.2023.00234"},{"key":"29_CR35","doi-asserted-by":"crossref","unstructured":"Mao, Y., Zhou, W., Lu, Z., et al.: Cmd: self-supervised 3d action representation learning with cross-modal mutual distillation. In: ECCV, pp. 734\u2013752 (2022)","DOI":"10.1007\/978-3-031-20062-5_42"},{"key":"29_CR36","unstructured":"Van der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. In: JMLR, pp. 7444\u20137452 (2018)"},{"key":"29_CR37","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: CVPR, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"29_CR38","doi-asserted-by":"crossref","unstructured":"Jin, Z., Wang, Y., Wang, Q., Shen, Y., et al.: SSRL: Self-supervised spatial-temporal representation learning for 3D action recognition. In: IEEE TCSVT (2023)","DOI":"10.1109\/TCSVT.2023.3284493"},{"key":"29_CR39","doi-asserted-by":"crossref","unstructured":"Guan, S., Yu, X., Huang, W., Fang, G. et al.: DMMG: dual min-max games for self-supervised skeleton-based action recognition. In: IEEE TIP (2023)","DOI":"10.1109\/TIP.2023.3338410"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8511-7_29","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T05:13:21Z","timestamp":1730524401000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8511-7_29"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,3]]},"ISBN":["9789819785100","9789819785117"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8511-7_29","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,3]]},"assertion":[{"value":"3 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}