{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,12,17]],"date-time":"2024-12-17T05:05:53Z","timestamp":1734411953298,"version":"3.30.2"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T00:00:00Z","timestamp":1733011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s00530-024-01592-6","type":"journal-article","created":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T14:01:48Z","timestamp":1733234508000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MKTZ: multi-semantic embedding and key frame masking techniques for zero-shot skeleton action recognition"],"prefix":"10.1007","volume":"30","author":[{"given":"Hongwei","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sheng","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zexi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,3]]},"reference":[{"key":"1592_CR1","doi-asserted-by":"crossref","unstructured":"Zhang, T., Wang, Q., Dong, X., Yu, W., Sun, H., Zhou, X., Zhen, A., Cui, S., Wu, D., He, Z.: Augmented self-mask attention transformer for naturalistic driving action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops, pp. 7108\u20137114 (2024)","DOI":"10.1109\/CVPRW63382.2024.00705"},{"key":"1592_CR2","doi-asserted-by":"crossref","unstructured":"Wang, X., Fang, Z., Li, X., Li, X., Chen, C., Liu, M.: Skeleton-in-context: Unified skeleton sequence modeling with in-context learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2436\u20132446 (2024)","DOI":"10.1109\/CVPR52733.2024.00236"},{"key":"1592_CR3","doi-asserted-by":"crossref","unstructured":"Keetha, N., Karhade, J., Jatavallabhula, K.M., Yang, G., Scherer, S., Ramanan, D., Luiten, J.: Splatam: Splat track & map 3d gaussians for dense rgb-d slam. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 21357\u201321366 (2024)","DOI":"10.1109\/CVPR52733.2024.02018"},{"key":"1592_CR4","unstructured":"Zhang, Z., Wang, X., Zhang, Z., Shen, G., Shen, S., Zhu, W.: Unsupervised graph neural architecture search with disentangled self-supervision. Advances in Neural Information Processing Systems 36 (2024)"},{"key":"1592_CR5","doi-asserted-by":"crossref","unstructured":"Guo, T., Liu, H., Chen, Z., Liu, M., Wang, T., Ding, R.: Contrastive learning from extremely augmented skeleton sequences for self-supervised action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 36, pp. 762\u2013770 (2022)","DOI":"10.1609\/aaai.v36i1.19957"},{"key":"1592_CR6","unstructured":"Li, J., Zhang, J., Schmidt, L., Ratner, A.J.: Characterizing the impacts of semi-supervised learning for weak supervision. Advances in Neural Information Processing Systems 36 (2024)"},{"key":"1592_CR7","doi-asserted-by":"crossref","unstructured":"Hou, W., Chen, S., Chen, S., Hong, Z., Wang, Y., Feng, X., Khan, S., Khan, F.S., You, X.: Visual-augmented dynamic semantic prototype for generative zero-shot learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 23627\u201323637 (2024)","DOI":"10.1109\/CVPR52733.2024.02230"},{"key":"1592_CR8","doi-asserted-by":"crossref","unstructured":"Chen, S., Hou, W., Khan, S., Khan, F.S.: Progressive semantic-guided vision transformer for zero-shot learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 23964\u201323974 (2024)","DOI":"10.1109\/CVPR52733.2024.02262"},{"key":"1592_CR9","doi-asserted-by":"crossref","unstructured":"Xu, C., Tan, R.T., Tan, Y., Chen, S., Wang, X., Wang, Y.: Auxiliary tasks benefit 3d skeleton-based human motion prediction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9509\u20139520 (2023)","DOI":"10.1109\/ICCV51070.2023.00872"},{"key":"1592_CR10","doi-asserted-by":"crossref","unstructured":"Li, M.-Z., Jia, Z., Zhang, Z., Ma, Z., Wang, L.: Multi-semantic fusion model for generalized zero-shot skeleton-based action recognition. In: International Conference on Image and Graphics, pp. 68\u201380 (2023). Springer","DOI":"10.1007\/978-3-031-46305-1_6"},{"key":"1592_CR11","doi-asserted-by":"crossref","unstructured":"Zhu, A., Ke, Q., Gong, M., Bailey, J.: Part-aware unified representation of language and skeleton for zero-shot action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18761\u201318770 (2024)","DOI":"10.1109\/CVPR52733.2024.01775"},{"key":"1592_CR12","doi-asserted-by":"crossref","unstructured":"Wray, M., Larlus, D., Csurka, G., Damen, D.: Fine-grained action retrieval through multiple parts-of-speech embeddings. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00054"},{"key":"1592_CR13","doi-asserted-by":"crossref","unstructured":"Xue, F., Budvytis, I., Cipolla, R.: Sfd2: Semantic-guided feature detection and description. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 5206\u20135216 (2023)","DOI":"10.1109\/CVPR52729.2023.00504"},{"key":"1592_CR14","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S.-E., Sheikh, Y.: Realtime multi-person 2d pose estimation using part affinity fields. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.143"},{"issue":"1","key":"1592_CR15","doi-asserted-by":"publisher","first-page":"116","DOI":"10.1145\/2398356.2398381","volume":"56","author":"J Shotton","year":"2013","unstructured":"Shotton, J., Sharp, T., Kipman, A., Fitzgibbon, A., Finocchio, M., Blake, A., Cook, M., Moore, R.: Real-time human pose recognition in parts from single depth images. Commun. ACM 56(1), 116\u2013124 (2013)","journal-title":"Commun. ACM"},{"key":"1592_CR16","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Yan, X., Cheng, Z.-Q., Yan, Y., Dai, Q., Hua, X.-S.: Blockgcn: Redefine topology awareness for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2049\u20132058 (2024)","DOI":"10.1109\/CVPR52733.2024.00200"},{"key":"1592_CR17","doi-asserted-by":"crossref","unstructured":"Peng, K., Yin, C., Zheng, J., Liu, R., Schneider, D., Zhang, J., Yang, K., Sarfraz, M.S., Stiefelhagen, R., Roitberg, A.: Navigating open set scenarios for skeleton-based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 4487\u20134496 (2024)","DOI":"10.1609\/aaai.v38i5.28247"},{"key":"1592_CR18","doi-asserted-by":"crossref","unstructured":"Zhou, H., Liu, Q., Wang, Y.: Learning discriminative representations for skeleton based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10608\u201310617 (2023)","DOI":"10.1109\/CVPR52729.2023.01022"},{"key":"1592_CR19","doi-asserted-by":"crossref","unstructured":"Du, Y., Wang, W., Wang, L.: Hierarchical recurrent neural network for skeleton based action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1110\u20131118 (2015)","DOI":"10.1109\/CVPR.2015.7298714"},{"key":"1592_CR20","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"1592_CR21","unstructured":"Huang, X., Zhou, H., Wang, J., Feng, H., Han, J., Ding, E., Wang, J., Wang, X., Liu, W., Feng, B.: Graph contrastive learning for skeleton-based action recognition. arXiv preprint arXiv:2301.10900 (2023)"},{"key":"1592_CR22","doi-asserted-by":"publisher","first-page":"1175","DOI":"10.1109\/TMM.2021.3139768","volume":"25","author":"R Wang","year":"2021","unstructured":"Wang, R., Liu, J., Ke, Q., Peng, D., Lei, Y.: Dear-net: Learning diversities for skeleton-based early action recognition. IEEE Trans. Multimedia 25, 1175\u20131189 (2021)","journal-title":"IEEE Trans. Multimedia"},{"key":"1592_CR23","doi-asserted-by":"publisher","first-page":"1061","DOI":"10.1109\/TMM.2021.3137745","volume":"25","author":"W Wang","year":"2021","unstructured":"Wang, W., Chang, F., Liu, C., Li, G., Wang, B.: Ga-net: a guidance aware network for skeleton-based early activity recognition. IEEE Trans. Multimedia 25, 1061\u20131073 (2021)","journal-title":"IEEE Trans. Multimedia"},{"key":"1592_CR24","doi-asserted-by":"crossref","unstructured":"Xin, W., Miao, Q., Liu, Y., Liu, R., Pun, C.-M., Shi, C.: Skeleton mixformer: Multivariate topology representation for skeleton-based action recognition. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 2211\u20132220 (2023)","DOI":"10.1145\/3581783.3611900"},{"key":"1592_CR25","doi-asserted-by":"crossref","unstructured":"Lin, L., Zhang, J., Liu, J.: Actionlet-dependent contrastive learning for unsupervised skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2363\u20132372 (2023)","DOI":"10.1109\/CVPR52729.2023.00234"},{"key":"1592_CR26","unstructured":"Xu, H., Gao, Y., Hui, Z., Li, J., Gao, X.: Language knowledge-assisted representation learning for skeleton-based action recognition. arXiv preprint arXiv:2305.12398 (2023)"},{"key":"1592_CR27","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084 (2019)","DOI":"10.18653\/v1\/D19-1410"},{"key":"1592_CR28","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021). PMLR"},{"key":"1592_CR29","doi-asserted-by":"crossref","unstructured":"Gupta, P., Sharma, D., Sarvadevabhatla, R.K.: Syntactically guided generative embeddings for zero-shot skeleton action recognition. In: 2021 IEEE International Conference on Image Processing (ICIP), pp. 439\u2013443 (2021). IEEE","DOI":"10.1109\/ICIP42928.2021.9506179"},{"key":"1592_CR30","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Qiang, W., Rao, A., Lin, N., Su, B., Wang, J.: Zero-shot skeleton-based action recognition via mutual information estimation and maximization. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5302\u20135310 (2023)","DOI":"10.1145\/3581783.3611888"},{"issue":"8","key":"1592_CR31","doi-asserted-by":"publisher","first-page":"832","DOI":"10.1109\/34.709601","volume":"20","author":"TK Ho","year":"1998","unstructured":"Ho, T.K.: The random subspace method for constructing decision forests. IEEE Trans. Pattern Anal. Mach. Intell. 20(8), 832\u2013844 (1998)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1592_CR32","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Duan, H., Rao, A., Su, B., Wang, J.: Self-supervised action representation learning from partial spatio-temporal skeleton sequences. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 3825\u20133833 (2023)","DOI":"10.1609\/aaai.v37i3.25495"},{"key":"1592_CR33","doi-asserted-by":"crossref","unstructured":"Shahroudy, A., Liu, J., Ng, T.-T., Wang, G.: Ntu rgb+ d: A large scale dataset for 3d human activity analysis. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1010\u20131019 (2016)","DOI":"10.1109\/CVPR.2016.115"},{"issue":"10","key":"1592_CR34","doi-asserted-by":"publisher","first-page":"2684","DOI":"10.1109\/TPAMI.2019.2916873","volume":"42","author":"J Liu","year":"2019","unstructured":"Liu, J., Shahroudy, A., Perez, M., Wang, G., Duan, L.-Y., Kot, A.C.: Ntu rgb+ d 120: A large-scale benchmark for 3d human activity understanding. IEEE Trans. Pattern Anal. Mach. Intell. 42(10), 2684\u20132701 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1592_CR35","doi-asserted-by":"crossref","unstructured":"Liu, J., Song, S., Liu, C., Li, Y., Hu, Y.: A benchmark dataset and comparison study for multi-modal human action analytics. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM) 16(2), 1\u201324 (2020)","DOI":"10.1145\/3365212"},{"key":"1592_CR36","doi-asserted-by":"crossref","unstructured":"Schonfeld, E., Ebrahimi, S., Sinha, S., Darrell, T., Akata, Z.: Generalized zero- and few-shot learning via aligned variational autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019)","DOI":"10.1109\/CVPR.2019.00844"},{"key":"1592_CR37","doi-asserted-by":"crossref","unstructured":"Hubert\u00a0Tsai, Y.-H., Huang, L.-K., Salakhutdinov, R.: Learning robust visual-semantic embeddings. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3571\u20133580 (2017)","DOI":"10.1109\/ICCV.2017.386"},{"key":"1592_CR38","unstructured":"Frome, A., Corrado, G.S., Shlens, J., Bengio, S., Dean, J., Ranzato, M., Mikolov, T.: Devise: A deep visual-semantic embedding model. Advances in neural information processing systems 26 (2013)"},{"key":"1592_CR39","unstructured":"Jasani, B., Mazagonwalla, A.: Skeleton based zero shot action recognition in joint pose-language semantic space. arXiv preprint arXiv:1911.11344 (2019)"},{"key":"1592_CR40","doi-asserted-by":"crossref","unstructured":"Wray, M., Larlus, D., Csurka, G., Damen, D.: Fine-grained action retrieval through multiple parts-of-speech embeddings. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 450\u2013459 (2019)","DOI":"10.1109\/ICCV.2019.00054"},{"key":"1592_CR41","unstructured":"Maaten, L., Hinton, G.: Visualizing data using t-sne. Journal of machine learning research 9(11) (2008)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01592-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01592-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01592-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T09:24:09Z","timestamp":1734341049000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01592-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12]]},"references-count":41,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["1592"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01592-6","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"type":"print","value":"0942-4962"},{"type":"electronic","value":"1432-1882"}],"subject":[],"published":{"date-parts":[[2024,12]]},"assertion":[{"value":"21 August 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 November 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there are no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"382"}}