{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:51:59Z","timestamp":1782201119944,"version":"3.54.5"},"reference-count":86,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T00:00:00Z","timestamp":1779408000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T00:00:00Z","timestamp":1779408000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100021171","name":"Guangdong Basic and Applied Basic Research Foundation","doi-asserted-by":"crossref","award":["No.2024A1515010456"],"award-info":[{"award-number":["No.2024A1515010456"]}],"id":[{"id":"10.13039\/501100021171","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Shenzhen Science and Technology Program","award":["No. JCYJ20240813111301003, JCYJ20230807090103008"],"award-info":[{"award-number":["No. JCYJ20240813111301003, JCYJ20230807090103008"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s00371-026-04522-x","type":"journal-article","created":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T17:12:47Z","timestamp":1779469967000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Frequency-enhanced diffusion models: curriculum-guided semantic alignment for zero-shot skeleton action recognition"],"prefix":"10.1007","volume":"42","author":[{"given":"Yuxi","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhengbo","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingyu","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiyu","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhigang","family":"Tu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,22]]},"reference":[{"key":"4522_CR1","doi-asserted-by":"publisher","first-page":"17165","DOI":"10.1007\/s11042-018-7108-9","volume":"78","author":"R Singh","year":"2019","unstructured":"Singh, R., Kushwaha, A.K.S., Srivastava, R.: Multi-view recognition system for human activity based on multiple features for video surveillance system. Multimed. Tools Appl. 78, 17165\u201317196 (2019)","journal-title":"Multimed. Tools Appl."},{"issue":"1","key":"4522_CR2","doi-asserted-by":"publisher","first-page":"70011","DOI":"10.1002\/cav.70011","volume":"36","author":"X Hong-qin","year":"2025","unstructured":"Hong-qin, X., Yuan-yuan, Z.: Advanced gesture recognition method based on fractional Fourier transform and relevance vector machine for smart home appliances. Comput. Anim. Virtual Worlds 36(1), 70011 (2025)","journal-title":"Comput. Anim. Virtual Worlds"},{"key":"4522_CR3","doi-asserted-by":"crossref","unstructured":"Yang, Y., Zhou, J., Hu, W., Tu, Z.: End-to-end pose-action recognition via implicit pose encoding and multi-scale skeleton modeling: Y. yang et al. Vis. Comput. 41, 9337\u20139353 (2025)","DOI":"10.1007\/s00371-025-03930-9"},{"issue":"5","key":"4522_CR4","doi-asserted-by":"publisher","first-page":"2774","DOI":"10.1109\/TSMC.2019.2916896","volume":"51","author":"K Aouaidjia","year":"2019","unstructured":"Aouaidjia, K., Sheng, B., Li, P., Kim, J., Feng, D.D.: Efficient body motion quantification and similarity evaluation using 3-d joints skeleton coordinates. IEEE Trans. Syst. Man Cybern. Syst. 51(5), 2774\u20132788 (2019)","journal-title":"IEEE Trans. Syst. Man Cybern. Syst."},{"issue":"4","key":"4522_CR5","doi-asserted-by":"publisher","first-page":"366","DOI":"10.1016\/j.vrih.2022.08.010","volume":"5","author":"X Hu","year":"2023","unstructured":"Hu, X., Bao, X., Wei, G., Li, Z.: Human-pose estimation based on weak supervision. Virtual Real. Intell. Hardw. 5(4), 366\u2013377 (2023)","journal-title":"Virtual Real. Intell. Hardw."},{"issue":"3","key":"4522_CR6","doi-asserted-by":"publisher","first-page":"807","DOI":"10.1109\/TCSVT.2016.2628339","volume":"28","author":"Y Hou","year":"2016","unstructured":"Hou, Y., Li, Z., Wang, P., Li, W.: Skeleton optical spectra-based action recognition using convolutional neural networks. IEEE Trans. Circuits Syst. Video Technol. 28(3), 807\u2013811 (2016)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"4522_CR7","doi-asserted-by":"crossref","unstructured":"Chi, H.-G., Ha, M.H., Chi, S., Lee, S.W., Huang, Q., Ramani, K.: Infogcn: representation learning for human skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20186\u201320196 (2022)","DOI":"10.1109\/CVPR52688.2022.01955"},{"issue":"5","key":"4522_CR8","doi-asserted-by":"publisher","first-page":"2191","DOI":"10.1007\/s00371-022-02473-7","volume":"39","author":"Z-X Qiu","year":"2023","unstructured":"Qiu, Z.-X., Zhang, H.-B., Deng, W.-M., Du, J.-X., Lei, Q., Zhang, G.-L.: Effective skeleton topology and semantics-guided adaptive graph convolution network for action recognition. Vis. Comput. 39(5), 2191\u20132203 (2023)","journal-title":"Vis. Comput."},{"key":"4522_CR9","doi-asserted-by":"crossref","unstructured":"Liu, H., Liu, Y., Chen, Y., Yuan, C., Li, B., Hu, W.: Transkeleton: hierarchical spatial-temporal transformer for skeleton-based action recognition. IEEE Trans Circuits Syst. Video Technol. 33, 4137\u20134148 (2023)","DOI":"10.1109\/TCSVT.2023.3240472"},{"key":"4522_CR10","doi-asserted-by":"crossref","unstructured":"Wu, W., Zheng, C., Yang, Z., Chen, C., Das, S., Lu, A.: Frequency guidance matters: Skeletal action recognition by frequency-aware mixed transformer. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp. 4660\u20134669 (2024)","DOI":"10.1145\/3664647.3681009"},{"key":"4522_CR11","doi-asserted-by":"crossref","unstructured":"Zhao, J., Dai, J., Zhou, F., Pan, J., Xu, H.: Dual-path spatio-temporal mamba for skeleton-based action recognition: J. Zhao et al. Vis. Comput. 41, 6507\u20136519 (2025)","DOI":"10.1007\/s00371-025-03950-5"},{"key":"4522_CR12","doi-asserted-by":"crossref","unstructured":"Xie, Z., Chen, J., Wang, Y., Xie, B.: Enhanced fine-grained relearning for skeleton-based action recognition. Vis. Comput. 41, 7983\u20137995 (2025)","DOI":"10.1007\/s00371-025-03850-8"},{"key":"4522_CR13","doi-asserted-by":"publisher","first-page":"7335","DOI":"10.1109\/TIP.2025.3627418","volume":"34","author":"Z Tu","year":"2025","unstructured":"Tu, Z., Zhang, Z., Gong, J., Yuan, J., Du, B.: Informative sample selection model for skeleton-based action recognition with limited training samples. IEEE Trans. Image Process. 34, 7335\u20137346 (2025)","journal-title":"IEEE Trans. Image Process."},{"key":"4522_CR14","doi-asserted-by":"crossref","unstructured":"Do, J., Kim, M.: Bridging the skeleton-text modality gap: diffusion-powered modality alignment for zero-shot skeleton-based action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12757\u201312768 (2025)","DOI":"10.1109\/ICCV51701.2025.01185"},{"key":"4522_CR15","doi-asserted-by":"crossref","unstructured":"Tsai, H., Huang, Y.-H., Salakhutdinov, L.-K., R.: Learning robust visual-semantic embeddings. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3571\u20133580 (2017)","DOI":"10.1109\/ICCV.2017.386"},{"key":"4522_CR16","doi-asserted-by":"crossref","unstructured":"Gupta, P., Sharma, D., Sarvadevabhatla, R.K.: Syntactically guided generative embeddings for zero-shot skeleton action recognition. In: 2021 IEEE International Conference on Image Processing (ICIP), pp. 439\u2013443. IEEE (2021)","DOI":"10.1109\/ICIP42928.2021.9506179"},{"key":"4522_CR17","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Qiang, W., Rao, A., Lin, N., Su, B., Wang, J.: Zero-shot skeleton-based action recognition via mutual information estimation and maximization. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5302\u20135310 (2023)","DOI":"10.1145\/3581783.3611888"},{"key":"4522_CR18","doi-asserted-by":"crossref","unstructured":"Li, M.-Z., Jia, Z., Zhang, Z., Ma, Z., Wang, L.: Multi-semantic fusion model for generalized zero-shot skeleton-based action recognition. In: International Conference on Image and Graphics, pp. 68\u201380. Springer (2023)","DOI":"10.1007\/978-3-031-46305-1_6"},{"key":"4522_CR19","doi-asserted-by":"crossref","unstructured":"Chen, Y., Guo, J., He, T., Lu, X., Wang, L.: Fine-grained side information guided dual-prompts for zero-shot skeleton action recognition. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp. 778\u2013786 (2024)","DOI":"10.1145\/3664647.3681196"},{"key":"4522_CR20","doi-asserted-by":"crossref","unstructured":"Zhu, A., Ke, Q., Gong, M., Bailey, J.: Part-aware unified representation of language and skeleton for zero-shot action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18761\u201318770 (2024)","DOI":"10.1109\/CVPR52733.2024.01775"},{"key":"4522_CR21","doi-asserted-by":"crossref","unstructured":"Li, S.-W., Wei, Z.-X., Chen, W.-J., Yu, Y.-H., Yang, C.-Y., Hsu, J.Y.-j.: Sa-dvae: improving zero-shot skeleton-based action recognition by disentangled variational autoencoders. In: European Conference on Computer Vision, pp. 447\u2013462. Springer (2025)","DOI":"10.1007\/978-3-031-72640-8_25"},{"key":"4522_CR22","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4522_CR23","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"4522_CR24","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Xu, L., Peng, D., Rahmani, H., Liu, J.: Diff-tracker: text-to-image diffusion models are unsupervised trackers. In: European Conference on Computer Vision, pp. 319\u2013337. Springer (2024)","DOI":"10.1007\/978-3-031-73390-1_19"},{"key":"4522_CR25","doi-asserted-by":"crossref","unstructured":"Wu, W., Guo, Z., Chen, C., Xue, H., Lu, A.: Frequency-semantic enhanced variational autoencoder for zero-shot skeleton-based action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 11122\u201311131 (2025)","DOI":"10.1109\/ICCV51701.2025.01035"},{"key":"4522_CR26","doi-asserted-by":"publisher","first-page":"23495","DOI":"10.52202\/068431-1707","volume":"35","author":"C Si","year":"2022","unstructured":"Si, C., Yu, W., Zhou, P., Zhou, Y., Wang, X., Yan, S.: Inception transformer. Adv. Neural. Inf. Process. Syst. 35, 23495\u201323509 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4522_CR27","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., Wang, O.: The unreasonable effectiveness of deep features as a perceptual metric. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 586\u2013595 (2018)","DOI":"10.1109\/CVPR.2018.00068"},{"key":"4522_CR28","unstructured":"Rahaman, N., Baratin, A., Arpit, D., Draxler, F., Lin, M., Hamprecht, F., Bengio, Y., Courville, A.: On the spectral bias of neural networks. In: International Conference on Machine Learning, pp. 5301\u20135310. PMLR (2019)"},{"key":"4522_CR29","doi-asserted-by":"crossref","unstructured":"Shahroudy, A., Liu, J., Ng, T.-T., Wang, G.: Ntu rgb+ d: a large scale dataset for 3d human activity analysis. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1010\u20131019 (2016)","DOI":"10.1109\/CVPR.2016.115"},{"issue":"10","key":"4522_CR30","doi-asserted-by":"publisher","first-page":"2684","DOI":"10.1109\/TPAMI.2019.2916873","volume":"42","author":"J Liu","year":"2019","unstructured":"Liu, J., Shahroudy, A., Perez, M., Wang, G., Duan, L.-Y., Kot, A.C.: Ntu rgb+ d 120: a large-scale benchmark for 3d human activity understanding. IEEE Trans. Pattern Anal. Mach. Intell. 42(10), 2684\u20132701 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4522_CR31","doi-asserted-by":"crossref","unstructured":"Liu, C., Hu, Y., Li, Y., Song, S., Liu, J.: Pku-mmd: A large scale benchmark for continuous multi-modal human action understanding. arXiv preprint arXiv:1703.07475 (2017)","DOI":"10.1145\/3132734.3132739"},{"key":"4522_CR32","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"4522_CR33","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P.: The kinetics human action video dataset (2017). (arXiv preprint)"},{"key":"4522_CR34","doi-asserted-by":"crossref","unstructured":"Schonfeld, E., Ebrahimi, S., Sinha, S., Darrell, T., Akata, Z.: Generalized zero-and few-shot learning via aligned variational autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8247\u20138255 (2019)","DOI":"10.1109\/CVPR.2019.00844"},{"key":"4522_CR35","doi-asserted-by":"crossref","unstructured":"Gupta, P., Sharma, D., Sarvadevabhatla, R.K.: Syntactically guided generative embeddings for zero-shot skeleton action recognition. In: 2021 IEEE International Conference on Image Processing (ICIP), pp. 439\u2013443. IEEE (2021)","DOI":"10.1109\/ICIP42928.2021.9506179"},{"key":"4522_CR36","doi-asserted-by":"crossref","unstructured":"Li, M.-Z., Jia, Z., Zhang, Z., Ma, Z., Wang, L.: Multi-semantic fusion model for generalized zero-shot skeleton-based action recognition. In: International Conference on Image and Graphics, pp. 68\u201380. Springer (2023)","DOI":"10.1007\/978-3-031-46305-1_6"},{"key":"4522_CR37","doi-asserted-by":"crossref","unstructured":"Li, S.-W., Wei, Z.-X., Chen, W.-J., Yu, Y.-H., Yang, C.-Y., Hsu, J.Y.-j.: Sa-dvae: Improving zero-shot skeleton-based action recognition by disentangled variational autoencoders. arXiv preprint arXiv:2407.13460 (2024)","DOI":"10.1007\/978-3-031-72640-8_25"},{"key":"4522_CR38","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Qiang, W., Rao, A., Lin, N., Su, B., Wang, J.: Zero-shot skeleton-based action recognition via mutual information estimation and maximization. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5302\u20135310 (2023)","DOI":"10.1145\/3581783.3611888"},{"key":"4522_CR39","doi-asserted-by":"crossref","unstructured":"Zhu, A., Ke, Q., Gong, M., Bailey, J.: Part-aware unified representation of language and skeleton for zero-shot action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18761\u201318770 (2024)","DOI":"10.1109\/CVPR52733.2024.01775"},{"key":"4522_CR40","doi-asserted-by":"crossref","unstructured":"Chen, Y., Guo, J., He, T., Wang, L.: Fine-grained side information guided dual-prompts for zero-shot skeleton action recognition (2024). (arXiv preprint)","DOI":"10.1145\/3664647.3681196"},{"key":"4522_CR41","unstructured":"Kuang, J., Wang, H., Han, C., Gui, J.: Zero-shot skeleton-based action recognition with dual visual-text alignment (2024). (arXiv preprint)"},{"key":"4522_CR42","doi-asserted-by":"crossref","unstructured":"Xu, H., Gao, Y., Li, J., Gao, X.: An information compensation framework for zero-shot skeleton-based action recognition. arXiv preprint arXiv:2406.00639 (2024)","DOI":"10.1109\/TMM.2025.3543004"},{"key":"4522_CR43","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR (2020)"},{"key":"4522_CR44","unstructured":"Brown, T.B.: Language models are few-shot learners (2020). (arXiv preprint)"},{"key":"4522_CR45","unstructured":"Tevet, G., Raab, S., Abu-Horany, B., Cohen-Or, D.: Human motion diffusion model. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"4522_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, M., Cai, Z., Pan, L., Hong, F., Guo, X., Yang, L., Liu, Z.: Motiondiffuse: Text-driven human motion generation with diffusion model. IEEE Trans. Pattern Anal. Mach. Intell. PAMI (2024)","DOI":"10.1109\/TPAMI.2024.3355414"},{"key":"4522_CR47","doi-asserted-by":"crossref","unstructured":"Zhang, P., Lan, C., Xing, J., Zeng, W., Xue, J., Zheng, N.: View adaptive recurrent neural networks for high performance human action recognition from skeleton data. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2117\u20132126 (2017)","DOI":"10.1109\/ICCV.2017.233"},{"key":"4522_CR48","doi-asserted-by":"crossref","unstructured":"Liu, J., Shahroudy, A., Xu, D., Wang, G.: Spatio-temporal lstm with trust gates for 3d human action recognition. In: Computer Vision-ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part III 14, pp. 816\u2013833. Springer (2016)","DOI":"10.1007\/978-3-319-46487-9_50"},{"key":"4522_CR49","unstructured":"Cai, D., Kang, Y., Yao, A., Chen, Y.: Ske2grid: skeleton-to-grid representation learning for action recognition. In: International Conference on Machine Learning (2023)"},{"key":"4522_CR50","doi-asserted-by":"crossref","unstructured":"Duan, H., Zhao, Y., Chen, K., Lin, D., Dai, B.: Revisiting skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2969\u20132978 (2022)","DOI":"10.1109\/CVPR52688.2022.00298"},{"key":"4522_CR51","doi-asserted-by":"crossref","unstructured":"Chen, Y., Zhang, Z., Yuan, C., Li, B., Deng, Y., Hu, W.: Channel-wise topology refinement graph convolution for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13359\u201313368 (2021)","DOI":"10.1109\/ICCV48922.2021.01311"},{"key":"4522_CR52","doi-asserted-by":"crossref","unstructured":"Chi, H.-G., Ha, M.H., Chi, S., Lee, S.W., Huang, Q., Ramani, K.: Infogcn: representation learning for human skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20186\u201320196 (2022)","DOI":"10.1109\/CVPR52688.2022.01955"},{"key":"4522_CR53","doi-asserted-by":"crossref","unstructured":"Liu, Z., Zhang, H., Chen, Z., Wang, Z., Ouyang, W.: Disentangling and unifying graph convolutions for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 143\u2013152 (2020)","DOI":"10.1109\/CVPR42600.2020.00022"},{"key":"4522_CR54","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Yan, X., Cheng, Z.-Q., Yan, Y., Dai, Q., Hua, X.-S.: Blockgcn: Redefine topology awareness for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2049\u20132058 (2024)","DOI":"10.1109\/CVPR52733.2024.00200"},{"key":"4522_CR55","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"4522_CR56","doi-asserted-by":"crossref","unstructured":"Cheng, K., Zhang, Y., He, X., Chen, W., Cheng, J., Lu, H.: Skeleton-based action recognition with shift graph convolutional network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 183\u2013192 (2020)","DOI":"10.1109\/CVPR42600.2020.00026"},{"key":"4522_CR57","doi-asserted-by":"crossref","unstructured":"Pang, Y., Ke, Q., Rahmani, H., Bailey, J., Liu, J.: Igformer: Interaction graph transformer for skeleton-based human interaction recognition. In: European Conference on Computer Vision, pp. 605\u2013622. Springer (2022)","DOI":"10.1007\/978-3-031-19806-9_35"},{"key":"4522_CR58","doi-asserted-by":"crossref","unstructured":"Do, J., Kim, M.: Skateformer: Skeletal-temporal transformer for human action recognition, (2024). (arXiv preprint)","DOI":"10.1007\/978-3-031-72940-9_23"},{"key":"4522_CR59","doi-asserted-by":"crossref","unstructured":"Zhao, J., Ning, K., Zhou, F., Pan, J., Xu, H., Dai, J.:Multi-level fusion tokens for enhanced self-supervised skeleton-based action recognition: J Zhao, et al. Vis. Comput. 42(1), 37 (2026)","DOI":"10.1007\/s00371-025-04285-x"},{"issue":"8","key":"4522_CR60","doi-asserted-by":"publisher","first-page":"5733","DOI":"10.1007\/s00371-023-03132-1","volume":"40","author":"S Sun","year":"2024","unstructured":"Sun, S., Jia, Z., Zhu, Y., Liu, G., Yu, Z.: Decoupled spatio-temporal grouping transformer for skeleton-based action recognition. Vis. Comput. 40(8), 5733\u20135745 (2024)","journal-title":"Vis. Comput."},{"issue":"10","key":"4522_CR61","doi-asserted-by":"publisher","first-page":"4501","DOI":"10.1007\/s00371-022-02603-1","volume":"39","author":"J Zhang","year":"2023","unstructured":"Zhang, J., Xie, W., Wang, C., Tu, R., Tu, Z.: Graph-aware transformer for skeleton-based action recognition. Vis. Comput. 39(10), 4501\u20134512 (2023)","journal-title":"Vis. Comput."},{"key":"4522_CR62","doi-asserted-by":"crossref","unstructured":"Yao, J., Chen, J., Niu, L., Sheng, B.: Scene-aware human pose generation using transformer. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 2847\u20132855 (2023)","DOI":"10.1145\/3581783.3612439"},{"key":"4522_CR63","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4522_CR64","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"issue":"3","key":"4522_CR65","doi-asserted-by":"publisher","first-page":"70040","DOI":"10.1002\/cav.70040","volume":"36","author":"J Peng","year":"2025","unstructured":"Peng, J., Liu, Z., Lin, J., He, G.: Precise motion inbetweening via bidirectional autoregressive diffusion models. Comput. Anim. Virtual Worlds 36(3), 70040 (2025)","journal-title":"Comput. Anim. Virtual Worlds"},{"key":"4522_CR66","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: Convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-assisted intervention-MICCAI 2015: 18th International Conference, Munich, Germany, October 5\u20139, 2015, Proceedings, Part III 18, pp. 234\u2013241. Springer (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"4522_CR67","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4195\u20134205 (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"4522_CR68","unstructured":"Chen, Z., Duan, Y., Wang, W., He, J., Lu, T., Dai, J., Qiao, Y.: Vision transformer adapter for dense predictions (2022). ArXiv:abs\/2205.08534"},{"issue":"1","key":"4522_CR69","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1109\/T-C.1974.223784","volume":"100","author":"N Ahmed","year":"1974","unstructured":"Ahmed, N., Natarajan, T., Rao, K.R.: Discrete cosine transform. IEEE Trans. Comput. 100(1), 90\u201393 (1974)","journal-title":"IEEE Trans. Comput."},{"issue":"3","key":"4522_CR70","doi-asserted-by":"publisher","first-page":"505","DOI":"10.1109\/83.826787","volume":"9","author":"A Polesel","year":"2000","unstructured":"Polesel, A., Ramponi, G., Mathews, V.J.: Image enhancement via adaptive unsharp masking. IEEE Trans. Image Process. 9(3), 505\u2013510 (2000)","journal-title":"IEEE Trans. Image Process."},{"key":"4522_CR71","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Zhou, C., Tu, Z.: Distilling inter-class distance for semantic segmentation (2022). (arXiv preprint)","DOI":"10.24963\/ijcai.2022\/235"},{"issue":"6","key":"4522_CR72","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou, J., Yu, B., Maybank, S.J., Tao, D.: Knowledge distillation: a survey. Int. J. Comput. Vis. 129(6), 1789\u20131819 (2021)","journal-title":"Int. J. Comput. Vis."},{"key":"4522_CR73","doi-asserted-by":"crossref","unstructured":"Choi, J., Lee, J., Shin, C., Kim, S., Kim, H., Yoon, S.: Perception prioritized training of diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11472\u201311481 (2022)","DOI":"10.1109\/CVPR52688.2022.01118"},{"key":"4522_CR74","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: International Conference on Learning Representations (2020)"},{"key":"4522_CR75","unstructured":"Loshchilov, I., Hutter, F.: SGDR: stochastic gradient descent with warm restarts. In: 5th International Conference on Learning Representations, ICLR 2017, Toulon, France, April 24\u201326, 2017, Conference Track Proceedings. OpenReview.net (2017). https:\/\/openreview.net\/forum?id=Skq89Scxx"},{"key":"4522_CR76","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: Proceedings of the 26th Annual International Conference on Machine Learning (ICML), pp. 41\u201348 (2009)","DOI":"10.1145\/1553374.1553380"},{"key":"4522_CR77","doi-asserted-by":"crossref","unstructured":"Tsai, H., Huang, Y.-H., Salakhutdinov, L.-K., R.: Learning robust visual-semantic embeddings. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3571\u20133580 (2017)","DOI":"10.1109\/ICCV.2017.386"},{"key":"4522_CR78","doi-asserted-by":"crossref","unstructured":"Wray, M., Larlus, D., Csurka, G., Damen, D.: Fine-grained action retrieval through multiple parts-of-speech embeddings. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 450\u2013459 (2019)","DOI":"10.1109\/ICCV.2019.00054"},{"key":"4522_CR79","unstructured":"Paszke, A., Gross, S., Chintala, S., Chanan, G., Yang, E., DeVito, Z., Lin, Z., Desmaison, A., Antiga, L., Lerer, A.: Automatic differentiation in pytorch (2017)"},{"key":"4522_CR80","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization (2017). (arXiv preprint)"},{"key":"4522_CR81","unstructured":"Loshchilov, I., Hutter, F.: Sgdr: Stochastic gradient descent with warm restarts (2016). (arXiv preprint)"},{"key":"4522_CR82","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F.L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S.: Gpt-4 technical report (2023). (arXiv preprint)"},{"key":"4522_CR83","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"4522_CR84","unstructured":"Ilharco, G., Wortsman, M., Carlini, N., Taori, R., Dave, A., Shankar, V., Namkoong, H., Miller, J., Hajishirzi, H., Farhadi, A., Schmidt, L.: Open Clip (2021)"},{"key":"4522_CR85","unstructured":"Frome, A., Corrado, G.S., Shlens, J., Bengio, S., Dean, J., Ranzato, M., Mikolov, T.: Devise: A deep visual-semantic embedding model. In: Advances in Neural Information Processing Systems, vol. 26 (2013)"},{"key":"4522_CR86","doi-asserted-by":"crossref","unstructured":"Cutler, R., Davis, L.S.: Robust real-time periodic motion detection, analysis, and applications. IEEE Trans. Pattern Anal. Mach. Intell. 22(8), 781\u2013796 (2002)","DOI":"10.1109\/34.868681"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04522-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-026-04522-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-026-04522-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T07:46:16Z","timestamp":1782200776000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-026-04522-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,22]]},"references-count":86,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["4522"],"URL":"https:\/\/doi.org\/10.1007\/s00371-026-04522-x","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5,22]]},"assertion":[{"value":"8 January 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 May 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"310"}}