{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,18]],"date-time":"2025-09-18T04:12:45Z","timestamp":1758168765511,"version":"3.44.0"},"reference-count":106,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2025,8,30]],"date-time":"2025-08-30T00:00:00Z","timestamp":1756512000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,30]],"date-time":"2025-08-30T00:00:00Z","timestamp":1756512000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["No.61673396","No.61673396","No.61673396","No.61673396"],"award-info":[{"award-number":["No.61673396","No.61673396","No.61673396","No.61673396"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cluster Comput"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s10586-025-05262-8","type":"journal-article","created":{"date-parts":[[2025,8,30]],"date-time":"2025-08-30T10:41:50Z","timestamp":1756550510000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Kdhiera: boosting self-supervised masked video modeling via hierarchical knowledge distillation"],"prefix":"10.1007","volume":"28","author":[{"given":"Yunlong","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Liang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingwen","family":"Shao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qian","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,8,30]]},"reference":[{"key":"5262_CR1","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"5262_CR2","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"5262_CR3","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.: Beit: Bert pre-training of image transformers. arXiv preprint arXiv:2106.08254 (2021)"},{"key":"5262_CR4","first-page":"552","volume":"37","author":"X Dong","year":"2023","unstructured":"Dong, X., Bao, J., Zhang, T., Chen, D., Zhang, W., Yuan, L., Chen, D., Wen, F., Yu, N., Guo, B.: Peco: perceptual codebook for Bert pre-training of vision transformers. Proc. AAAI Confer. Artif. Intell. 37, 552\u2013560 (2023)","journal-title":"Proc. AAAI Confer. Artif. Intell."},{"key":"5262_CR5","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"5262_CR6","doi-asserted-by":"crossref","unstructured":"Wang, R., Chen, D., Wu, Z., Chen, Y., Dai, X., Liu, M., Jiang, Y.G., Zhou, L., Yuan, L.: Bevt: Bert pretraining of video transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14733\u201314743 (2022)","DOI":"10.1109\/CVPR52688.2022.01432"},{"key":"5262_CR7","first-page":"10078","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., Wang, L.: Videomae: masked autoencoders are data-efficient learners for self-supervised video pre-training. Adv. Neural. Inf. Process. Syst. 35, 10078\u201310093 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR8","first-page":"35946","volume":"35","author":"C Feichtenhofer","year":"2022","unstructured":"Feichtenhofer, C., Li, Y., He, K., et al.: Masked autoencoders as spatiotemporal learners. Adv. Neural. Inf. Process. Syst. 35, 35946\u201335958 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR9","unstructured":"Zhou, J., Wei, C., Wang, H., Shen, W., Xie, C., Yuille, A., Kong, T.: ibot: image bert pre-training with online tokenizer. arXiv preprint arXiv:2111.07832 (2021)"},{"key":"5262_CR10","doi-asserted-by":"crossref","unstructured":"Wang, R., Chen, D., Wu, Z., Chen, Y., Dai, X., Liu, M., Yuan, L., Jiang, Y.G.: Masked video distillation: Rethinking masked feature modeling for self-supervised video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6312\u20136322 (2023)","DOI":"10.1109\/CVPR52729.2023.00611"},{"key":"5262_CR11","doi-asserted-by":"crossref","unstructured":"Qing, Z., Zhang, S., Huang, Z., Wang, X., Wang, Y., Lv, Y., Gao, C., Sang, N.: Mar: masked autoencoders for efficient action recognition. IEEE Trans. Multimedia 26, 218\u2013233 (2023)","DOI":"10.1109\/TMM.2023.3263288"},{"key":"5262_CR12","first-page":"19997","volume":"35","author":"L Huang","year":"2022","unstructured":"Huang, L., You, S., Zheng, M., Wang, F., Qian, C., Yamasaki, T.: Green hierarchical vision transformer for masked image modeling. Adv. Neural. Inf. Process. Syst. 35, 19997\u201320010 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR13","unstructured":"Li, X., Wang, W., Yang, L., Yang, J.: Uniform masking: enabling MAE pre-training for pyramid-based vision transformers with locality. arXiv preprint arXiv:2205.10063 (2022)"},{"key":"5262_CR14","unstructured":"Ryali, C., Hu, Y.T., Bolya, D., Wei, C., Fan, H., Huang, P.Y., Aggarwal, V., Chowdhury, A., Poursaeed, O., Hoffman, J., et\u00a0al.: Hiera: a hierarchical vision transformer without the bells-and-whistles. In: International Conference on Machine Learning (PMLR), pp. 29441\u201329454 (2023)"},{"key":"5262_CR15","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning (PMLR), pp. 8748\u20138763 (2021)"},{"key":"5262_CR16","doi-asserted-by":"crossref","unstructured":"Li, K., Wang, Y., Li, Y., Wang, Y., He, Y., Wang, L., Qiao, Y.: Unmasked teacher: Towards training-efficient video foundation models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19948\u201319960 (2023)","DOI":"10.1109\/ICCV51070.2023.01826"},{"key":"5262_CR17","unstructured":"Hou, Z., Sun, F., Chen, Y.K., Xie, Y., Kung, S.Y.: Milan: masked image pretraining on language assisted representation. arXiv preprint arXiv:2208.06049 (2022)"},{"key":"5262_CR18","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P., et\u00a0al.: The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"5262_CR19","doi-asserted-by":"crossref","unstructured":"Goyal, R., Ebrahimi\u00a0Kahou, S., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., Mueller-Freitag, M., et\u00a0al.: The \u201csomething something\u201d video database for learning and evaluating visual common sense. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"5262_CR20","doi-asserted-by":"crossref","unstructured":"Gu, C., Sun, C., Ross, D.A., Vondrick, C., Pantofaru, C., Li, Y., Vijayanarasimhan, S., Toderici, G., Ricco, S., Sukthankar, R., et\u00a0al.: Ava: a video dataset of spatio-temporally localized atomic visual actions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6047\u20136056 (2018)","DOI":"10.1109\/CVPR.2018.00633"},{"key":"5262_CR21","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C.: X3d: expanding architectures for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 203\u2013213 (2020)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"5262_CR22","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"5262_CR23","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., Han, S.: Tsm: temporal shift module for efficient video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7083\u20137093 (2019)","DOI":"10.1109\/ICCV.2019.00718"},{"issue":"11","key":"5262_CR24","doi-asserted-by":"publisher","first-page":"2740","DOI":"10.1109\/TPAMI.2018.2868668","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., Van Gool, L.: Temporal segment networks for action recognition in videos. IEEE Trans. Pattern Anal. Mach. Intell. 41(11), 2740\u20132755 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5262_CR25","doi-asserted-by":"crossref","unstructured":"Dong, X., Bao, J., Chen, D., Zhang, W., Yu, N., Yuan, L., Chen, D., Guo, B.: Cswin transformer: a general vision transformer backbone with cross-shaped windows. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12124\u201312134 (2022)","DOI":"10.1109\/CVPR52688.2022.01181"},{"key":"5262_CR26","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"5262_CR27","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: Vivit: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6836\u20136846 (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"5262_CR28","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: ICML, vol.\u00a02 , p.\u00a04 (2021)"},{"key":"5262_CR29","first-page":"12493","volume":"34","author":"M Patrick","year":"2021","unstructured":"Patrick, M., Campbell, D., Asano, Y., Misra, I., Metze, F., Feichtenhofer, C., Vedaldi, A., Henriques, J.F.: Keeping your eye on the ball: trajectory attention in video transformers. Adv. Neural. Inf. Process. Syst. 34, 12493\u201312506 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR30","first-page":"12786","volume":"34","author":"M Ryoo","year":"2021","unstructured":"Ryoo, M., Piergiovanni, A., Arnab, A., Dehghani, M., Angelova, A.: Tokenlearner: adaptive space-time tokenization for videos. Adv. Neural. Inf. Process. Syst. 34, 12786\u201312797 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR31","first-page":"19594","volume":"34","author":"A Bulat","year":"2021","unstructured":"Bulat, A., Perez Rua, J.M., Sudhakaran, S., Martinez, B., Tzimiropoulos, G.: Space-time mixing attention for video transformer. Adv. Neural. Inf. Process. Syst. 34, 19594\u201319607 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR32","doi-asserted-by":"crossref","unstructured":"Fan, H., Xiong, B., Mangalam, K., Li, Y., Yan, Z., Malik, J., Feichtenhofer, C.: Multiscale vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6824\u20136835 (2021)","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"5262_CR33","doi-asserted-by":"crossref","unstructured":"Li, Y., Wu, C.Y., Fan, H., Mangalam, K., Xiong, B., Malik, J., Feichtenhofer, C.: Mvitv2: improved multiscale vision transformers for classification and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"5262_CR34","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., Cao, Y., Wei, Y., Zhang, Z., Lin, S., Hu, H.: Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3202\u20133211 (2022)","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"5262_CR35","unstructured":"Wang, R., Wu, Z., Chen, D., Chen, Y., Dai, X., Liu, M., Zhou, L., Yuan, L., Jiang, Y.G.: Video mobile-former: video recognition with efficient global spatial-temporal modeling. arXiv preprint arXiv:2208.12257 (2022)"},{"key":"5262_CR36","unstructured":"Li, K., Wang, Y., Gao, P., Song, G., Liu, Y., Li, H., Qiao, Y.: Uniformer: unified transformer for efficient spatiotemporal representation learning. arXiv preprint arXiv:2201.04676 (2022)"},{"key":"5262_CR37","unstructured":"Li, K., Wang, Y., He, Y., Li, Y., Wang, Y., Wang, L., Qiao, Y.: Uniformerv2: spatiotemporal learning by arming image vits with video uniformer. arXiv preprint arXiv:2211.09552 (2022)"},{"key":"5262_CR38","unstructured":"Ermolov, A., Siarohin, A., Sangineto, E., Sebe, N.: Whitening for self-supervised representation learning. In: International Conference on Machine Learning (PMLR), pp. 3015\u20133024 (2021)"},{"key":"5262_CR39","unstructured":"Goyal, P., Caron, M., Lefaudeux, B., Xu, M., Wang, P., Pai, V., Singh, M., Liptchinsky, V., Misra, I., Joulin, A., et\u00a0al.: Self-supervised pretraining of visual features in the wild. arXiv preprint arXiv:2103.01988 (2021)"},{"key":"5262_CR40","unstructured":"Zbontar, J., Jing, L., Misra, I., LeCun, Y., Deny, S.: Barlow twins: self-supervised learning via redundancy reduction. In: International Conference on Machine Learning (PMLR), pp. 12310\u201312320 (2021)"},{"key":"5262_CR41","unstructured":"Chen, M., Radford, A., Child, R., Wu, J., Jun, H., Luan, D., Sutskever, I.: Generative pretraining from pixels. In: International Conference on Machine Learning (PMLR), pp. 1691\u20131703 (2020)"},{"key":"5262_CR42","first-page":"13165","volume":"34","author":"Z Li","year":"2021","unstructured":"Li, Z., Chen, Z., Yang, F., Li, W., Zhu, Y., Zhao, C., Deng, R., Wu, L., Zhao, R., Tang, M., et al.: Mst: masked self-supervised transformer for visual representation. Adv. Neural. Inf. Process. Syst. 34, 13165\u201313176 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR43","doi-asserted-by":"crossref","unstructured":"Ma, X., Liu, C., Xie, C., Ye, L., Deng, Y., Ji, X.: Disjoint masking with joint distillation for efficient masked image modeling. IEEE Trans. Multimedia 26, 3077\u20133087 (2023)","DOI":"10.1109\/TMM.2023.3306840"},{"issue":"1","key":"5262_CR44","doi-asserted-by":"publisher","first-page":"208","DOI":"10.1007\/s11263-023-01852-4","volume":"132","author":"X Chen","year":"2024","unstructured":"Chen, X., Ding, M., Wang, X., Xin, Y., Mo, S., Wang, Y., Han, S., Luo, P., Zeng, G., Wang, J.: Context autoencoder for self-supervised representation learning. Int. J. Comput. Vis. 132(1), 208\u2013223 (2024)","journal-title":"Int. J. Comput. Vis."},{"key":"5262_CR45","unstructured":"Ramesh, A., Pavlov, M., Goh, G., Gray, S., Voss, C., Radford, A., Chen, M., Sutskever, I.: Zero-shot text-to-image generation. In: International Conference on Machine Learning (PMLR), pp. 8821\u20138831 (2021)"},{"key":"5262_CR46","unstructured":"Rolfe, J.T.: Discrete variational autoencoders. arXiv preprint arXiv:1609.02200 (2016)"},{"key":"5262_CR47","unstructured":"Van Den\u00a0Oord, A., Vinyals, O.: Neural discrete representation learning. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"5262_CR48","doi-asserted-by":"crossref","unstructured":"Wei, C., Fan, H., Xie, S., Wu, C.Y., Yuille, A., Feichtenhofer, C.: Masked feature prediction for self-supervised visual pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14668\u201314678 (2022)","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"5262_CR49","unstructured":"Baevski, A., Hsu, W.N., Xu, Q., Babu, A., Gu, J., Auli, M.: Data2vec: a general framework for self-supervised learning in speech, vision and language. In: International Conference on Machine Learning (PMLR), pp. 1298\u20131312 (2022)"},{"key":"5262_CR50","doi-asserted-by":"crossref","unstructured":"Dong, X., Bao, J., Zhang, T., Chen, D., Zhang, W., Yuan, L., Chen, D., Wen, F., Yu, N.: Bootstrapped masked autoencoders for vision bert pretraining. In: European Conference on Computer Vision, pp. 247\u2013264. Springer (2022)","DOI":"10.1007\/978-3-031-20056-4_15"},{"key":"5262_CR51","doi-asserted-by":"crossref","unstructured":"Chen, Y., Liu, Y., Jiang, D., Zhang, X., Dai, W., Xiong, H., Tian, Q.: Sdae: self-distillated masked autoencoder. In: European Conference on Computer Vision, pp. 108\u2013124. Springer (2022)","DOI":"10.1007\/978-3-031-20056-4_7"},{"key":"5262_CR52","doi-asserted-by":"crossref","unstructured":"Assran, M., Caron, M., Misra, I., Bojanowski, P., Bordes, F., Vincent, P., Joulin, A., Rabbat, M., Ballas, N.: Masked siamese networks for label-efficient learning. In: European Conference on Computer Vision, pp. 456\u2013473. Springer (2022)","DOI":"10.1007\/978-3-031-19821-2_26"},{"key":"5262_CR53","doi-asserted-by":"crossref","unstructured":"Caron, M., Touvron, H., Misra, I., J\u00e9gou, H., Mairal, J., Bojanowski, P., Joulin, A.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"5262_CR54","doi-asserted-by":"crossref","unstructured":"Kakogeorgiou, I., Gidaris, S., Psomas, B., Avrithis, Y., Bursuc, A., Karantzalos, K., Komodakis, N.: What to hide from your students: attention-guided masked image modeling. In: European Conference on Computer Vision, pp. 300\u2013318. Springer (2022)","DOI":"10.1007\/978-3-031-20056-4_18"},{"key":"5262_CR55","doi-asserted-by":"crossref","unstructured":"Feng, Z., Zhang, S.: Evolved part masking for self-supervised learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10386\u201310395 (2023)","DOI":"10.1109\/CVPR52729.2023.01001"},{"key":"5262_CR56","first-page":"14290","volume":"35","author":"G Li","year":"2022","unstructured":"Li, G., Zheng, H., Liu, D., Wang, C., Su, B., Zheng, C.: Semmae: semantic-guided masking for learning masked autoencoders. Adv. Neural. Inf. Process. Syst. 35, 14290\u201314302 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"5262_CR57","unstructured":"Shi, Y., Siddharth, N., Torr, P., Kosiorek, A.R.: Adversarial masking for self-supervised learning. In: International Conference on Machine Learning (PMLR), pp. 20026\u201320040 (2022)"},{"key":"5262_CR58","doi-asserted-by":"crossref","unstructured":"Wang, H., Tang, Y., Wang, Y., Guo, J., Deng, Z.H., Han, K.: Masked image modeling with local multi-scale reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2122\u20132131 (2023)","DOI":"10.1109\/CVPR52729.2023.00211"},{"key":"5262_CR59","unstructured":"Topcuoglu, U.M., Akag\u00fcnd\u00fcz, E.: Local masking meets progressive freezing: crafting efficient vision transformers for self-supervised learning. arXiv preprint arXiv:2312.02194 (2023)"},{"key":"5262_CR60","doi-asserted-by":"crossref","unstructured":"Assran, M., Duval, Q., Misra, I., Bojanowski, P., Vincent, P., Rabbat, M., LeCun, Y., Ballas, N.: Self-supervised learning from images with a joint-embedding predictive architecture. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15619\u201315629 (2023)","DOI":"10.1109\/CVPR52729.2023.01499"},{"key":"5262_CR61","unstructured":"Chen, J., Hu, M., Li, B., Elhoseiny, M.: Efficient self-supervised vision pretraining with local masked reconstruction. arXiv preprint arXiv:2206.00790 (2022)"},{"key":"5262_CR62","doi-asserted-by":"crossref","unstructured":"Chen, X., He, K.: Exploring simple siamese representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15750\u201315758 (2021)","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"5262_CR63","doi-asserted-by":"crossref","unstructured":"Chen, X., Xie, S., He, K.: An empirical study of training self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9640\u20139649 (2021)","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"5262_CR64","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, R., Shen, C., Kong, T., Li, L.: Dense contrastive learning for self-supervised visual pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3024\u20133033 (2021)","DOI":"10.1109\/CVPR46437.2021.00304"},{"key":"5262_CR65","doi-asserted-by":"crossref","unstructured":"Xie, Z., Lin, Y., Zhang, Z., Cao, Y., Lin, S., Hu, H.: Propagate yourself: exploring pixel-level consistency for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern recognition, pp. 16684\u201316693 (2021)","DOI":"10.1109\/CVPR46437.2021.01641"},{"key":"5262_CR66","unstructured":"Mishra, S., Robinson, J., Chang, H., Jacobs, D., Sarna, A., Maschinot, A., Krishnan, D.: A simple, efficient and scalable contrastive masked autoencoder for learning visual representations. arXiv preprint arXiv:2210.16870 (2022)"},{"key":"5262_CR67","doi-asserted-by":"crossref","unstructured":"Huang, Z., Jin, X., Lu, C., Hou, Q., Cheng, M.M., Fu, D., Shen, X., Feng, J.: Contrastive masked autoencoders are stronger vision learners. IEEE Trans. Pattern Anal. Mach. Intell. 46(4), 2506\u20132517 (2023)","DOI":"10.1109\/TPAMI.2023.3336525"},{"key":"5262_CR68","unstructured":"Yi, K., Ge, Y., Li, X., Yang, S., Li, D., Wu, J., Shan, Y., Qie, X.: Masked image modeling with denoising contrast. arXiv preprint arXiv:2205.09616 (2022)"},{"key":"5262_CR69","doi-asserted-by":"crossref","unstructured":"Zhou, Q., Yu, C., Luo, H., Wang, Z., Li, H.: Mimco: masked image modeling pre-training with contrastive teacher. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4487\u20134495 (2022)","DOI":"10.1145\/3503161.3548173"},{"key":"5262_CR70","doi-asserted-by":"crossref","unstructured":"Dong, X., Bao, J., Zheng, Y., Zhang, T., Chen, D., Yang, H., Zeng, M., Zhang, W., Yuan, L., Chen, D., et\u00a0al.: Maskclip: masked self-distillation advances contrastive language-image pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10995\u201311005 (2023)","DOI":"10.1109\/CVPR52729.2023.01058"},{"key":"5262_CR71","unstructured":"Liu, Y., Zhang, S., Chen, J., Chen, K., Lin, D.: Pixmim: rethinking pixel reconstruction in masked image modeling. arXiv preprint arXiv:2303.02416 (2023)"},{"key":"5262_CR72","first-page":"1649","volume":"37","author":"H Liu","year":"2023","unstructured":"Liu, H., Jiang, X., Li, X., Guo, A., Hu, Y., Jiang, D., Ren, B.: The devil is in the frequency: geminated gestalt autoencoder for self-supervised visual pre-training. Proc. AAAI Confer. Artif. Intell. 37, 1649\u20131656 (2023)","journal-title":"Proc. AAAI Confer. Artif. Intell."},{"key":"5262_CR73","unstructured":"Xie, J., Li, W., Zhan, X., Liu, Z., Ong, Y.S., Loy, C.C.: Masked frequency modeling for self-supervised visual pre-training. arXiv preprint arXiv:2206.07706 (2022)"},{"key":"5262_CR74","doi-asserted-by":"crossref","unstructured":"Recasens, A., Luc, P., Alayrac, J.B., Wang, L., Strub, F., Tallec, C., Malinowski, M., P\u0103tr\u0103ucean, V., Altch\u00e9, F., Valko, M., et\u00a0al.: Broaden your views for self-supervised video learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1255\u20131265 (2021)","DOI":"10.1109\/ICCV48922.2021.00129"},{"key":"5262_CR75","doi-asserted-by":"crossref","unstructured":"Xu, D., Xiao, J., Zhao, Z., Shao, J., Xie, D., Zhuang, Y.: Self-supervised spatiotemporal learning via video clip order prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10334\u201310343 (2019)","DOI":"10.1109\/CVPR.2019.01058"},{"key":"5262_CR76","doi-asserted-by":"crossref","unstructured":"Benaim, S., Ephrat, A., Lang, O., Mosseri, I., Freeman, W.T., Rubinstein, M., Irani, M., Dekel, T.: Speednet: learning the speediness in videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9922\u20139931 (2020)","DOI":"10.1109\/CVPR42600.2020.00994"},{"key":"5262_CR77","doi-asserted-by":"crossref","unstructured":"Pan, T., Song, Y., Yang, T., Jiang, W., Liu, W.: Videomoco: contrastive video representation learning with temporally adversarial examples. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11205\u201311214 (2021)","DOI":"10.1109\/CVPR46437.2021.01105"},{"key":"5262_CR78","doi-asserted-by":"crossref","unstructured":"Guo, S., Xiong, Z., Zhong, Y., Wang, L., Guo, X., Han, B., Huang, W.: Cross-architecture self-supervised video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19270\u201319279 (2022)","DOI":"10.1109\/CVPR52688.2022.01867"},{"key":"5262_CR79","doi-asserted-by":"crossref","unstructured":"Qian, R., Meng, T., Gong, B., Yang, M.H., Wang, H., Belongie, S., Cui, Y.: Spatiotemporal contrastive video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6964\u20136974 (2021)","DOI":"10.1109\/CVPR46437.2021.00689"},{"key":"5262_CR80","unstructured":"Patraucean, V., Handa, A., Cipolla, R.: Spatio-temporal video autoencoder with differentiable memory. arXiv preprint arXiv:1511.06309 (2015)"},{"key":"5262_CR81","unstructured":"Srivastava, N., Mansimov, E., Salakhudinov, R.: Unsupervised learning of video representations using lstms. In: International Conference on Machine Learning (PMLR), pp. 843\u2013852 (2015)"},{"key":"5262_CR82","unstructured":"Yan, W., Zhang, Y., Abbeel, P., Srinivas, A.: Videogpt: video generation using vq-vae and transformers. arXiv preprint arXiv:2104.10157 (2021)"},{"key":"5262_CR83","doi-asserted-by":"crossref","unstructured":"Girdhar, R., El-Nouby, A., Singh, M., Alwala, K.V., Joulin, A., Misra, I.: Omnimae: single model masked pretraining on images and videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10406\u201310417 (2023)","DOI":"10.1109\/CVPR52729.2023.01003"},{"key":"5262_CR84","unstructured":"Gupta, A., Wu, J., Deng, J., Li, F.F.: Siamese masked autoencoders. Adv. Neural Inf. Process. Syst. 36, 40676\u201340693 (2023)"},{"key":"5262_CR85","unstructured":"Tan, H., Lei, J., Wolf, T., Bansal, M.: Vimpac: Video pre-training via masked token prediction and contrastive learning. arXiv preprint arXiv:2106.11250 (2021)"},{"key":"5262_CR86","unstructured":"Wang, Y., Li, K., Li, Y., He, Y., Huang, B., Zhao, Z., Zhang, H., Xu, J., Liu, Y., Wang, Z., et\u00a0al.: Internvideo: general video foundation models via generative and discriminative learning. arXiv preprint arXiv:2212.03191 (2022)"},{"issue":"6","key":"5262_CR87","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou, J., Yu, B., Maybank, S.J., Tao, D.: Knowledge distillation: a survey. Int. J. Comput. Vis. 129(6), 1789\u20131819 (2021)","journal-title":"Int. J. Comput. Vis."},{"key":"5262_CR88","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)"},{"key":"5262_CR89","unstructured":"Hinton, G.E., Srivastava, N., Krizhevsky, A., Sutskever, I., Salakhutdinov, R.R.: Improving neural networks by preventing co-adaptation of feature detectors. arXiv preprint arXiv:1207.0580 (2012)"},{"key":"5262_CR90","unstructured":"Shen, Z., Savvides, M.: Meal v2: Boosting vanilla resnet-50 to 80%+ top-1 accuracy on imagenet without tricks. arXiv preprint arXiv:2009.08453 (2020)"},{"key":"5262_CR91","doi-asserted-by":"crossref","unstructured":"Shen, Z., Xing, E.: A fast knowledge distillation framework for visual recognition. In: European Conference on Computer Vision, pp. 673\u2013690. Springer (2022)","DOI":"10.1007\/978-3-031-20053-3_39"},{"key":"5262_CR92","doi-asserted-by":"crossref","unstructured":"Huang, W., Peng, Z., Dong, L., Wei, F., Jiao, J., Ye, Q.: Generic-to-specific distillation of masked autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15996\u201316005 (2023)","DOI":"10.1109\/CVPR52729.2023.01535"},{"key":"5262_CR93","unstructured":"Yang, Z., Li, Z., Zeng, A., Li, Z., Yuan, C., Li, Y.: Vitkd: practical guidelines for vit feature knowledge distillation. arXiv preprint arXiv:2209.02432 (2022)"},{"key":"5262_CR94","first-page":"1","volume":"132","author":"P Gao","year":"2023","unstructured":"Gao, P., Lin, Z., Zhang, R., Fang, R., Li, H., Li, H., Qiao, Y.: Mimic before reconstruct: enhancing masked autoencoders with feature mimicking. Int. J. Comput. Vis. 132, 1\u201311 (2023)","journal-title":"Int. J. Comput. Vis."},{"key":"5262_CR95","unstructured":"Peng, Z., Dong, L., Bao, H., Ye, Q., Wei, F.: A unified view of masked image modeling. arXiv preprint arXiv:2210.10615 (2022)"},{"key":"5262_CR96","doi-asserted-by":"crossref","unstructured":"Bai, Y., Wang, Z., Xiao, J., Wei, C., Wang, H., Yuille, A.L., Zhou, Y., Xie, C.: Masked autoencoders enable efficient knowledge distillers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 24256\u201324265 (2023)","DOI":"10.1109\/CVPR52729.2023.02323"},{"key":"5262_CR97","doi-asserted-by":"crossref","unstructured":"Xue, H., Gao, P., Li, H., Qiao, Y., Sun, H., Li, H., Luo, J.: Stare at what you see: masked image modeling without reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22732\u201322741 (2023)","DOI":"10.1109\/CVPR52729.2023.02177"},{"key":"5262_CR98","doi-asserted-by":"crossref","unstructured":"Ranzinger, M., Heinrich, G., Kautz, J., Molchanov, P.: Am-radio: agglomerative vision foundation model reduce all domains into one. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12490\u201312500 (2024)","DOI":"10.1109\/CVPR52733.2024.01187"},{"issue":"6","key":"5262_CR99","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-CNN: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"5262_CR100","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., He, K.: Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7794\u20137803 (2018)","DOI":"10.1109\/CVPR.2018.00813"},{"key":"5262_CR101","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wang, L., Wu, W., Qian, C., Lu, T.: Tam: temporal adaptive module for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13708\u201313718 (2021)","DOI":"10.1109\/ICCV48922.2021.01345"},{"key":"5262_CR102","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Feiszli, M.: Video classification with channel-separated convolutional networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5552\u20135561 (2019)","DOI":"10.1109\/ICCV.2019.00565"},{"key":"5262_CR103","doi-asserted-by":"crossref","unstructured":"Huang, B., Zhao, Z., Zhang, G., Qiao, Y., Wang, L.: Mgmae: Motion guided masking for video masked autoencoding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13493\u201313504 (2023)","DOI":"10.1109\/ICCV51070.2023.01241"},{"key":"5262_CR104","unstructured":"Yang, H., Huang, D., Wen, B., Wu, J., Yao, H., Jiang, Y., Zhu, X., Yuan, Z.: Self-supervised video representation learning with motion-aware masked autoencoders. arXiv preprint arXiv:2210.04154 (2022)"},{"key":"5262_CR105","first-page":"11669","volume":"34","author":"Z Liu","year":"2020","unstructured":"Liu, Z., Luo, D., Wang, Y., Wang, L., Tai, Y., Wang, C., Li, J., Huang, F., Lu, T.: Teinet: towards an efficient architecture for video recognition. Proc. AAAI Conf. Artif. Intell. 34, 11669\u201311676 (2020)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"5262_CR106","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Xiong, B., Girshick, R., He, K.: A large-scale study on unsupervised spatiotemporal representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3299\u20133309 (2021)","DOI":"10.1109\/CVPR46437.2021.00331"}],"container-title":["Cluster Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-025-05262-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10586-025-05262-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10586-025-05262-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,17]],"date-time":"2025-09-17T21:23:26Z","timestamp":1758144206000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10586-025-05262-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,30]]},"references-count":106,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["5262"],"URL":"https:\/\/doi.org\/10.1007\/s10586-025-05262-8","relation":{},"ISSN":["1386-7857","1573-7543"],"issn-type":[{"type":"print","value":"1386-7857"},{"type":"electronic","value":"1573-7543"}],"subject":[],"published":{"date-parts":[[2025,8,30]]},"assertion":[{"value":"25 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 January 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 March 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 August 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"566"}}