{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T04:29:32Z","timestamp":1781497772191,"version":"3.54.1"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T00:00:00Z","timestamp":1736812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T00:00:00Z","timestamp":1736812800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"The High-level Talent Innovation and Entrepreneurship Project of Quanzhou City","award":["2023C013R"],"award-info":[{"award-number":["2023C013R"]}]},{"name":"The High-level Talent Innovation and Entrepreneurship Project of Quanzhou City","award":["2023C013R"],"award-info":[{"award-number":["2023C013R"]}]},{"name":"The Natural Science Foundation for Outstanding Young Scholars of Fujian Province","award":["2022J06023"],"award-info":[{"award-number":["2022J06023"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1007\/s00530-024-01646-9","type":"journal-article","created":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T13:13:39Z","timestamp":1736860419000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["A strong benchmark for yoga action recognition based on lightweight pose estimation model"],"prefix":"10.1007","volume":"31","author":[{"given":"Liangtai","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weiwei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Banghui","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaobin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianqing","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,1,14]]},"reference":[{"issue":"3","key":"1646_CR1","doi-asserted-by":"publisher","first-page":"242","DOI":"10.3109\/09540261.2016.1160878","volume":"28","author":"R Govindaraj","year":"2016","unstructured":"Govindaraj, R., Karmani, S., Varambally, S., Gangadhar, B.: Yoga and physical exercise-a review and comparison. Int. Rev. Psychiatry 28(3), 242\u2013253 (2016)","journal-title":"Int. Rev. Psychiatry"},{"key":"1646_CR2","doi-asserted-by":"crossref","unstructured":"Duan, H., Zhao, Y., Chen, K., Lin, D., Dai, B.: Revisiting skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2969\u20132978 (2022)","DOI":"10.1109\/CVPR52688.2022.00298"},{"issue":"1","key":"1646_CR3","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/TPAMI.2019.2929257","volume":"43","author":"Z Cao","year":"2021","unstructured":"Cao, Z., Hidalgo, G., Simon, T., Wei, S.-E., Sheikh, Y.: Openpose: Realtime multi-person 2d pose estimation using part affinity fields. IEEE Trans. Pattern Anal. Mach. Intell. 43(1), 172\u2013186 (2021). https:\/\/doi.org\/10.1109\/TPAMI.2019.2929257","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1646_CR4","first-page":"38571","volume":"35","author":"Y Xu","year":"2022","unstructured":"Xu, Y., Zhang, J., Zhang, Q., Tao, D.: Vitpose: Simple vision transformer baselines for human pose estimation. Adv. Neural. Inf. Process. Syst. 35, 38571\u201338584 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1646_CR5","unstructured":"Bazarevsky, V., Grishchenko, I., Raveendran, K., Zhu, T., Zhang, F., Grundmann, M.: Blazepose: On-device real-time body pose tracking. arXiv preprint arXiv:2006.10204 (2020)"},{"key":"1646_CR6","first-page":"1","volume":"70","author":"R Bajpai","year":"2021","unstructured":"Bajpai, R., Joshi, D.: Movenet: A deep neural network for joint profile prediction across variable walking speeds and slopes. IEEE Trans. Instrum. Meas. 70, 1\u201311 (2021)","journal-title":"IEEE Trans. Instrum. Meas."},{"issue":"1","key":"1646_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3603618","volume":"56","author":"C Zheng","year":"2023","unstructured":"Zheng, C., Wu, W., Chen, C., Yang, T., Zhu, S., Shen, J., Kehtarnavaz, N., Shah, M.: Deep learning-based human pose estimation: A survey. ACM Comput. Surv. 56(1), 1\u201337 (2023)","journal-title":"ACM Comput. Surv."},{"key":"1646_CR8","unstructured":"Jiang, T., Lu, P., Zhang, L., Ma, N., Han, R., Lyu, C., Li, Y., Chen, K.: Rtmpose: Real-time multi-person pose estimation based on mmpose. arXiv e-prints, 2303 (2023)"},{"key":"1646_CR9","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5693\u20135703 (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"key":"1646_CR10","doi-asserted-by":"crossref","unstructured":"Toshev, A., Szegedy, C.: Deeppose: Human pose estimation via deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1653\u20131660 (2014)","DOI":"10.1109\/CVPR.2014.214"},{"key":"1646_CR11","unstructured":"Zhou, X., Wang, D., Kr\u00e4henb\u00fchl, P.: Objects as Points (2019)"},{"key":"1646_CR12","doi-asserted-by":"crossref","unstructured":"Shi, D., Wei, X., Li, L., Ren, Y., Tan, W.: End-to-end multi-person pose estimation with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11069\u201311078 (2022)","DOI":"10.1109\/CVPR52688.2022.01079"},{"key":"1646_CR13","doi-asserted-by":"crossref","unstructured":"Li, J., Bian, S., Zeng, A., Wang, C., Pang, B., Liu, W., Lu, C.: Human pose regression with residual log-likelihood estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11025\u201311034 (2021)","DOI":"10.1109\/ICCV48922.2021.01084"},{"key":"1646_CR14","doi-asserted-by":"crossref","unstructured":"Ye, S., Zhang, Y., Hu, J., Cao, L., Zhang, S., Shen, L., Wang, J., Ding, S., Ji, R.: Distilpose: Tokenized pose regression with heatmap distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2163\u20132172 (2023)","DOI":"10.1109\/CVPR52729.2023.00215"},{"key":"1646_CR15","doi-asserted-by":"crossref","unstructured":"Li, Y., Yang, S., Liu, P., Zhang, S., Wang, Y., Wang, Z., Yang, W., Xia, S.-T.: Simcc: A simple coordinate classification perspective for human pose estimation. In: European Conference on Computer Vision, pp. 89\u2013106 (2022). Springer","DOI":"10.1007\/978-3-031-20068-7_6"},{"key":"1646_CR16","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)"},{"key":"1646_CR17","doi-asserted-by":"crossref","unstructured":"Zhang, F., Zhu, X., Ye, M.: Fast human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3517\u20133526 (2019)","DOI":"10.1109\/CVPR.2019.00363"},{"key":"1646_CR18","doi-asserted-by":"crossref","unstructured":"Li, Z., Ye, J., Song, M., Huang, Y., Pan, Z.: Online knowledge distillation for efficient pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11740\u201311750 (2021)","DOI":"10.1109\/ICCV48922.2021.01153"},{"key":"1646_CR19","doi-asserted-by":"crossref","unstructured":"Weinzaepfel, P., Br\u00e9gier, R., Combaluzier, H., Leroy, V., Rogez, G.: Dope: Distillation of part experts for whole-body 3d pose estimation in the wild. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXVI 16, pp. 380\u2013397 (2020). Springer","DOI":"10.1007\/978-3-030-58574-7_23"},{"key":"1646_CR20","unstructured":"Zhang, F., Bazarevsky, V., Vakunov, A., Tkachenka, A., Sung, G., Chang, C.-L., Grundmann, M.: MediaPipe Hands: On-device Real-time Hand Tracking (2020)"},{"key":"1646_CR21","doi-asserted-by":"crossref","unstructured":"Wu, W., Yin, W., Guo, F.: Learning and self-instruction expert system for yoga. In: 2010 2nd International Workshop on Intelligent Systems and Applications, pp. 1\u20134 (2010). IEEE","DOI":"10.1109\/IWISA.2010.5473592"},{"key":"1646_CR22","doi-asserted-by":"crossref","unstructured":"Luo, Z., Yang, W., Ding, Z.Q., Liu, L., Chen, I.-M., Yeo, S.H., Ling, K.V., Duh, H.B.-L.: \u201cleft arm up!\u201d interactive yoga training in virtual environment. In: 2011 IEEE Virtual Reality Conference, pp. 261\u2013262 (2011). IEEE","DOI":"10.1109\/VR.2011.5759498"},{"key":"1646_CR23","doi-asserted-by":"crossref","unstructured":"Agrawal, Y., Shah, Y., Sharma, A.: Implementation of machine learning technique for identification of yoga poses. In: 2020 IEEE 9th International Conference on Communication Systems and Network Technologies (CSNT), pp. 40\u201343 (2020). Ieee","DOI":"10.1109\/CSNT48778.2020.9115758"},{"issue":"6","key":"1646_CR24","doi-asserted-by":"publisher","first-page":"476","DOI":"10.1007\/s42979-022-01376-7","volume":"3","author":"M Chasmai","year":"2022","unstructured":"Chasmai, M., Das, N., Bhardwaj, A., Garg, R.: A view independent classification framework for yoga postures. SN computer science 3(6), 476 (2022)","journal-title":"SN computer science"},{"issue":"7","key":"1646_CR25","doi-asserted-by":"publisher","first-page":"9515","DOI":"10.1109\/JSEN.2021.3055898","volume":"21","author":"S Liaqat","year":"2021","unstructured":"Liaqat, S., Dashtipour, K., Arshad, K., Assaleh, K., Ramzan, N.: A hybrid posture detection framework: Integrating machine learning and deep neural networks. IEEE Sens. J. 21(7), 9515\u20139522 (2021)","journal-title":"IEEE Sens. J."},{"key":"1646_CR26","doi-asserted-by":"crossref","unstructured":"Narayanan, S.S., Misra, D.K., Arora, K., Rai, H.: Yoga pose detection using deep learning techniques. In: Proceedings of the International Conference on Innovative Computing & Communication (ICICC) (2021)","DOI":"10.2139\/ssrn.3842656"},{"issue":"12","key":"1646_CR27","doi-asserted-by":"publisher","first-page":"16551","DOI":"10.1007\/s12652-022-03910-0","volume":"14","author":"S Garg","year":"2023","unstructured":"Garg, S., Saxena, A., Gupta, R.: Yoga pose classification: a cnn and mediapipe inspired deep learning approach for real-world application. J. Ambient. Intell. Humaniz. Comput. 14(12), 16551\u201316562 (2023)","journal-title":"J. Ambient. Intell. Humaniz. Comput."},{"key":"1646_CR28","doi-asserted-by":"crossref","unstructured":"Bera, A., Nasipuri, M., Krejcar, O., Bhattacharjee, D.: Fine-grained sports, yoga, and dance postures recognition: A benchmark analysis. IEEE Transactions on Instrumentation and Measurement (2023)","DOI":"10.1109\/TIM.2023.3293564"},{"key":"1646_CR29","doi-asserted-by":"crossref","unstructured":"Srinivasan, T.: Dynamic and static asana practices. Medknow (2016)","DOI":"10.4103\/0973-6131.171724"},{"key":"1646_CR30","doi-asserted-by":"publisher","first-page":"1155","DOI":"10.1109\/ACCESS.2017.2778011","volume":"6","author":"A Ullah","year":"2017","unstructured":"Ullah, A., Ahmad, J., Muhammad, K., Sajjad, M., Baik, S.W.: Action recognition in video sequences using deep bi-directional lstm with cnn features. IEEE access 6, 1155\u20131166 (2017)","journal-title":"IEEE access"},{"key":"1646_CR31","doi-asserted-by":"crossref","unstructured":"Sun, B., Ye, X., Yan, T., Wang, Z., Li, H., Wang, Z.: Fine-grained action recognition with robust motion representation decoupling and concentration. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4779\u20134788 (2022)","DOI":"10.1145\/3503161.3548046"},{"key":"1646_CR32","unstructured":"Kim, S.: 3DYoga90: A Hierarchical Video Dataset for Yoga Pose Understanding (2023)"},{"key":"1646_CR33","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"1646_CR34","doi-asserted-by":"crossref","unstructured":"Shi, L., Zhang, Y., Cheng, J., Lu, H.: Two-stream adaptive graph convolutional networks for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12026\u201312035 (2019)","DOI":"10.1109\/CVPR.2019.01230"},{"key":"1646_CR35","doi-asserted-by":"crossref","unstructured":"Duan, H., Wang, J., Chen, K., Lin, D.: Pyskl: Towards good practices for skeleton action recognition. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 7351\u20137354 (2022)","DOI":"10.1145\/3503161.3548546"},{"key":"1646_CR36","doi-asserted-by":"crossref","unstructured":"Chen, C.-F.R., Panda, R., Ramakrishnan, K., Feris, R., Cohn, J., Oliva, A., Fan, Q.: Deep analysis of cnn-based spatio-temporal representations for action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6165\u20136175 (2021)","DOI":"10.1109\/CVPR46437.2021.00610"},{"key":"1646_CR37","doi-asserted-by":"crossref","unstructured":"Noor, N., Park, I.K.: A lightweight skeleton-based 3d-cnn for real-time fall detection and action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2179\u20132188 (2023)","DOI":"10.1109\/ICCVW60793.2023.00232"},{"key":"1646_CR38","doi-asserted-by":"crossref","unstructured":"Sun, B., Ye, X., Wang, Z., Li, H., Wang, Z.: Exploring coarse-to-fine action token localization and interaction for fine-grained video action recognition. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5070\u20135078 (2023)","DOI":"10.1145\/3581783.3612206"},{"key":"1646_CR39","doi-asserted-by":"crossref","unstructured":"Ahn, D., Kim, S., Hong, H., Ko, B.C.: Star-transformer: a spatio-temporal cross attention transformer for human action recognition. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 3330\u20133339 (2023)","DOI":"10.1109\/WACV56688.2023.00333"},{"issue":"7","key":"1646_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3654671","volume":"20","author":"B Sun","year":"2024","unstructured":"Sun, B., Ye, X., Yan, T., Wang, Z., Li, H., Wang, Z.: Discriminative segment focus network for fine-grained video action recognition. ACM Trans. Multimed. Comput. Commun. Appl. 20(7), 1\u201320 (2024)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"1646_CR41","unstructured":"Zagoruyko, S., Komodakis, N.: Paying More Attention to Attention: Improving the Performance of Convolutional Neural Networks via Attention Transfer (2017)"},{"key":"1646_CR42","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S.-E., Sheikh, Y.: Realtime multi-person 2d pose estimation using part affinity fields. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7291\u20137299 (2017)","DOI":"10.1109\/CVPR.2017.143"},{"key":"1646_CR43","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pp. 740\u2013755 (2014). Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1646_CR44","unstructured":"Yang, J., Zeng, A., Liu, S., Li, F., Zhang, R., Zhang, L.: Explicit box detection unifies end-to-end multi-person pose estimation. In: International Conference on Learning Representations (2023). https:\/\/openreview.net\/forum?id=s4WVupnJjmX"},{"key":"1646_CR45","doi-asserted-by":"publisher","unstructured":"Jiang, T., Lu, P., Zhang, L., Ma, N., Han, R., Lyu, C., Li, Y., Chen, K.: RTMPose: Real-Time Multi-Person Pose Estimation based on MMPose. arXiv (2023). https:\/\/doi.org\/10.48550\/ARXIV.2303.07399 . https:\/\/arxiv.org\/abs\/2303.07399","DOI":"10.48550\/ARXIV.2303.07399"},{"key":"1646_CR46","doi-asserted-by":"publisher","first-page":"9349","DOI":"10.1007\/s00521-019-04232-7","volume":"31","author":"SK Yadav","year":"2019","unstructured":"Yadav, S.K., Singh, A., Gupta, A., Raheja, J.L.: Real-time yoga recognition using deep learning. Neural Comput. Appl. 31, 9349\u20139361 (2019)","journal-title":"Neural Comput. Appl."},{"key":"1646_CR47","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1440\u20131448 (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"1646_CR48","unstructured":"Maaten, L., Hinton, G.: Visualizing data using t-sne. Journal of Machine Learning Research 9(11) (2008)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01646-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01646-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01646-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,28]],"date-time":"2025-02-28T11:06:01Z","timestamp":1740740761000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01646-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,14]]},"references-count":48,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,2]]}},"alternative-id":["1646"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01646-9","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,14]]},"assertion":[{"value":"3 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 December 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 January 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors of this research paper declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"66"}}