{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T10:56:52Z","timestamp":1784977012114,"version":"3.55.0"},"reference-count":76,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T00:00:00Z","timestamp":1770595200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T00:00:00Z","timestamp":1770595200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s00530-025-02204-7","type":"journal-article","created":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T14:35:38Z","timestamp":1770647738000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Exploring multi-transformer with fine-grained prompt-driven coupled with diffusion model for 3D human pose estimation"],"prefix":"10.1007","volume":"32","author":[{"given":"Sathiyamoorthi","family":"Arthanari","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sathishkumar","family":"Moorthy","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jae Hoon","family":"Jeong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,2,9]]},"reference":[{"key":"2204_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2025.107210","volume":"186","author":"S Arthanari","year":"2025","unstructured":"Arthanari, S., Elayaperumal, D., Joo, Y.H.: Learning temporal regularized spatial-aware deep correlation filter tracking via adaptive channel selection. Neural Networks 186, 107210 (2025)","journal-title":"Neural Networks"},{"key":"2204_CR2","doi-asserted-by":"crossref","unstructured":"Moorthy, S., KS, S.S., Arthanari, S., Jeong, J.H., Joo, Y.H.: Learning disruptor-suppressed response variation-aware multi-regularized correlation filter for visual tracking. J. Vis. Commun. Image Represent., 104458 (2025)","DOI":"10.1016\/j.jvcir.2025.104458"},{"key":"2204_CR3","unstructured":"KS, S.S., Jeong, J.H., Joo, Y.H.: A multi-level hybrid siamese network using box adaptive and classification approach for robust tracking. Multimed. Tools Appl., 1\u201326 (2024)"},{"key":"2204_CR4","doi-asserted-by":"publisher","first-page":"502","DOI":"10.1016\/j.ins.2023.02.009","volume":"629","author":"D Elayaperumal","year":"2023","unstructured":"Elayaperumal, D., Joo, Y.H.: Learning spatial variance-key surrounding-aware tracking via multi-expert deep feature fusion. Inf. Sci. 629, 502\u2013519 (2023)","journal-title":"Inf. Sci."},{"key":"2204_CR5","volume":"136","author":"S Arthanari","year":"2025","unstructured":"Arthanari, S., Moorthy, S., Jeong, J.H., Joo, Y.H.: Adaptive spatially regularized target attribute-aware background suppressed deep correlation filter for object tracking. Signal Proces.: Image Commun. 136, 117305 (2025)","journal-title":"Signal Proces.: Image Commun."},{"issue":"5","key":"2204_CR6","doi-asserted-by":"publisher","first-page":"3260","DOI":"10.1109\/TCSVT.2023.3318557","volume":"34","author":"L Zhou","year":"2023","unstructured":"Zhou, L., Chen, Y., Wang, J.: Dual-path transformer for 3d human pose estimation. IEEE Trans. Circuits Syst. Video Technol. 34(5), 3260\u20133270 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2204_CR7","doi-asserted-by":"crossref","unstructured":"Xie, B., Liu, G., Deng, F., Lu, M.: Aitepose: Learning an end-to-end monocular 3d human pose estimator via auxiliary-information-driven training enhancement. IEEE Trans. Circ. Syst. Video Technol. (2025)","DOI":"10.1109\/TCSVT.2025.3570967"},{"key":"2204_CR8","doi-asserted-by":"crossref","unstructured":"Zhang, J., Tu, Z., Yang, J., Chen, Y., Yuan, J.: Mixste: Seq2seq mixed spatio-temporal encoder for 3d human pose estimation in video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13232\u201313242 (2022)","DOI":"10.1109\/CVPR52688.2022.01288"},{"key":"2204_CR9","doi-asserted-by":"crossref","unstructured":"Arthanari, S., Jeong, J.H., Joo, Y.H.: Exploiting multi-transformer encoder with multiple-hypothesis aggregation via diffusion model for 3d human pose estimation. Multimed. Tools Appl., 1\u201329 (2024)","DOI":"10.1007\/s11042-024-20179-x"},{"key":"2204_CR10","doi-asserted-by":"crossref","unstructured":"Li, W., Liu, H., Tang, H., Wang, P., Van\u00a0Gool, L.: Mhformer: Multi-hypothesis transformer for 3d human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13147\u201313156 (2022)","DOI":"10.1109\/CVPR52688.2022.01280"},{"issue":"5","key":"2204_CR11","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1007\/s00530-024-01451-4","volume":"30","author":"S Arthanari","year":"2024","unstructured":"Arthanari, S., Jeong, J.H., Joo, Y.H.: Exploring multi-level transformers with feature frame padding network for 3d human pose estimation. Multimedia Syst. 30(5), 243 (2024)","journal-title":"Multimedia Syst."},{"key":"2204_CR12","volume":"136","author":"S Arthanari","year":"2025","unstructured":"Arthanari, S., Moorthy, S., Jeong, J.H., Joo, Y.H.: Adaptive spatially regularized target attribute-aware background suppressed deep correlation filter for object tracking. Signal Processing: Image Commun. 136, 117305 (2025)","journal-title":"Signal Processing: Image Commun."},{"issue":"14","key":"2204_CR13","doi-asserted-by":"publisher","first-page":"2279","DOI":"10.3390\/math12142279","volume":"12","author":"SS Kuppusami Sakthivel","year":"2024","unstructured":"Kuppusami Sakthivel, S.S., Moorthy, S., Arthanari, S., Jeong, J.H., Joo, Y.H.: Learning a context-aware environmental residual correlation filter via deep convolution features for visual object tracking. Math. 12(14), 2279 (2024)","journal-title":"Math."},{"key":"2204_CR14","doi-asserted-by":"publisher","first-page":"467","DOI":"10.1016\/j.ins.2021.06.084","volume":"577","author":"D Elayaperumal","year":"2021","unstructured":"Elayaperumal, D., Joo, Y.H.: Robust visual object tracking using context-based spatial variation via multi-feature fusion. Inf. Sci. 577, 467\u2013482 (2021)","journal-title":"Inf. Sci."},{"key":"2204_CR15","doi-asserted-by":"crossref","unstructured":"Cai, Y., Ge, L., Liu, J., Cai, J., Cham, T.-J., Yuan, J., Thalmann, N.M.: Exploiting spatial-temporal relationships for 3d pose estimation via graph convolutional networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2272\u20132281 (2019)","DOI":"10.1109\/ICCV.2019.00236"},{"key":"2204_CR16","doi-asserted-by":"crossref","unstructured":"Liu, J., Rojas, J., Li, Y., Liang, Z., Guan, Y., Xi, N., Zhu, H.: A graph attention spatio-temporal convolutional network for 3d human pose estimation in video. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 3374\u20133380 (2021). IEEE","DOI":"10.1109\/ICRA48506.2021.9561605"},{"key":"2204_CR17","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1016\/j.neucom.2021.11.007","volume":"487","author":"Y Wu","year":"2022","unstructured":"Wu, Y., Kong, D., Wang, S., Li, J., Yin, B.: Hpgcn: Hierarchical poselet-guided graph convolutional network for 3d pose estimation. Neurocomputing 487, 243\u2013256 (2022)","journal-title":"Neurocomputing"},{"key":"2204_CR18","doi-asserted-by":"crossref","unstructured":"Yu, B.X., Zhang, Z., Liu, Y., Zhong, S.-h., Liu, Y., Chen, C.W.: Gla-gcn: Global-local adaptive graph convolutional network for 3d human pose estimation from monocular video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8818\u20138829 (2023)","DOI":"10.1109\/ICCV51070.2023.00810"},{"key":"2204_CR19","doi-asserted-by":"publisher","first-page":"4212","DOI":"10.1109\/TIP.2023.3275914","volume":"32","author":"MT Hassan","year":"2023","unstructured":"Hassan, M.T., Ben Hamza, A.: Regular splitting graph network for 3d human pose estimation. IEEE Trans. Image Process. 32, 4212\u20134222 (2023). https:\/\/doi.org\/10.1109\/TIP.2023.3275914","journal-title":"IEEE Trans. Image Process."},{"key":"2204_CR20","doi-asserted-by":"crossref","unstructured":"Zheng, C., Zhu, S., Mendieta, M., Yang, T., Chen, C., Ding, Z.: 3d human pose estimation with spatial and temporal transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11656\u201311665 (2021)","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"2204_CR21","doi-asserted-by":"crossref","unstructured":"Xie, Y., Hong, C., Zhuang, W., Liu, L., Li, J.: Hogformer: high-order graph convolution transformer for 3d human pose estimation. Int. J. Machine Learning and Cybernetics, 1\u201312 (2024)","DOI":"10.1007\/s13042-024-02262-9"},{"key":"2204_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110446","volume":"152","author":"Z Chen","year":"2024","unstructured":"Chen, Z., Dai, J., Bai, J., Pan, J.: Dgformer: Dynamic graph transformer for 3d human pose estimation. Pattern Recogn. 152, 110446 (2024)","journal-title":"Pattern Recogn."},{"key":"2204_CR23","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2024.109606","volume":"139","author":"S Moorthy","year":"2025","unstructured":"Moorthy, S., KS, S.S., Arthanari, S., Jeong, J.H., Joo, Y.H.: Hybrid multi-attention transformer for robust video object detection. Eng. Appl. Artif. Intell. 139, 109606 (2025)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"2204_CR24","doi-asserted-by":"crossref","unstructured":"Xu, J., Guo, Y., Peng, Y.: Finepose: Fine-grained prompt-driven 3d human pose estimation via diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 561\u2013570 (2024)","DOI":"10.1109\/CVPR52733.2024.00060"},{"key":"2204_CR25","doi-asserted-by":"crossref","unstructured":"Shan, W., Zhang, Y., Zhang, X., Wang, S., Zhou, X., Ma, S., Gao, W.: Diffusion-based hypotheses generation and joint-level hypotheses aggregation for 3d human pose estimation. IEEE Trans. Circuits Syst. Video Tech. (2024)","DOI":"10.1109\/TCSVT.2024.3415348"},{"issue":"2","key":"2204_CR26","doi-asserted-by":"publisher","first-page":"911","DOI":"10.1109\/TCSVT.2023.3286402","volume":"34","author":"Z Tang","year":"2023","unstructured":"Tang, Z., Hao, Y., Li, J., Hong, R.: Ftcm: Frequency-temporal collaborative module for efficient 3d human pose estimation in video. IEEE Trans. Circuits Syst. Video Technol. 34(2), 911\u2013923 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2204_CR27","doi-asserted-by":"crossref","unstructured":"Ma, X., Su, J., Wang, C., Ci, H., Wang, Y.: Context modeling in 3d human pose estimation: A unified perspective. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6238\u20136247 (2021)","DOI":"10.1109\/CVPR46437.2021.00617"},{"key":"2204_CR28","doi-asserted-by":"crossref","unstructured":"Fang, H.-S., Xu, Y., Wang, W., Liu, X., Zhu, S.-C.: Learning pose grammar to encode human body configuration for 3d pose estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.12270"},{"issue":"4","key":"2204_CR29","doi-asserted-by":"publisher","first-page":"4122","DOI":"10.1109\/TPAMI.2022.3188716","volume":"45","author":"H Shuai","year":"2023","unstructured":"Shuai, H., Wu, L., Liu, Q.: Adaptive multi-view and temporal fusing transformer for 3d human pose estimation. IEEE Trans. Pattern Anal. Mach. Intell. 45(4), 4122\u20134135 (2023). https:\/\/doi.org\/10.1109\/TPAMI.2022.3188716","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2204_CR30","doi-asserted-by":"publisher","first-page":"1832","DOI":"10.1109\/TMM.2022.3171102","volume":"25","author":"G Hua","year":"2023","unstructured":"Hua, G., Liu, H., Li, W., Zhang, Q., Ding, R., Xu, X.: Weakly-supervised 3d human pose estimation with cross-view u-shaped graph convolutional network. IEEE Trans. Multimed. 25, 1832\u20131843 (2023). https:\/\/doi.org\/10.1109\/TMM.2022.3171102","journal-title":"IEEE Trans. Multimed."},{"key":"2204_CR31","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110925","volume":"158","author":"W Li","year":"2025","unstructured":"Li, W., Liu, M., Liu, H., Guo, T., Wang, T., Tang, H., Sebe, N.: Graphmlp: A graph mlp-like architecture for 3d human pose estimation. Pattern Recogn. 158, 110925 (2025)","journal-title":"Pattern Recogn."},{"key":"2204_CR32","unstructured":"Lu, J., Lin, J., Dou, H., Zeng, A., Deng, Y., Zhang, Y., Wang, H.: Dposer: Diffusion model as robust 3d human pose prior. arXiv preprint arXiv:2312.05541 (2023)"},{"key":"2204_CR33","doi-asserted-by":"crossref","unstructured":"Cai, Q., Hu, X., Hou, S., Yao, L., Huang, Y.: Disentangled diffusion-based 3d human pose estimation with hierarchical spatial and temporal denoiser. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 882\u2013890 (2024)","DOI":"10.1609\/aaai.v38i2.27847"},{"key":"2204_CR34","doi-asserted-by":"crossref","unstructured":"Chen, Z., Dai, J., Pan, J., Zhou, F.: Diffusion model with temporal constraint for 3d human pose estimation. Vis. Comput., 1\u201317 (2024)","DOI":"10.1007\/s00371-024-03763-y"},{"key":"2204_CR35","doi-asserted-by":"crossref","unstructured":"Zheng, H., Li, H., Shi, B., Dai, W., Wang, B., Sun, Y., Guo, M., Xiong, H.: Actionprompt: Action-guided 3d human pose estimation with text and pose prompting. In: 2023 IEEE International Conference on Multimedia and Expo (ICME), pp. 2657\u20132662 (2023). IEEE","DOI":"10.1109\/ICME55011.2023.00452"},{"key":"2204_CR36","doi-asserted-by":"crossref","unstructured":"Hu, S., Zheng, C., Zhou, Z., Chen, C., Sukthankar, G.: Lamp: Leveraging language prompts for multi-person pose estimation. In: 2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 3759\u20133766 (2023). IEEE","DOI":"10.1109\/IROS55552.2023.10341430"},{"key":"2204_CR37","doi-asserted-by":"crossref","unstructured":"Chen, H., He, J.-Y., Xiang, W., Cheng, Z.-Q., Liu, W., Liu, H., Luo, B., Geng, Y., Xie, X.: Hdformer: High-order directed transformer for 3d human pose estimation (2023). arXiv preprint arXiv:2302.01825","DOI":"10.24963\/ijcai.2023\/65"},{"issue":"7","key":"2204_CR38","doi-asserted-by":"publisher","first-page":"1100","DOI":"10.3390\/math13071100","volume":"13","author":"S Moorthy","year":"2025","unstructured":"Moorthy, S., Moon, Y.-K.: Hybrid multi-attention network for audio-visual emotion recognition through multimodal feature fusion. Mathematics 13(7), 1100 (2025)","journal-title":"Mathematics"},{"key":"2204_CR39","unstructured":"Qian, X., Tang, Y., Zhang, N., Han, M., Xiao, J., Huang, M.-C., Lin, R.-S.: Hstformer: Hierarchical spatial-temporal transformers for 3d human pose estimation (2023). arXiv preprint arXiv:2301.07322"},{"key":"2204_CR40","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Lu, Y., Liu, B., Zhao, Z., Chu, Q., Yu, N.: Evopose: A recursive transformer for 3d human pose estimation with kinematic structure priors. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 1\u20135 (2023). IEEE","DOI":"10.1109\/ICASSP49357.2023.10095302"},{"issue":"4","key":"2204_CR41","doi-asserted-by":"publisher","first-page":"2555","DOI":"10.1007\/s00371-023-02936-5","volume":"40","author":"A Diaz-Arias","year":"2024","unstructured":"Diaz-Arias, A., Shin, D.: Convformer: parameter reduction in transformer models for 3d human pose estimation by leveraging dynamic multi-headed convolutional attention. Vis. Comput. 40(4), 2555\u20132569 (2024)","journal-title":"Vis. Comput."},{"key":"2204_CR42","doi-asserted-by":"crossref","unstructured":"Choi, J., Shim, D., Kim, H.J.: Diffupose: Monocular 3d human pose estimation via denoising diffusion probabilistic model. In: 2023 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 3773\u20133780 (2023). IEEE","DOI":"10.1109\/IROS55552.2023.10342204"},{"key":"2204_CR43","doi-asserted-by":"crossref","unstructured":"Kang, H., Wang, Y., Liu, M., Wu, D., Liu, P., Yuan, X., Yang, W.: Diffusion-based pose refinement and multi-hypothesis generation for 3d human pose estimation. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5130\u20135134 (2024). IEEE","DOI":"10.1109\/ICASSP48485.2024.10445850"},{"key":"2204_CR44","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2204_CR45","unstructured":"Zhang, M., Cai, Z., Pan, L., Hong, F., Guo, X., Yang, L., Liu, Z.: Motiondiffuse: Text-driven human motion generation with diffusion model (2022). arXiv preprint arXiv:2208.15001"},{"key":"2204_CR46","doi-asserted-by":"crossref","unstructured":"Pavllo, D., Feichtenhofer, C., Grangier, D., Auli, M.: 3d human pose estimation in video with temporal convolutions and semi-supervised training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7753\u20137762 (2019)","DOI":"10.1109\/CVPR.2019.00794"},{"key":"2204_CR47","doi-asserted-by":"crossref","unstructured":"Zeng, A., Sun, X., Huang, F., Liu, M., Xu, Q., Lin, S.: Srnet: Improving generalization in 3d human pose estimation with a split-and-recombine approach. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16, pp. 507\u2013523 (2020). Springer","DOI":"10.1007\/978-3-030-58568-6_30"},{"key":"2204_CR48","doi-asserted-by":"crossref","unstructured":"Zheng, C., Zhu, S., Mendieta, M., Yang, T., Chen, C., Ding, Z.: 3d human pose estimation with spatial and temporal transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 11656\u201311665 (2021)","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"2204_CR49","doi-asserted-by":"crossref","unstructured":"Shan, W., Lu, H., Wang, S., Zhang, X., Gao, W.: Improving robustness and accuracy via relative information encoding in 3d human pose estimation. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 3446\u20133454 (2021)","DOI":"10.1145\/3474085.3475504"},{"issue":"1","key":"2204_CR50","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1109\/TCSVT.2021.3057267","volume":"32","author":"T Chen","year":"2021","unstructured":"Chen, T., Fang, C., Shen, X., Zhu, Y., Chen, Z., Luo, J.: Anatomy-aware 3d human pose estimation with bone-based pose decomposition. IEEE Trans. Circuits Syst. Video Technol. 32(1), 198\u2013209 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"2204_CR51","doi-asserted-by":"crossref","unstructured":"Hu, W., Zhang, C., Zhan, F., Zhang, L., Wong, T.-T.: Conditional directed graph convolution for 3d human pose estimation. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 602\u2013611 (2021)","DOI":"10.1145\/3474085.3475219"},{"key":"2204_CR52","doi-asserted-by":"crossref","unstructured":"Zhan, Y., Li, F., Weng, R., Choi, W.: Ray3d: ray-based 3d human pose estimation for monocular absolute 3d localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13116\u201313125 (2022)","DOI":"10.1109\/CVPR52688.2022.01277"},{"key":"2204_CR53","doi-asserted-by":"publisher","first-page":"1282","DOI":"10.1109\/TMM.2022.3141231","volume":"25","author":"W Li","year":"2022","unstructured":"Li, W., Liu, H., Ding, R., Liu, M., Wang, P., Yang, W.: Exploiting temporal contexts with strided transformer for 3d human pose estimation. IEEE Trans. Multimedia 25, 1282\u20131293 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"2204_CR54","doi-asserted-by":"publisher","first-page":"4278","DOI":"10.1109\/TIP.2022.3182269","volume":"31","author":"Y Xue","year":"2022","unstructured":"Xue, Y., Chen, J., Gu, X., Ma, H., Ma, H.: Boosting monocular 3d human pose estimation with part aware attention. IEEE Trans. Image Process. 31, 4278\u20134291 (2022)","journal-title":"IEEE Trans. Image Process."},{"key":"2204_CR55","doi-asserted-by":"crossref","unstructured":"Shan, W., Liu, Z., Zhang, X., Wang, S., Ma, S., Gao, W.: P-stmo: Pre-trained spatial temporal many-to-one model for 3d human pose estimation. In: European Conference on Computer Vision, pp. 461\u2013478 (2022). Springer","DOI":"10.1007\/978-3-031-20065-6_27"},{"key":"2204_CR56","doi-asserted-by":"crossref","unstructured":"Zhang, J., Tu, Z., Yang, J., Chen, Y., Yuan, J.: Mixste: Seq2seq mixed spatio-temporal encoder for 3d human pose estimation in video, in 2022 ieee. In: CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13222\u201313232 (2022)","DOI":"10.1109\/CVPR52688.2022.01288"},{"key":"2204_CR57","doi-asserted-by":"publisher","first-page":"8712","DOI":"10.1109\/TMM.2023.3240455","volume":"25","author":"Z Tang","year":"2023","unstructured":"Tang, Z., Li, J., Hao, Y., Hong, R.: Mlp-jcg: Multi-layer perceptron with joint-coordinate gating for efficient 3d human pose estimation. IEEE Trans. Multimed. 25, 8712\u20138724 (2023). https:\/\/doi.org\/10.1109\/TMM.2023.3240455","journal-title":"IEEE Trans. Multimed."},{"key":"2204_CR58","doi-asserted-by":"crossref","unstructured":"Einfalt, M., Ludwig, K., Lienhart, R.: Uplift and upsample: Efficient 3d human pose estimation with uplifting transformers. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2903\u20132913 (2023)","DOI":"10.1109\/WACV56688.2023.00292"},{"key":"2204_CR59","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2023.104863","volume":"140","author":"X Liu","year":"2023","unstructured":"Liu, X., Tang, H.: Strformer: Spatial-temporal-retemporal transformer for 3d human pose estimation. Image Vis. Comput. 140, 104863 (2023)","journal-title":"Image Vis. Comput."},{"key":"2204_CR60","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.110116","volume":"147","author":"S Du","year":"2024","unstructured":"Du, S., Yuan, Z., Lai, P., Ikenaga, T.: Joypose: Jointly learning evolutionary data augmentation and anatomy-aware global-local representation for 3d human pose estimation. Pattern Recogn. 147, 110116 (2024)","journal-title":"Pattern Recogn."},{"key":"2204_CR61","doi-asserted-by":"crossref","unstructured":"Tang, Z., Qiu, Z., Hao, Y., Hong, R., Yao, T.: 3d human pose estimation with spatio-temporal criss-cross attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4790\u20134799 (2023)","DOI":"10.1109\/CVPR52729.2023.00464"},{"key":"2204_CR62","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Zheng, C., Liu, M., Wang, P., Chen, C.: Poseformerv2: Exploring frequency domain for efficient and robust 3d human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8877\u20138886 (2023)","DOI":"10.1109\/CVPR52729.2023.00857"},{"key":"2204_CR63","doi-asserted-by":"crossref","unstructured":"Peng, Q., Zheng, C., Chen, C.: A dual-augmentor framework for domain generalization in 3d human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2240\u20132249 (2024)","DOI":"10.1109\/CVPR52733.2024.00218"},{"key":"2204_CR64","doi-asserted-by":"crossref","unstructured":"Yu, B.X., Zhang, Z., Liu, Y., Zhong, S.-h., Liu, Y., Chen, C.W.: Gla-gcn: Global-local adaptive graph convolutional network for 3d human pose estimation from monocular video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8818\u20138829 (2023)","DOI":"10.1109\/ICCV51070.2023.00810"},{"key":"2204_CR65","doi-asserted-by":"crossref","unstructured":"Cai, J., Liu, M., Liu, H., Zhou, S., Li, W.: Nanohtnet: Nano human topology network for efficient 3d human pose estimation. IEEE Transactions on Image Processing (2025)","DOI":"10.1109\/TIP.2025.3608662"},{"key":"2204_CR66","doi-asserted-by":"crossref","unstructured":"Wei, M., Xie, X., Zhong, Y., Shi, G.: Learning pyramid-structured long-range dependencies for 3d human pose estimation. IEEE Transactions on Multimedia (2025)","DOI":"10.1109\/TMM.2025.3535349"},{"key":"2204_CR67","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110925","volume":"158","author":"W Li","year":"2025","unstructured":"Li, W., Liu, M., Liu, H., Guo, T., Wang, T., Tang, H., Sebe, N.: Graphmlp: A graph mlp-like architecture for 3d human pose estimation. Pattern Recogn. 158, 110925 (2025)","journal-title":"Pattern Recogn."},{"key":"2204_CR68","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110446","volume":"152","author":"Z Chen","year":"2024","unstructured":"Chen, Z., Dai, J., Bai, J., Pan, J.: Dgformer: Dynamic graph transformer for 3d human pose estimation. Pattern Recogn. 152, 110446 (2024)","journal-title":"Pattern Recogn."},{"key":"2204_CR69","doi-asserted-by":"crossref","unstructured":"Sharma, S., Varigonda, P.T., Bindal, P., Sharma, A., Jain, A.: Monocular 3d human pose estimation by generation and ordinal ranking. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2325\u20132334 (2019)","DOI":"10.1109\/ICCV.2019.00241"},{"key":"2204_CR70","doi-asserted-by":"crossref","unstructured":"Li, C., Lee, G.H.: Weakly supervised generative network for multiple 3d human pose hypotheses (2020). arXiv preprint arXiv:2008.05770","DOI":"10.5244\/C.34.88"},{"key":"2204_CR71","doi-asserted-by":"crossref","unstructured":"Oikarinen, T., Hannah, D., Kazerounian, S.: Graphmdn: Leveraging graph structure and deep learning to solve inverse problems. In: 2021 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20139 (2021). IEEE","DOI":"10.1109\/IJCNN52387.2021.9534301"},{"key":"2204_CR72","doi-asserted-by":"crossref","unstructured":"Wehrbein, T., Rudolph, M., Rosenhahn, B., Wandt, B.: Probabilistic monocular 3d human pose estimation with normalizing flows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11199\u201311208 (2021)","DOI":"10.1109\/ICCV48922.2021.01101"},{"key":"2204_CR73","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109631","volume":"141","author":"W Li","year":"2023","unstructured":"Li, W., Liu, H., Tang, H., Wang, P.: Multi-hypothesis representation learning for transformer-based 3d human pose estimation. Pattern Recogn. 141, 109631 (2023)","journal-title":"Pattern Recogn."},{"key":"2204_CR74","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvcir.2023.103890","volume":"95","author":"X Xiang","year":"2023","unstructured":"Xiang, X., Zhang, K., Qiao, Y., El Saddik, A.: Emhiformer: An enhanced multi-hypothesis interaction transformer for 3d human pose estimation in video. J. Vis. Commun. Image Represent. 95, 103890 (2023)","journal-title":"J. Vis. Commun. Image Represent."},{"key":"2204_CR75","doi-asserted-by":"crossref","unstructured":"Li, C., Lee, G.H.: Generating multiple hypotheses for 3d human pose estimation with mixture density network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9887\u20139895 (2019)","DOI":"10.1109\/CVPR.2019.01012"},{"key":"2204_CR76","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, Z., Peng, Y., Zhang, Z., Yu, G., Sun, J.: Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7103\u20137112 (2018)","DOI":"10.1109\/CVPR.2018.00742"}],"updated-by":[{"DOI":"10.1007\/s00530-026-02336-4","type":"correction","label":"Correction","source":"publisher","updated":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T00:00:00Z","timestamp":1778025600000}}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02204-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-02204-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-02204-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T05:49:27Z","timestamp":1778046567000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-02204-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,9]]},"references-count":76,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["2204"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-02204-7","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,2,9]]},"assertion":[{"value":"17 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 February 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 March 2026","order":5,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Update","order":6,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The original online version of this article was revised to update corresponding author name.","order":7,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 May 2026","order":8,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Correction","order":9,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"A Correction to this paper has been published:","order":10,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"https:\/\/doi.org\/10.1007\/s00530-026-02336-4","URL":"https:\/\/doi.org\/10.1007\/s00530-026-02336-4","order":11,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The author declares that they have no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"149"}}