{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T12:02:44Z","timestamp":1784894564686,"version":"3.55.0"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Science and Technology Innovation Team Training Plan Project of Tangshan City","award":["18130221A"],"award-info":[{"award-number":["18130221A"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00530-026-02487-4","type":"journal-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T14:45:15Z","timestamp":1783781115000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["FLTGFormer: a dual-stream transformer-GCN network enhanced by frequency loss for monocular 3D human pose estimation"],"prefix":"10.1007","volume":"32","author":[{"given":"Xuejing","family":"Gu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zibo","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Congzhe","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,11]]},"reference":[{"key":"2487_CR1","doi-asserted-by":"publisher","first-page":"128049","DOI":"10.1016\/j.neucom.2024.128049","volume":"596","author":"Y Liu","year":"2024","unstructured":"Liu, Y., Qiu, C., Zhang, Z.: Deep learning for 3D human pose estimation and mesh recovery: A survey. Neurocomputing. 596, 128049 (2024)","journal-title":"Neurocomputing"},{"key":"2487_CR2","doi-asserted-by":"crossref","unstructured":"Huo, R., Gao, Q., Qi, J., Ju, Z.: 3d human pose estimation in video for human-computer\/robot interaction. In: Proceedings of the 16th International Conference on Intelligent Robotics and Applications (ICIRA), vol. 14273, pp. 176\u2013187. Springer (2023)","DOI":"10.1007\/978-981-99-6498-7_16"},{"key":"2487_CR3","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1007\/978-981-19-1520-8_43","volume-title":"Pattern Recognition and Data Analysis with Applications","author":"P Sharma","year":"2022","unstructured":"Sharma, P., Shah, B.B., Prakash, C.: A pilot study on human pose estimation for sports analysis. In: Gupta, D., Goswami, R.S., Banerjee, S., Tanveer, M., Pachori, R.B. (eds.) Pattern Recognition and Data Analysis with Applications, pp. 533\u2013544. Springer, Singapore (2022)"},{"key":"2487_CR4","doi-asserted-by":"publisher","first-page":"e2574","DOI":"10.7717\/peerj-cs.2574","volume":"11","author":"S Salisu","year":"2025","unstructured":"Salisu, S., Danyaro, K.U., Nasser, M., Hayder, I.M., Younis, H.A.: Review of models for estimating 3D human pose using deep learning. PeerJ Comput. Sci. 11, e2574 (2025)","journal-title":"PeerJ Comput. Sci."},{"key":"2487_CR5","doi-asserted-by":"publisher","first-page":"103275","DOI":"10.1016\/j.cviu.2021.103275","volume":"212","author":"Y Desmarais","year":"2021","unstructured":"Desmarais, Y., Mottet, D., Slangen, P., Montesinos, P.: A review of 3D human pose estimation algorithms for markerless motion capture. Comput. Vis. Image Underst. 212, 103275 (2021)","journal-title":"Comput. Vis. Image Underst."},{"key":"2487_CR6","doi-asserted-by":"crossref","unstructured":"Fu, Y., Huang, C., Li, J., Kong, H., Tian, Y., Li, H., Zhang, Z.: HDiffTG: A lightweight hybrid diffusion-transformer-GCN architecture for 3D human pose estimation. In: Proceedings of the International Joint Conference on Neural Networks (IJCNN), pp. 1\u20138. IEEE (2025)","DOI":"10.1109\/IJCNN64981.2025.11227571"},{"key":"2487_CR7","unstructured":"Ye, M., Yang, L., Zhu, H., Zheng, Z., Wang, X., Lo, Y.: Dual-stream transformer-gcn model with contextualized representations learning for monocular 3d human pose estimation. arXiv preprint arXiv:250401764 (2025)"},{"key":"2487_CR8","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30, pp. 5998\u20136008 (2017)"},{"key":"2487_CR9","doi-asserted-by":"crossref","unstructured":"Mehraban, S., Adeli, V., Taati, B.: Motionagformer: Enhancing 3d human pose estimation with a transformer-gcnformer network. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6920\u20136930 (2024)","DOI":"10.1109\/WACV57701.2024.00677"},{"key":"2487_CR10","unstructured":"Kipf, T.N., Welling, M.: Semi-supervised classification with graph convolutional networks. In: International Conference on Learning Representations (ICLR) (2017)"},{"key":"2487_CR11","unstructured":"Meng, W., Luo, Y., Li, X., Jiang, D., Zhang, Z.: PolaFormer: Polarity-aware linear attention for vision transformers. In: International Conference on Learning Representations (ICLR) (2025)"},{"key":"2487_CR12","unstructured":"Liu, Y., Hu, T., Zhang, H., Wu, H., Wang, S., Ma, L., Long, M.: iTransformer: Inverted transformers are effective for time series forecasting. In: International Conference on Learning Representations (ICLR) (2024)"},{"key":"2487_CR13","unstructured":"Nie, Y., Nguyen, N.H., Sinthong, P., Kalagnanam, J.: A time series is worth 64 words: Long-term forecasting with transformers. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"2487_CR14","doi-asserted-by":"crossref","unstructured":"Wang, H., Pan, L., Shen, Y., Chen, Z., Yang, D., Yang, Y., Zhang, S., Liu, X., Li, H., Tao, D.: FreDF: Learning to forecast in the frequency domain. In: International Conference on Learning Representations (ICLR) (2025)","DOI":"10.1201\/9781003612742-3"},{"issue":"7","key":"2487_CR15","doi-asserted-by":"publisher","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","volume":"36","author":"C Ionescu","year":"2014","unstructured":"Ionescu, C., Papava, D., Olaru, V., Sminchisescu, C.: Human3.6\u00a0M: large scale datasets and predictive methods for 3D human sensing in natural environments. IEEE Trans. Pattern Anal. Mach. Intell. 36(7), 1325\u20131339 (2014)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2487_CR16","doi-asserted-by":"publisher","first-page":"703","DOI":"10.1007\/s11263-020-01398-9","volume":"129","author":"Z Zhang","year":"2021","unstructured":"Zhang, Z., Wang, C., Qiu, W., Qin, W., Zeng, W.: AdaFuse: Adaptive multiview fusion for accurate human pose estimation in the wild. Int. J. Comput. Vis. 129, 703\u2013718 (2021)","journal-title":"Int. J. Comput. Vis."},{"key":"2487_CR17","doi-asserted-by":"crossref","unstructured":"Sun, X., Xiao, B., Wei, F., Liang, S., Wei, Y.: Integral human pose regression. In: European Conference on Computer Vision, pp. 529\u2013545. Springer (2018)","DOI":"10.1007\/978-3-030-01231-1_33"},{"key":"2487_CR18","doi-asserted-by":"crossref","unstructured":"Martinez, J., Hossain, R., Romero, J., Little, J.J.: A simple yet effective baseline for 3D human pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2640\u20132649 (2017)","DOI":"10.1109\/ICCV.2017.288"},{"key":"2487_CR19","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, Z., Peng, Y., Zhang, Z., Yu, G., Sun, J.: Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7103\u20137112. IEEE (2018)","DOI":"10.1109\/CVPR.2018.00742"},{"key":"2487_CR20","doi-asserted-by":"crossref","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation. In: European Conference on Computer Vision, pp. 483\u2013499. Springer (2016)","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"2487_CR21","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5693\u20135703 (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"key":"2487_CR22","doi-asserted-by":"crossref","unstructured":"Zheng, C., Zhu, S., Mendieta, M., Yang, T., Chen, C., Ding, Z.: 3D human pose estimation with spatial and temporal transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11656\u201311665 (2021)","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"2487_CR23","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Zheng, C., Liu, M., Wang, P., Chen, C.: PoseFormerV2: Exploring frequency domain for efficient and robust 3D human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8877\u20138886 (2023)","DOI":"10.1109\/CVPR52729.2023.00857"},{"key":"2487_CR24","doi-asserted-by":"crossref","unstructured":"Li, W., Liu, H., Tang, H., Wang, P., Van Gool, L.: MHFormer: Multi-hypothesis transformer for 3D human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13147\u201313156 (2022)","DOI":"10.1109\/CVPR52688.2022.01280"},{"key":"2487_CR25","doi-asserted-by":"crossref","unstructured":"Shan, W., Liu, Z., Zhang, X., Wang, S., Ma, S., Gao, W.: P-STMO: Pre-trained spatial temporal many-to-one model for 3D human pose estimation. In: European Conference on Computer Vision, pp. 461\u2013478. Springer (2022)","DOI":"10.1007\/978-3-031-20065-6_27"},{"key":"2487_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, J., Tu, Z., Yang, J., Chen, Y., Yuan, J.: MixSTE: Seq2seq mixed spatio-temporal encoder for 3D human pose estimation in video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13232\u201313242 (2022)","DOI":"10.1109\/CVPR52688.2022.01288"},{"key":"2487_CR27","doi-asserted-by":"crossref","unstructured":"Tang, Z., Qiu, Z., Hao, Y., Hong, R., Yao, T.: 3D human pose estimation with spatio-temporal criss-cross attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4790\u20134799 (2023)","DOI":"10.1109\/CVPR52729.2023.00464"},{"key":"2487_CR28","doi-asserted-by":"crossref","unstructured":"Zhu, W., Ma, X., Liu, Z., Liu, L., Wu, W., Wang, Y.: MotionBERT: A unified perspective on learning human motion representations. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15085\u201315099 (2023)","DOI":"10.1109\/ICCV51070.2023.01385"},{"key":"2487_CR29","doi-asserted-by":"crossref","unstructured":"Liu, J., Liu, M., Liu, H., Li, W.: TCPFormer: Learning temporal correlation with implicit pose proxy for 3D human pose estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, pp. 5478\u20135486. (2025)","DOI":"10.1609\/aaai.v39i5.32583"},{"key":"2487_CR30","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1007\/s00530-024-01451-4","volume":"30","author":"S Arthanari","year":"2024","unstructured":"Arthanari, S., Jeong, J.H., Joo, Y.H.: Exploring multi-level transformers with feature frame padding network for 3D human pose estimation. Multimedia Syst. 30, 243 (2024)","journal-title":"Multimedia Syst."},{"key":"2487_CR31","doi-asserted-by":"publisher","first-page":"26417","DOI":"10.1007\/s11042-024-20179-x","volume":"84","author":"S Arthanari","year":"2025","unstructured":"Arthanari, S., Jeong, J.H., Joo, Y.H.: Exploiting multi-transformer encoder with multiple-hypothesis aggregation via diffusion model for 3D human pose estimation. Multimed Tools Appl. 84, 26417\u201326445 (2025)","journal-title":"Multimed Tools Appl."},{"key":"2487_CR32","doi-asserted-by":"publisher","first-page":"149","DOI":"10.1007\/s00530-025-02204-7","volume":"32","author":"S Arthanari","year":"2026","unstructured":"Arthanari, S., Moorthy, S., Jeong, J.H.: Exploring multi-transformer with fine-grained prompt-driven coupled with diffusion model for 3D human pose estimation. Multimedia Syst. 32, 149 (2026)","journal-title":"Multimedia Syst."},{"key":"2487_CR33","doi-asserted-by":"crossref","unstructured":"Zhao, L., Peng, X., Tian, Y., Kapadia, M., Metaxas, D.N.: Semantic graph convolutional networks for 3D human pose regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3425\u20133435 (2019)","DOI":"10.1109\/CVPR.2019.00354"},{"issue":"3","key":"2487_CR34","doi-asserted-by":"publisher","first-page":"1429","DOI":"10.1109\/TPAMI.2020.3019139","volume":"44","author":"H Ci","year":"2022","unstructured":"Ci, H., Ma, X., Wang, C., Wang, Y.: Locally connected network for monocular 3D human pose estimation. IEEE Trans. Pattern Anal. Mach. Intell. 44(3), 1429\u20131442 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2487_CR35","doi-asserted-by":"crossref","unstructured":"Yu, B.X.B., Zhang, Z., Liu, Y., Zhong, S., Liu, Y., Chen, C.W.: GLA-GCN: Global-local adaptive graph convolutional network for 3D human pose estimation from monocular video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8818\u20138829 (2023)","DOI":"10.1109\/ICCV51070.2023.00810"},{"key":"2487_CR36","doi-asserted-by":"crossref","unstructured":"Peng, J., Zhou, Y., Mok, P.Y.: KTPFormer: Kinematics and trajectory prior knowledge-enhanced transformer for 3D human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1123\u20131132 (2024)","DOI":"10.1109\/CVPR52733.2024.00113"},{"key":"2487_CR37","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1007\/s00530-025-02074-z","volume":"32","author":"F Su","year":"2026","unstructured":"Su, F., Wang, J., STGSFormer: A 3d Human Pose Estimation Model That Integrates GCN and Self-attention in the Spatio-temporal Domain. Multimed. Syst. 32, 30 (2026)","journal-title":"Multimed. Syst."},{"key":"2487_CR38","doi-asserted-by":"publisher","first-page":"111955","DOI":"10.1016\/j.knosys.2024.111955","volume":"297","author":"Y Zhang","year":"2024","unstructured":"Zhang, Y., Sun, H.: Optimizing intrinsic representation for tracking. Knowl. Based Syst. 297, 111955 (2024)","journal-title":"Knowl. Based Syst."},{"key":"2487_CR39","doi-asserted-by":"publisher","first-page":"106839","DOI":"10.1016\/j.neunet.2024.106839","volume":"181","author":"Y Zhang","year":"2025","unstructured":"Zhang, Y., Pan, H., Wang, J.: Enabling deformation slack in tracking with temporally even correlation filters. Neural Netw. 181, 106839 (2025)","journal-title":"Neural Netw."},{"key":"2487_CR40","doi-asserted-by":"publisher","first-page":"107587","DOI":"10.1016\/j.neunet.2025.107587","volume":"189","author":"Y Zhang","year":"2025","unstructured":"Zhang, Y., Sun, H.: Decoding split-frequency representation for cross-scale tracking. Neural Netw. 189, 107587 (2025)","journal-title":"Neural Netw."},{"key":"2487_CR41","doi-asserted-by":"crossref","unstructured":"Mehta, D., Rhodin, H., Casas, D., Fua, P., Sotnychenko, O., Xu, W., Theobalt, C.: Monocular 3D human pose estimation in the wild using improved CNN supervision. In: International Conference on 3D Vision (3DV), pp. 506\u2013516. IEEE (2017)","DOI":"10.1109\/3DV.2017.00064"},{"key":"2487_CR42","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (ICLR) (2019)"},{"key":"2487_CR43","doi-asserted-by":"crossref","unstructured":"von Marcard, T., Henschel, R., Black, M.J., Rosenhahn, B., Pons-Moll, G.: Recovering accurate 3D human pose in the wild using IMUs and a moving camera. In: European Conference on Computer Vision, pp. 614\u2013631. Springer (2018)","DOI":"10.1007\/978-3-030-01249-6_37"},{"key":"2487_CR44","first-page":"1","volume":"72","author":"Y Zhang","year":"2023","unstructured":"Zhang, Y., Pan, H., Wang, J., Sun, W.: Facing completely occluded short-term tracking based on correlation filters. IEEE Trans. Instrum. Meas. 72, 1\u201315 (2023)","journal-title":"IEEE Trans. Instrum. Meas."},{"issue":"8","key":"2487_CR45","doi-asserted-by":"publisher","first-page":"3747","DOI":"10.1109\/TCYB.2025.3574326","volume":"55","author":"H Sun","year":"2025","unstructured":"Sun, H., Zhang, Y., Zhang, H., Qiu, X., Rudas, I.J.: Learning distance constrained transformation for video tracking in car-following. IEEE Trans. Cybern. 55(8), 3747\u20133759 (2025)","journal-title":"IEEE Trans. Cybern"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02487-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02487-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02487-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T11:31:53Z","timestamp":1784892713000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02487-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":45,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2487"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02487-4","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"3 March 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare no conflict of interest.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"420"}}