{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:09:06Z","timestamp":1784218146450,"version":"3.55.0"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"15","license":[{"start":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T00:00:00Z","timestamp":1758326400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T00:00:00Z","timestamp":1758326400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1007\/s00371-025-04175-2","type":"journal-article","created":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T05:36:29Z","timestamp":1758346589000},"page":"12603-12619","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing 3D human pose estimation via spatio-temporal dual-stream fusion"],"prefix":"10.1007","volume":"41","author":[{"given":"Junfen","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuhan","family":"Cui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bojun","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,20]]},"reference":[{"key":"4175_CR1","doi-asserted-by":"publisher","unstructured":"Cheng, Y., Yi, P., Liu, R., Dong, J., Zhou, D., Zhang, Q.: Human-robot interaction method combining human pose estimation and motion intention recognition. In: 2021 IEEE 24th International Conference on Computer Supported Cooperative Work in Design (CSCWD), pp. 958\u2013963 (2021). https:\/\/doi.org\/10.1109\/CSCWD49262.2021.9437772","DOI":"10.1109\/CSCWD49262.2021.9437772"},{"key":"4175_CR2","doi-asserted-by":"publisher","first-page":"103448","DOI":"10.1016\/j.jvcir.2022.103448","volume":"83","author":"M Hassanin","year":"2022","unstructured":"Hassanin, M., Radwan, I., Khan, S., Tahtali, M.: Learning discriminative representations for multi-label image recognition. J. Vis. Commun. Image Represent. 83, 103448 (2022)","journal-title":"J. Vis. Commun. Image Represent."},{"key":"4175_CR3","unstructured":"Ahad, M.A.R., Antar, A.D., Shahid, O.: Vision-based action understanding for assistive healthcare: A short review. In: CVPR Workshops, 2 (2019)"},{"key":"4175_CR4","doi-asserted-by":"crossref","unstructured":"Ahmedt-Aristizabal, D., Nguyen, K., Denman, S., Sarfraz, M.S., Sridharan, S., Dionisio, S., Fookes, C.: Vision-based mouth motion analysis in epilepsy: A 3d perspective. In: 2019 41st Annual International Conference of the IEEE Engineering in Medicine and Biology Society (EMBC), pp. 1625\u20131629. IEEE (2019)","DOI":"10.1109\/EMBC.2019.8857656"},{"key":"4175_CR5","doi-asserted-by":"crossref","unstructured":"Li, S., Chan, A.B.: 3d human pose estimation from monocular images with deep convolutional neural network. In: Asian Conference on Computer Vision, pp. 332\u2013347. Springer (2014)","DOI":"10.1007\/978-3-319-16808-1_23"},{"key":"4175_CR6","doi-asserted-by":"crossref","unstructured":"Moon, G., Chang, J.Y., Lee, K.M.: Camera distance-aware top-down approach for 3d multi-person pose estimation from a single RGB image. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10133\u201310142 (2019)","DOI":"10.1109\/ICCV.2019.01023"},{"issue":"1","key":"4175_CR7","doi-asserted-by":"publisher","first-page":"429","DOI":"10.1007\/s00371-021-02339-4","volume":"39","author":"K Wang","year":"2023","unstructured":"Wang, K., Zhang, G., Yang, J.: 3d human pose and shape estimation with dense correspondence from a single depth image. Vis. Comput. 39(1), 429\u2013441 (2023)","journal-title":"Vis. Comput."},{"key":"4175_CR8","doi-asserted-by":"crossref","unstructured":"Martinez, J., Hossain, R., Romero, J., Little, J.J.: A simple yet effective baseline for 3d human pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2640\u20132649 (2017)","DOI":"10.1109\/ICCV.2017.288"},{"key":"4175_CR9","doi-asserted-by":"crossref","unstructured":"Fang, H.-S., Xu, Y., Wang, W., Liu, X., Zhu, S.-C.: Learning pose grammar to encode human body configuration for 3d pose estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 32 (2018)","DOI":"10.1609\/aaai.v32i1.12270"},{"key":"4175_CR10","doi-asserted-by":"crossref","unstructured":"Gong, K., Zhang, J., Feng, J.: PoseAug: A differentiable pose augmentation framework for 3d human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8575\u20138584 (2021)","DOI":"10.1109\/CVPR46437.2021.00847"},{"key":"4175_CR11","doi-asserted-by":"crossref","unstructured":"Ci, H., Wang, C., Ma, X., Wang, Y.: Optimizing network structure for 3d human pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2262\u20132271 (2019)","DOI":"10.1109\/ICCV.2019.00235"},{"key":"4175_CR12","doi-asserted-by":"crossref","unstructured":"Li, W., Liu, H., Tang, H., Wang, P., Van\u00a0Gool, L.: Mhformer: Multi-hypothesis transformer for 3d human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13147\u201313156 (2022)","DOI":"10.1109\/CVPR52688.2022.01280"},{"key":"4175_CR13","doi-asserted-by":"crossref","unstructured":"Ma, X., Su, J., Wang, C., Ci, H., Wang, Y.: Context modeling in 3d human pose estimation: A unified perspective. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6238\u20136247 (2021)","DOI":"10.1109\/CVPR46437.2021.00617"},{"key":"4175_CR14","doi-asserted-by":"crossref","unstructured":"Mehraban, S., Adeli, V., Taati, B.: Motionagformer: Enhancing 3d human pose estimation with a transformer-gcnformer network. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6920\u20136930 (2024)","DOI":"10.1109\/WACV57701.2024.00677"},{"key":"4175_CR15","doi-asserted-by":"crossref","unstructured":"Yu, B.X., Zhang, Z., Liu, Y., Zhong, S.-h., Liu, Y., Chen, C.W.: Gla-gcn: Global-local adaptive graph convolutional network for 3d human pose estimation from monocular video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8818\u20138829 (2023)","DOI":"10.1109\/ICCV51070.2023.00810"},{"key":"4175_CR16","doi-asserted-by":"crossref","unstructured":"Zheng, C., Zhu, S., Mendieta, M., Yang, T., Chen, C., Ding, Z.: 3d human pose estimation with spatial and temporal transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11656\u201311665 (2021)","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"4175_CR17","doi-asserted-by":"crossref","unstructured":"Peng, K., Yin, C., Zheng, J., Liu, R., Schneider, D., Zhang, J., Yang, K., Sarfraz, M.S., Stiefelhagen, R., Roitberg, A.: Navigating open set scenarios for skeleton-based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 4487\u20134496 (2024)","DOI":"10.1609\/aaai.v38i5.28247"},{"key":"4175_CR18","doi-asserted-by":"crossref","unstructured":"Xu, Y., Peng, K., Wen, D., Liu, R., Zheng, J., Chen, Y., Zhang, J., Roitberg, A., Yang, K., Stiefelhagen, R.: Skeleton-based human action recognition with noisy labels. In: 2024 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 4716\u20134723. IEEE (2024)","DOI":"10.1109\/IROS58592.2024.10801681"},{"key":"4175_CR19","doi-asserted-by":"publisher","first-page":"1489","DOI":"10.1109\/TMM.2023.3235300","volume":"25","author":"K Peng","year":"2023","unstructured":"Peng, K., Roitberg, A., Yang, K., Zhang, J., Stiefelhagen, R.: Delving deep into one-shot skeleton-based action recognition with diverse occlusions. IEEE Trans. Multimedia 25, 1489\u20131504 (2023)","journal-title":"IEEE Trans. Multimedia"},{"key":"4175_CR20","doi-asserted-by":"publisher","unstructured":"Zhu, H., Wei, P., Xu, Z.: A spatio-temporal enhanced graph-transformer autoencoder embedded pose for anomaly detection. IET Comput. Vision 18(3), 405\u2013419 (2024). https:\/\/doi.org\/10.1049\/cvi2.12257https:\/\/ietresearch.onlinelibrary.wiley.com\/doi\/pdf\/10.1049\/cvi2.12257","DOI":"10.1049\/cvi2.12257"},{"key":"4175_CR21","doi-asserted-by":"crossref","unstructured":"Zhao, L., Peng, X., Tian, Y., Kapadia, M., Metaxas, D.N.: Semantic graph convolutional networks for 3d human pose regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3425\u20133435 (2019)","DOI":"10.1109\/CVPR.2019.00354"},{"issue":"8","key":"4175_CR22","doi-asserted-by":"publisher","first-page":"5883","DOI":"10.1007\/s00371-023-03142-z","volume":"40","author":"R Jia","year":"2024","unstructured":"Jia, R., Yang, H., Zhao, L., Wu, X., Zhang, Y.: Mpa-gnet: multi-scale parallel adaptive graph network for 3d human pose estimation. Vis. Comput. 40(8), 5883\u20135899 (2024)","journal-title":"Vis. Comput."},{"issue":"5","key":"4175_CR23","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1007\/s00530-024-01451-4","volume":"30","author":"S Arthanari","year":"2024","unstructured":"Arthanari, S., Jeong, J.H., Joo, Y.H.: Exploring multi-level transformers with feature frame padding network for 3d human pose estimation. Multimedia Syst. 30(5), 243 (2024)","journal-title":"Multimedia Syst."},{"key":"4175_CR24","unstructured":"Qian, X., Tang, Y., Zhang, N., Han, M., Xiao, J., Huang, M.-C., Lin, R.-S.: Hstformer: Hierarchical spatial-temporal transformers for 3d human pose estimation. arXiv preprint arXiv:2301.07322 (2023)"},{"key":"4175_CR25","doi-asserted-by":"crossref","unstructured":"Zhao, W., Wang, W., Tian, Y.: Graformer: Graph-oriented transformer for 3d pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20438\u201320447 (2022)","DOI":"10.1109\/CVPR52688.2022.01979"},{"key":"4175_CR26","doi-asserted-by":"crossref","unstructured":"Li, H., Shi, B., Dai, W., Zheng, H., Wang, B., Sun, Y., Guo, M., Li, C., Zou, J., Xiong, H.: Pose-oriented transformer with uncertainty-guided refinement for 2d-to-3d human pose estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 1296\u20131304 (2023)","DOI":"10.1609\/aaai.v37i1.25213"},{"key":"4175_CR27","doi-asserted-by":"crossref","unstructured":"Gong, J., Foo, L.G., Fan, Z., Ke, Q., Rahmani, H., Liu, J.: Diffpose: Toward more reliable 3d pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13041\u201313051 (2023)","DOI":"10.1109\/CVPR52729.2023.01253"},{"key":"4175_CR28","unstructured":"Yue, K., Sun, M., Yuan, Y., Zhou, F., Ding, E., Xu, F.: Compact generalized non-local network. Advances in neural information processing systems, vol. 31 (2018)"},{"key":"4175_CR29","doi-asserted-by":"crossref","unstructured":"Zhang, J., Tu, Z., Yang, J., Chen, Y., Yuan, J.: Mixste: Seq2seq mixed spatio-temporal encoder for 3d human pose estimation in video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13232\u201313242 (2022)","DOI":"10.1109\/CVPR52688.2022.01288"},{"key":"4175_CR30","doi-asserted-by":"crossref","unstructured":"Pavllo, D., Feichtenhofer, C., Grangier, D., Auli, M.: 3d human pose estimation in video with temporal convolutions and semi-supervised training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7753\u20137762 (2019)","DOI":"10.1109\/CVPR.2019.00794"},{"issue":"7","key":"4175_CR31","doi-asserted-by":"publisher","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","volume":"36","author":"C Ionescu","year":"2013","unstructured":"Ionescu, C., Papava, D., Olaru, V., Sminchisescu, C.: Human 3.6 m: large scale datasets and predictive methods for 3d human sensing in natural environments. IEEE Trans. Pattern Anal. Mach. Intell. 36(7), 1325\u20131339 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4175_CR32","doi-asserted-by":"crossref","unstructured":"Mehta, D., Rhodin, H., Casas, D., Fua, P., Sotnychenko, O., Xu, W., Theobalt, C.: Monocular 3d human pose estimation in the wild using improved cnn supervision. In: 2017 International Conference on 3D Vision (3DV), pp. 506\u2013516. IEEE (2017)","DOI":"10.1109\/3DV.2017.00064"},{"key":"4175_CR33","doi-asserted-by":"crossref","unstructured":"Tang, Z., Qiu, Z., Hao, Y., Hong, R., Yao, T.: 3d human pose estimation with spatio-temporal criss-cross attention. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4790\u20134799 (2023)","DOI":"10.1109\/CVPR52729.2023.00464"},{"key":"4175_CR34","doi-asserted-by":"crossref","unstructured":"Zhao, Q., Zheng, C., Liu, M., Wang, P., Chen, C.: Poseformerv2: Exploring frequency domain for efficient and robust 3d human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8877\u20138886 (2023)","DOI":"10.1109\/CVPR52729.2023.00857"},{"key":"4175_CR35","doi-asserted-by":"crossref","unstructured":"Zhu, W., Ma, X., Liu, Z., Liu, L., Wu, W., Wang, Y.: Motionbert: A unified perspective on learning human motion representations. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15085\u201315099 (2023)","DOI":"10.1109\/ICCV51070.2023.01385"},{"key":"4175_CR36","doi-asserted-by":"crossref","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part VIII 14, pp. 483\u2013499. Springer (2016)","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"4175_CR37","doi-asserted-by":"crossref","unstructured":"Wang, J., Yan, S., Xiong, Y., Lin, D.: Motion guided 3d pose estimation from videos. In: European Conference on Computer Vision, pp. 764\u2013780. Springer (2020)","DOI":"10.1007\/978-3-030-58601-0_45"},{"key":"4175_CR38","doi-asserted-by":"publisher","first-page":"1282","DOI":"10.1109\/TMM.2022.3141231","volume":"25","author":"W Li","year":"2022","unstructured":"Li, W., Liu, H., Ding, R., Liu, M., Wang, P., Yang, W.: Exploiting temporal contexts with strided transformer for 3d human pose estimation. IEEE Trans. Multimedia 25, 1282\u20131293 (2022)","journal-title":"IEEE Trans. Multimedia"},{"key":"4175_CR39","doi-asserted-by":"crossref","unstructured":"Shan, W., Liu, Z., Zhang, X., Wang, S., Ma, S., Gao, W.: P-stmo: Pre-trained spatial temporal many-to-one model for 3d human pose estimation. In: European Conference on Computer Vision, pp. 461\u2013478. Springer (2022)","DOI":"10.1007\/978-3-031-20065-6_27"},{"key":"4175_CR40","doi-asserted-by":"crossref","unstructured":"Peng, J., Zhou, Y., Mok, P.: Ktpformer: Kinematics and trajectory prior knowledge-enhanced transformer for 3d human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1123\u20131132 (2024)","DOI":"10.1109\/CVPR52733.2024.00113"},{"key":"4175_CR41","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Zisserman, A.: Convolutional two-stream network fusion for video action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1933\u20131941 (2016)","DOI":"10.1109\/CVPR.2016.213"},{"key":"4175_CR42","unstructured":"Ye, M., Yang, L., Zhu, H., Zheng, Z., Wang, X., Lo, Y.: Dual-stream transformer-gcn model with contextualized representations learning for monocular 3d human pose estimation. arXiv preprint arXiv:2504.01764 (2025)"},{"issue":"4","key":"4175_CR43","doi-asserted-by":"publisher","first-page":"3471","DOI":"10.1109\/TKDE.2021.3125020","volume":"35","author":"G Huo","year":"2021","unstructured":"Huo, G., Zhang, Y., Gao, J., Wang, B., Hu, Y., Yin, B.: Caegcn: cross-attention fusion based enhanced graph convolutional network for clustering. IEEE Trans. Knowl. Data Eng. 35(4), 3471\u20133483 (2021)","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"4175_CR44","doi-asserted-by":"crossref","unstructured":"Bruce, X., Liu, Y., Zhang, X., Zhong, S.-H., Chan, K.C.: MMNet: a model-based multimodal network for human action recognition in RGB-d videos. IEEE Trans. Pattern Anal. Mach. Intell. 45(3), 3522\u20133538 (2022)","DOI":"10.1109\/TPAMI.2022.3177813"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04175-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-04175-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04175-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,20]],"date-time":"2025-11-20T13:16:46Z","timestamp":1763644606000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-04175-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,20]]},"references-count":44,"journal-issue":{"issue":"15","published-print":{"date-parts":[[2025,12]]}},"alternative-id":["4175"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-04175-2","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,20]]},"assertion":[{"value":"13 May 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 August 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 September 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}