{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T02:32:40Z","timestamp":1783823560696,"version":"3.55.0"},"reference-count":66,"publisher":"Springer Science and Business Media LLC","issue":"23","license":[{"start":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T00:00:00Z","timestamp":1725926400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T00:00:00Z","timestamp":1725926400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-024-20179-x","type":"journal-article","created":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T04:05:26Z","timestamp":1725941126000},"page":"26417-26445","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Exploiting multi-transformer encoder with multiple-hypothesis aggregation via diffusion model for 3D human pose estimation"],"prefix":"10.1007","volume":"84","author":[{"given":"Sathiyamoorthi","family":"Arthanari","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jae Hoon","family":"Jeong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Young Hoon","family":"Joo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,9,10]]},"reference":[{"key":"20179_CR1","doi-asserted-by":"crossref","unstructured":"Fan L, Jiang K, Zhou W, Gao Z, Luo Y (2024) 3d human pose estimation from video via multi-scale multi-level spatial temporal features. Multimed Tools Appl 1\u201320","DOI":"10.2139\/ssrn.4379238"},{"issue":"23","key":"20179_CR2","doi-asserted-by":"publisher","first-page":"32883","DOI":"10.1007\/s11042-022-13079-5","volume":"81","author":"R Gu","year":"2022","unstructured":"Gu R, Jiang Z, Wang G, McQuade K, Hwang J-N (2022) Unsupervised universal hierarchical multi-person 3d pose estimation for natural scenes. Multimed Tools Appl 81(23):32883\u201332906","journal-title":"Multimed Tools Appl"},{"issue":"8","key":"20179_CR3","doi-asserted-by":"publisher","first-page":"22995","DOI":"10.1007\/s11042-023-16369-8","volume":"83","author":"Y Liu","year":"2024","unstructured":"Liu Y, Cheng X, Ikenaga T (2024) Motion-aware and data-independent model based multi-view 3d pose refinement for volleyball spike analysis. Multimed Tools Appl 83(8):22995\u201323018","journal-title":"Multimed Tools Appl"},{"issue":"10","key":"20179_CR4","doi-asserted-by":"publisher","first-page":"6642","DOI":"10.1109\/TCSVT.2022.3177320","volume":"32","author":"L Yan","year":"2022","unstructured":"Yan L, Ma S, Wang Q, Chen Y, Zhang X, Savakis A, Liu D (2022) Video captioning using global-local representation. IEEE Transactions on Circuits and Systems for Video Technology 32(10):6642\u20136656","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"issue":"1","key":"20179_CR5","doi-asserted-by":"publisher","first-page":"393","DOI":"10.1109\/TCSVT.2022.3202574","volume":"33","author":"L Yan","year":"2022","unstructured":"Yan L, Wang Q, Ma S, Wang J, Yu C (2022) Solve the puzzle of instance segmentation in videos: a weakly supervised framework with spatio-temporal collaboration. IEEE Transactions on Circuits and Systems for Video Technology 33(1):393\u2013406","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"20179_CR6","doi-asserted-by":"crossref","unstructured":"Cai Y, Ge L, Liu J, Cai J, Cham T-J, Yuan J, Thalmann NM (2019) Exploiting spatial-temporal relationships for 3d pose estimation via graph convolutional networks. In: Proceedings of the IEEE\/CVF international conference on computer vision pp 2272\u20132281","DOI":"10.1109\/ICCV.2019.00236"},{"key":"20179_CR7","doi-asserted-by":"crossref","unstructured":"Liu R, Shen J, Wang H, Chen C, Cheung S.-c, Asari V (2020) Attention mechanism exploits temporal contexts: real-time 3d human pose reconstruction. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5064\u20135073","DOI":"10.1109\/CVPR42600.2020.00511"},{"key":"20179_CR8","doi-asserted-by":"crossref","unstructured":"Liu J, Rojas J, Li Y, Liang Z, Guan Y, Xi N, Zhu H (2021) A graph attention spatio-temporal convolutional network for 3d human pose estimation in video. In: 2021 IEEE International conference on robotics and automation (ICRA), IEEE, pp 3374\u20133380","DOI":"10.1109\/ICRA48506.2021.9561605"},{"key":"20179_CR9","doi-asserted-by":"publisher","first-page":"243","DOI":"10.1016\/j.neucom.2021.11.007","volume":"487","author":"Y Wu","year":"2022","unstructured":"Wu Y, Kong D, Wang S, Li J, Yin B (2022) Hpgcn: hierarchical poselet-guided graph convolutional network for 3d pose estimation. Neurocomputing 487:243\u2013256","journal-title":"Neurocomputing"},{"key":"20179_CR10","doi-asserted-by":"publisher","first-page":"4212","DOI":"10.1109\/TIP.2023.3275914","volume":"32","author":"MT Hassan","year":"2023","unstructured":"Hassan MT, Ben Hamza A (2023) Regular splitting graph network for 3d human pose estimation. IEEE Trans Image Process 32:4212\u20134222. https:\/\/doi.org\/10.1109\/TIP.2023.3275914","journal-title":"IEEE Trans Image Process"},{"key":"20179_CR11","doi-asserted-by":"crossref","unstructured":"Zheng C, Zhu S, Mendieta M, Yang T, Chen C, Ding Z (2021) 3d human pose estimation with spatial and temporal transformers. In: Proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp 11656\u201311665","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"20179_CR12","doi-asserted-by":"crossref","unstructured":"Zhang J, Tu Z, Yang J, Chen Y, Yuan J (2022) Mixste: Seq2seq mixed spatio-temporal encoder for 3d human pose estimation in video, in 2022 IEEE. In: CVF Conference on computer vision and pattern recognition (CVPR), pp 13222\u201313232","DOI":"10.1109\/CVPR52688.2022.01288"},{"key":"20179_CR13","doi-asserted-by":"crossref","unstructured":"Li W, Liu H, Tang H, Wang P, Van\u00a0Gool L (2022) Mhformer: multi-hypothesis transformer for 3d human pose estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 13147\u201313156","DOI":"10.1109\/CVPR52688.2022.01280"},{"key":"20179_CR14","doi-asserted-by":"publisher","first-page":"4278","DOI":"10.1109\/TIP.2022.3182269","volume":"31","author":"Y Xue","year":"2022","unstructured":"Xue Y, Chen J, Gu X, Ma H, Ma H (2022) Boosting monocular 3d human pose estimation with part aware attention. IEEE Trans Image Process 31:4278\u20134291","journal-title":"IEEE Trans Image Process"},{"key":"20179_CR15","doi-asserted-by":"publisher","first-page":"1282","DOI":"10.1109\/TMM.2022.3141231","volume":"25","author":"W Li","year":"2022","unstructured":"Li W, Liu H, Ding R, Liu M, Wang P, Yang W (2022) Exploiting temporal contexts with strided transformer for 3d human pose estimation. IEEE Trans Multimed 25:1282\u20131293","journal-title":"IEEE Trans Multimed"},{"key":"20179_CR16","doi-asserted-by":"crossref","unstructured":"Holmquist K, Wandt B (2023) Diffpose: Multi-hypothesis human pose estimation using diffusion models. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 15977\u201315987","DOI":"10.1109\/ICCV51070.2023.01464"},{"issue":"11","key":"20179_CR17","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2020) Generative adversarial networks. Commun ACM 63(11):139\u2013144","journal-title":"Commun ACM"},{"key":"20179_CR18","doi-asserted-by":"crossref","unstructured":"Ma X, Su J, Wang C, Ci H, Wang Y (2021) Context modeling in 3d human pose estimation: a unified perspective. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6238\u20136247","DOI":"10.1109\/CVPR46437.2021.00617"},{"key":"20179_CR19","doi-asserted-by":"crossref","unstructured":"Fang H-S, Xu Y, Wang W, Liu X, Zhu S-C (2018) Learning pose grammar to encode human body configuration for 3d pose estimation. In: Proceedings of the AAAI conference on artificial intelligence, vol 32","DOI":"10.1609\/aaai.v32i1.12270"},{"issue":"4","key":"20179_CR20","doi-asserted-by":"publisher","first-page":"4122","DOI":"10.1109\/TPAMI.2022.3188716","volume":"45","author":"H Shuai","year":"2023","unstructured":"Shuai H, Wu L, Liu Q (2023) Adaptive multi-view and temporal fusing transformer for 3d human pose estimation. IEEE Trans Pattern Anal Mach Intell 45(4):4122\u20134135. https:\/\/doi.org\/10.1109\/TPAMI.2022.3188716","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"20179_CR21","doi-asserted-by":"publisher","first-page":"1832","DOI":"10.1109\/TMM.2022.3171102","volume":"25","author":"G Hua","year":"2023","unstructured":"Hua G, Liu H, Li W, Zhang Q, Ding R, Xu X (2023) Weakly-supervised 3d human pose estimation with cross-view u-shaped graph convolutional network. IEEE Trans Multimed 25:1832\u20131843. https:\/\/doi.org\/10.1109\/TMM.2022.3171102","journal-title":"IEEE Trans Multimed"},{"issue":"2","key":"20179_CR22","doi-asserted-by":"publisher","first-page":"1781","DOI":"10.1109\/TPAMI.2022.3164344","volume":"45","author":"K Lee","year":"2023","unstructured":"Lee K, Kim W, Lee S (2023) From human pose similarity metric to 3d human pose estimator: Temporal propagating lstm networks. IEEE Trans Pattern Anal Mach Intell 45(2):1781\u20131797. https:\/\/doi.org\/10.1109\/TPAMI.2022.3164344","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"20179_CR23","doi-asserted-by":"crossref","unstructured":"Zou Z, Tang W (2021) Modulated graph convolutional network for 3d human pose estimation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 11477\u201311487","DOI":"10.1109\/ICCV48922.2021.01128"},{"key":"20179_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2023.104841","volume":"140","author":"L Chen","year":"2023","unstructured":"Chen L, Liu Q (2023) Relation-balanced graph convolutional network for 3d human pose estimation. Image Vision Comput 140:104841","journal-title":"Image Vision Comput"},{"key":"20179_CR25","unstructured":"Li W, Liu H, Guo T, Ding R, Tang H (2022) Graphmlp: a graph mlp-like architecture for 3d human pose estimation. arXiv:2206.06420"},{"key":"20179_CR26","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez A.N, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst"},{"key":"20179_CR27","doi-asserted-by":"crossref","unstructured":"Cui Y, Yan L, Cao Z, Liu D (2021) Tf-blender: temporal feature blender for video object detection. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 8138\u20138147","DOI":"10.1109\/ICCV48922.2021.00803"},{"key":"20179_CR28","doi-asserted-by":"crossref","unstructured":"Geng Z, Liang L, Ding T, Zharkov I (2022) Rstt: real-time spatial temporal transformer for space-time video super-resolution. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 17441\u201317451","DOI":"10.1109\/CVPR52688.2022.01692"},{"key":"20179_CR29","doi-asserted-by":"crossref","unstructured":"Lu Y, Wang Q, Ma S, Geng T, Chen YV, Chen H, Liu D (2023) Transflow: transformer as flow learner. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 18063\u201318073","DOI":"10.1109\/CVPR52729.2023.01732"},{"key":"20179_CR30","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S et al (2020) An image is worth 16x16 words: transformers for image recognition at scale. arXiv:2010.11929"},{"key":"20179_CR31","doi-asserted-by":"crossref","unstructured":"Zhao Q, Zheng C, Liu M, Wang P, Chen C (2023) Poseformerv2: exploring frequency domain for efficient and robust 3d human pose estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8877\u20138886","DOI":"10.1109\/CVPR52729.2023.00857"},{"key":"20179_CR32","doi-asserted-by":"crossref","unstructured":"Shan W, Liu Z, Zhang X, Wang S, Ma S, Gao W (2022) P-stmo: pre-trained spatial temporal many-to-one model for 3d human pose estimation. In: European conference on computer vision, Springer, pp 461\u2013478","DOI":"10.1007\/978-3-031-20065-6_27"},{"key":"20179_CR33","doi-asserted-by":"crossref","unstructured":"Choi J, Shim D, Kim HJ (2023) Diffupose: Monocular 3d human pose estimation via denoising diffusion probabilistic model. In: 2023 IEEE\/RSJ international conference on intelligent robots and systems (IROS), IEEE, pp 3773\u20133780","DOI":"10.1109\/IROS55552.2023.10342204"},{"key":"20179_CR34","doi-asserted-by":"crossref","unstructured":"Kang H, Wang Y, Liu M, Wu D, Liu P, Yuan X, Yang W (2024) Diffusion-based pose refinement and multi-hypothesis generation for 3d human pose estimation. In: ICASSP 2024-2024 IEEE international conference on acoustics, speech and signal processing (ICASSP), IEEE, pp 5130\u20135134","DOI":"10.1109\/ICASSP48485.2024.10445850"},{"key":"20179_CR35","unstructured":"Han C, Liang JC, Wang Q, Rabbani M, Dianat S, Rao R, Wu YN, Liu D (2024) Image translation as diffusion visual programmers. arXiv:2401.09742"},{"key":"20179_CR36","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho J, Jain A, Abbeel P (2020) Denoising diffusion probabilistic models. Adv Neural Inf Process Syst 33:6840\u20136851","journal-title":"Adv Neural Inf Process Syst"},{"key":"20179_CR37","doi-asserted-by":"publisher","unstructured":"Choi J, Shim D, Kim HJ (2023) Diffupose: monocular 3d human pose estimation via denoising diffusion probabilistic model. In: 2023 IEEE\/RSJ International conference on intelligent robots and systems (IROS), pp 3773\u20133780. https:\/\/doi.org\/10.1109\/IROS55552.2023.10342204","DOI":"10.1109\/IROS55552.2023.10342204"},{"key":"20179_CR38","unstructured":"Song J, Meng C, Ermon S (2020) Denoising diffusion implicit models. arXiv:2010.02502"},{"key":"20179_CR39","doi-asserted-by":"crossref","unstructured":"Li C, Lee GH (2019) Generating multiple hypotheses for 3d human pose estimation with mixture density network. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9887\u20139895","DOI":"10.1109\/CVPR.2019.01012"},{"key":"20179_CR40","doi-asserted-by":"crossref","unstructured":"Oikarinen T, Hannah D, Kazerounian S (2021) Graphmdn: leveraging graph structure and deep learning to solve inverse problems. In: 2021 International joint conference on neural networks (IJCNN), IEEE, pp 1\u20139","DOI":"10.1109\/IJCNN52387.2021.9534301"},{"key":"20179_CR41","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2022.108871","volume":"250","author":"B Yu","year":"2022","unstructured":"Yu B, Jiao L, Liu X, Li L, Liu F, Yang S, Tang X (2022) Entire deformable convnets for semantic segmentation. Knowl-Based Syst 250:108871","journal-title":"Knowl-Based Syst"},{"key":"20179_CR42","doi-asserted-by":"crossref","unstructured":"Sharma S, Varigonda PT, Bindal P, Sharma A, Jain A (2019) Monocular 3d human pose estimation by generation and ordinal ranking. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 2325\u20132334","DOI":"10.1109\/ICCV.2019.00241"},{"key":"20179_CR43","doi-asserted-by":"crossref","unstructured":"Ionescu C, Papava D, Olaru V, Sminchisescu C (2013) Human3. 6m: large scale datasets and predictive methods for 3d human sensing in natural environments. IEEE Trans Pattern Anal Mach Intell 36(7):1325\u20131339","DOI":"10.1109\/TPAMI.2013.248"},{"key":"20179_CR44","doi-asserted-by":"crossref","unstructured":"Mehta D, Rhodin H, Casas D, Fua P, Sotnychenko O, Xu W, Theobalt C (2017) Monocular 3d human pose estimation in the wild using improved cnn supervision. In: 2017 International conference on 3D vision (3DV), IEEE, pp 506\u2013516","DOI":"10.1109\/3DV.2017.00064"},{"key":"20179_CR45","doi-asserted-by":"crossref","unstructured":"Chen Y, Wang Z, Peng Y, Zhang Z, Yu G, Sun J (2018) Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 7103\u20137112","DOI":"10.1109\/CVPR.2018.00742"},{"key":"20179_CR46","doi-asserted-by":"crossref","unstructured":"Pavllo D, Feichtenhofer C, Grangier D, Auli M (2019) 3d human pose estimation in video with temporal convolutions and semi-supervised training. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7753\u20137762","DOI":"10.1109\/CVPR.2019.00794"},{"key":"20179_CR47","doi-asserted-by":"crossref","unstructured":"Zeng A, Sun X, Huang F, Liu M, Xu Q, Lin S (2020) Srnet: improving generalization in 3d human pose estimation with a split-and-recombine approach. In: Computer vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16, Springer, pp 507\u2013523","DOI":"10.1007\/978-3-030-58568-6_30"},{"key":"20179_CR48","doi-asserted-by":"crossref","unstructured":"Shan W, Lu H, Wang S, Zhang X, Gao W (2021) Improving robustness and accuracy via relative information encoding in 3d human pose estimation. In: Proceedings of the 29th ACM international conference on multimedia, pp 3446\u20133454","DOI":"10.1145\/3474085.3475504"},{"issue":"1","key":"20179_CR49","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1109\/TCSVT.2021.3057267","volume":"32","author":"T Chen","year":"2021","unstructured":"Chen T, Fang C, Shen X, Zhu Y, Chen Z, Luo J (2021) Anatomy-aware 3d human pose estimation with bone-based pose decomposition. IEEE Trans Circuits Syst Video Technol 32(1):198\u2013209","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"20179_CR50","doi-asserted-by":"crossref","unstructured":"Hu W, Zhang C, Zhan F, Zhang L, Wong T-T (2021) Conditional directed graph convolution for 3d human pose estimation. In: Proceedings of the 29th ACM international conference on multimedia, pp 602\u2013611","DOI":"10.1145\/3474085.3475219"},{"key":"20179_CR51","doi-asserted-by":"crossref","unstructured":"Zhan Y, Li F, Weng R, Choi W (2022) Ray3d: ray-based 3d human pose estimation for monocular absolute 3d localization. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 13116\u201313125","DOI":"10.1109\/CVPR52688.2022.01277"},{"key":"20179_CR52","doi-asserted-by":"publisher","first-page":"1282","DOI":"10.1109\/TMM.2022.3141231","volume":"25","author":"W Li","year":"2022","unstructured":"Li W, Liu H, Ding R, Liu M, Wang P, Yang W (2022) Exploiting temporal contexts with strided transformer for 3d human pose estimation. IEEE Trans Multimed 25:1282\u20131293","journal-title":"IEEE Trans Multimed"},{"key":"20179_CR53","doi-asserted-by":"publisher","first-page":"4278","DOI":"10.1109\/TIP.2022.3182269","volume":"31","author":"Y Xue","year":"2022","unstructured":"Xue Y, Chen J, Gu X, Ma H, Ma H (2022) Boosting monocular 3d human pose estimation with part aware attention. IEEE Trans Image Process 31:4278\u20134291","journal-title":"IEEE Trans Image Process"},{"key":"20179_CR54","doi-asserted-by":"publisher","first-page":"8712","DOI":"10.1109\/TMM.2023.3240455","volume":"25","author":"Z Tang","year":"2023","unstructured":"Tang Z, Li J, Hao Y, Hong R (2023) Mlp-jcg: multi-layer perceptron with joint-coordinate gating for efficient 3d human pose estimation. IEEE Trans Multimed 25:8712\u20138724. https:\/\/doi.org\/10.1109\/TMM.2023.3240455","journal-title":"IEEE Trans Multimed"},{"key":"20179_CR55","doi-asserted-by":"crossref","unstructured":"Einfalt M, Ludwig K, Lienhart R (2023) Uplift and upsample: efficient 3d human pose estimation with uplifting transformers. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 2903\u20132913","DOI":"10.1109\/WACV56688.2023.00292"},{"key":"20179_CR56","doi-asserted-by":"publisher","first-page":"104863","DOI":"10.1016\/j.imavis.2023.104863","volume":"140","author":"X Liu","year":"2023","unstructured":"Liu X, Tang H (2023) Strformer: spatial-temporal-retemporal transformer for 3d human pose estimation. Image Vision Comput 140:104863","journal-title":"Image Vision Comput"},{"key":"20179_CR57","doi-asserted-by":"crossref","unstructured":"Chen H, He J-Y, Xiang W, Liu W, Cheng Z-Q, Liu H, Luo B, Geng Y, Xie X (2023) Hdformer: high-order directed transformer for 3d human pose estimation. arXiv:2302.01825","DOI":"10.24963\/ijcai.2023\/65"},{"key":"20179_CR58","unstructured":"Qian X, Tang Y, Zhang N, Han M, Xiao J, Huang M-C, Lin R-S (2023) Hstformer: hierarchical spatial-temporal transformers for 3d human pose estimation. arXiv:2301.07322"},{"key":"20179_CR59","doi-asserted-by":"publisher","first-page":"110116","DOI":"10.1016\/j.patcog.2023.110116","volume":"147","author":"S Du","year":"2024","unstructured":"Du S, Yuan Z, Lai P, Ikenaga T (2024) Joypose: jointly learning evolutionary data augmentation and anatomy-aware global-local representation for 3d human pose estimation. Pattern Recognit 147:110116","journal-title":"Pattern Recognit"},{"key":"20179_CR60","doi-asserted-by":"crossref","unstructured":"Tang Z, Qiu Z, Hao Y, Hong R, Yao T (2023) 3d human pose estimation with spatio-temporal criss-cross attention. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4790\u20134799","DOI":"10.1109\/CVPR52729.2023.00464"},{"key":"20179_CR61","doi-asserted-by":"crossref","unstructured":"Peng Q, Zheng C, Chen C (2024) A dual-augmentor framework for domain generalization in 3d human pose estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2240\u20132249","DOI":"10.1109\/CVPR52733.2024.00218"},{"key":"20179_CR62","doi-asserted-by":"crossref","unstructured":"Yu BX, Zhang Z, Liu Y, Zhong S-h, Liu Y, Chen CW (2023) Gla-gcn: global-local adaptive graph convolutional network for 3d human pose estimation from monocular video. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 8818\u20138829","DOI":"10.1109\/ICCV51070.2023.00810"},{"key":"20179_CR63","doi-asserted-by":"crossref","unstructured":"Li C, Lee GH (2020) Weakly supervised generative network for multiple 3d human pose hypotheses. arXiv:2008.05770","DOI":"10.5244\/C.34.88"},{"key":"20179_CR64","doi-asserted-by":"crossref","unstructured":"Wehrbein T, Rudolph M, Rosenhahn B, Wandt B (2021) Probabilistic monocular 3d human pose estimation with normalizing flows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 11199\u201311208","DOI":"10.1109\/ICCV48922.2021.01101"},{"key":"20179_CR65","doi-asserted-by":"crossref","unstructured":"Li W, Liu H, Tang H, Wang P (2023) Multi-hypothesis representation learning for transformer-based 3d human pose estimation. Pattern Recognit 141:109631","DOI":"10.1016\/j.patcog.2023.109631"},{"key":"20179_CR66","doi-asserted-by":"publisher","first-page":"103890","DOI":"10.1016\/j.jvcir.2023.103890","volume":"95","author":"X Xiang","year":"2023","unstructured":"Xiang X, Zhang K, Qiao Y, El Saddik A (2023) Emhiformer: an enhanced multi-hypothesis interaction transformer for 3d human pose estimation in video. J Visual Commun Image Represent 95:103890","journal-title":"J Visual Commun Image Represent"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-20179-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-024-20179-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-024-20179-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,5]],"date-time":"2025-09-05T22:07:31Z","timestamp":1757110051000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-024-20179-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,10]]},"references-count":66,"journal-issue":{"issue":"23","published-online":{"date-parts":[[2025,7]]}},"alternative-id":["20179"],"URL":"https:\/\/doi.org\/10.1007\/s11042-024-20179-x","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,10]]},"assertion":[{"value":"11 June 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 July 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 August 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 September 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The author declares that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}]}}