{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T04:01:32Z","timestamp":1769054492957,"version":"3.49.0"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2023,9,14]],"date-time":"2023-09-14T00:00:00Z","timestamp":1694649600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,14]],"date-time":"2023-09-14T00:00:00Z","timestamp":1694649600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Strategic Priority Research Program of the Chinese Academy of Sciences","award":["XDC02070600"],"award-info":[{"award-number":["XDC02070600"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-15713-2","type":"journal-article","created":{"date-parts":[[2023,9,14]],"date-time":"2023-09-14T13:50:52Z","timestamp":1694699452000},"page":"28357-28372","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Keypoint-based contextual representations for hand pose estimation"],"prefix":"10.1007","volume":"83","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5202-9911","authenticated-orcid":false,"given":"Weiwei","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rong","family":"Du","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shudong","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,9,14]]},"reference":[{"key":"15713_CR1","doi-asserted-by":"crossref","unstructured":"Athitsos V, Sclaro S (2003) Estimating 3d hand pose from a cluttered image. 2003 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) 2:II\u2013432","DOI":"10.1109\/CVPR.2003.1211500"},{"key":"15713_CR2","first-page":"10835","volume":"2019","author":"A Boukhayma","year":"2019","unstructured":"Boukhayma A, de Bem R, Torr PHS (2019) 3d hand shape and pose from images in the wild. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 2019:10835\u201310844","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"15713_CR3","doi-asserted-by":"crossref","unstructured":"Brahmbhatt S, Tang C, Twigg CD, et al (2020) Contactpose: A dataset of grasps with object contact and hand pose. European Conference on Computer Vision (ECCV)","DOI":"10.1007\/978-3-030-58601-0_22"},{"key":"15713_CR4","doi-asserted-by":"crossref","unstructured":"Cai Y, Ge L, Cai J, et al (2018) Weakly-supervised 3d hand pose estimation from monocular rgb images. European Conference on Computer Vision (ECCV)","DOI":"10.1007\/978-3-030-01231-1_41"},{"key":"15713_CR5","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/TPAMI.2019.2929257","volume":"43","author":"Z Cao","year":"2021","unstructured":"Cao Z, Hidalgo G, Simon T et al (2021) Openpose: Realtime multi-person 2d pose estimation using part affinity fields. IEEE Transactions on Pattern Analysis and Machine Intelligence 43:172\u2013186","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"15713_CR6","doi-asserted-by":"crossref","unstructured":"Chen Y, Ma H, Kong D, et al (2020) Nonparametric structure regularization machine for 2d hand pose estimation. 2020 IEEE Winter Conference on Applications of Computer Vision (WACV) 370\u2013379","DOI":"10.1109\/WACV45572.2020.9093271"},{"key":"15713_CR7","doi-asserted-by":"crossref","unstructured":"Chen Y, Tu Z, Kang D, et al (2021) Model-based 3d hand reconstruction via self-supervised learning. 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 10,446\u201310,455","DOI":"10.1109\/CVPR46437.2021.01031"},{"key":"15713_CR8","doi-asserted-by":"crossref","unstructured":"Ci H, Wang C, Ma X, et al (2019) Optimizing network structure for 3d human pose estimation. 2019 IEEE\/CVF International Conference on Computer Vision (ICCV) 2262\u20132271","DOI":"10.1109\/ICCV.2019.00235"},{"key":"15713_CR9","doi-asserted-by":"crossref","unstructured":"de\u00a0La\u00a0GorceMartin, FleetDavid J, ParagiosNikos (2011) Model-based 3d hand pose estimation from monocular video. IEEE Transactions on Pattern Analysis and Machine Intelligence","DOI":"10.1109\/TPAMI.2011.33"},{"key":"15713_CR10","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, et al (2009) Imagenet: A large-scale hierarchical image database. 2009 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"15713_CR11","doi-asserted-by":"crossref","unstructured":"Fu J, Liu J, Tian H, et al (2019) Dual attention network for scene segmentation. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 3141\u20133149","DOI":"10.1109\/CVPR.2019.00326"},{"key":"15713_CR12","first-page":"10825","volume":"2019","author":"L Ge","year":"2019","unstructured":"Ge L, Ren Z, Li Y et al (2019) 3d hand shape and pose estimation from a single rgb image. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 2019:10825\u201310834","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"15713_CR13","doi-asserted-by":"crossref","unstructured":"Hasson Y, Varol G, Tzionas D, et\u00a0al (2019) Learning joint reconstruction of hands and manipulated objects. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 11,799\u201311,808","DOI":"10.1109\/CVPR.2019.01208"},{"key":"15713_CR14","unstructured":"Ioffe S, Szegedy C (2015) Batch normalization: Accelerating deep network training by reducing internal covariate shift. International conference on machine learning"},{"key":"15713_CR15","doi-asserted-by":"crossref","unstructured":"Iqbal U, Molchanov P, Breuel TM, et\u00a0al (2018) Hand pose estimation via latent 2.5d heatmap regression. Proceedings of the European Conference on Computer Vision (ECCV)","DOI":"10.1007\/978-3-030-01252-6_8"},{"key":"15713_CR16","doi-asserted-by":"crossref","unstructured":"Joo H, Simon T, Sheikh Y (2018) Total capture: A 3d deformation model for tracking faces, hands, and bodies. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 8320\u20138329","DOI":"10.1109\/CVPR.2018.00868"},{"key":"15713_CR17","doi-asserted-by":"crossref","unstructured":"Kulon D, G\u00fcler RA, Kokkinos I, et\u00a0al (2020) Weakly-supervised mesh-convolutional hand reconstruction in the wild. 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 4989\u20134999","DOI":"10.1109\/CVPR42600.2020.00504"},{"key":"15713_CR18","unstructured":"Kulon D, Wang H, G\u00fcler RA, et\u00a0al (2019) Single image 3d hand reconstruction with mesh convolutions. BMVC"},{"key":"15713_CR19","first-page":"5079","volume":"2018","author":"G Moon","year":"2018","unstructured":"Moon G, Chang JY, Lee KM (2018) V2v-posenet: Voxel-to-voxel prediction network for accurate 3d hand and human pose estimation from a single depth map. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 2018:5079\u20135088","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"15713_CR20","doi-asserted-by":"crossref","unstructured":"Moon G, Yu SI, Wen H, et\u00a0al (2020) Interhand2.6m: A dataset and baseline for 3d interacting hand pose estimation from a single rgb image. In: European Conference on Computer Vision (ECCV)","DOI":"10.1007\/978-3-030-58565-5_33"},{"key":"15713_CR21","doi-asserted-by":"crossref","unstructured":"Mueller F, Bernard F, Sotnychenko O, et\u00a0al (2018) Ganerated hands for real-time 3d hand tracking from monocular rgb. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 49\u201359","DOI":"10.1109\/CVPR.2018.00013"},{"key":"15713_CR22","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1016\/j.cviu.2017.10.006","volume":"164","author":"N Neverova","year":"2017","unstructured":"Neverova N, Wolf C, Nebout F et al (2017) Hand pose estimation through semi-supervised and weakly-supervised learning. Comput Vis Image Underst 164:56\u201367","journal-title":"Comput Vis Image Underst"},{"key":"15713_CR23","doi-asserted-by":"crossref","unstructured":"Oikonomidis I, Kyriazis N, Argyros AA (2011) Efficient model-based 3d tracking of hand articulations using kinect. BMVC","DOI":"10.5244\/C.25.101"},{"key":"15713_CR24","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3130800.3130883","volume":"36","author":"J Romero","year":"2017","unstructured":"Romero J, Tzionas D, Black MJ (2017) Embodied hands. ACM Transactions on Graphics (TOG) 36:1\u201317","journal-title":"Embodied hands. ACM Transactions on Graphics (TOG)"},{"key":"15713_CR25","doi-asserted-by":"crossref","unstructured":"Spurr A, Song J, Park S, et\u00a0al (2018) Cross-modal deep variational hand pose estimation. 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 89\u201398","DOI":"10.1109\/CVPR.2018.00017"},{"key":"15713_CR26","doi-asserted-by":"crossref","unstructured":"Taheri O, Ghorbani N, Black MJ, et\u00a0al (2020) Grab: A dataset of whole-body human grasping of objects. European Conference on Computer Vision (ECCV)","DOI":"10.1007\/978-3-030-58548-8_34"},{"key":"15713_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2980179.2980226","volume":"35","author":"A Tkach","year":"2016","unstructured":"Tkach A, Pauly M, Tagliasacchi A (2016) Sphere-meshes for real-time hand modeling and tracking. ACM Transactions on Graphics (TOG) 35:1\u201311","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"15713_CR28","doi-asserted-by":"crossref","unstructured":"Toshev A, Szegedy C (2014) Deeppose: Human pose estimation via deep neural networks. 2014 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) 1653\u20131660","DOI":"10.1109\/CVPR.2014.214"},{"key":"15713_CR29","unstructured":"Vaswani A, Shazeer NM, Parmar N, et\u00a0al (2017) Attention is all you need. Advances in neural information processing systems 5998\u20136008"},{"key":"15713_CR30","first-page":"7794","volume":"2018","author":"X Wang","year":"2018","unstructured":"Wang X, Girshick RB, Gupta AK et al (2018) Non-local neural networks. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 2018:7794\u20137803","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"15713_CR31","doi-asserted-by":"publisher","first-page":"3258","DOI":"10.1109\/TCSVT.2018.2879980","volume":"29","author":"Y Wang","year":"2019","unstructured":"Wang Y, Peng C, Liu Y (2019) Mask-pose cascaded cnn for 2d hand pose estimation from single color image. IEEE Transactions on Circuits and Systems for Video Technology 29:3258\u20133268","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"15713_CR32","doi-asserted-by":"publisher","first-page":"2977","DOI":"10.1109\/TIP.2019.2955280","volume":"29","author":"Y Wang","year":"2020","unstructured":"Wang Y, Zhang B, Peng C (2020) Srhandnet: Real-time 2d hand pose estimation with simultaneous region localization. IEEE Transactions on Image Processing 29:2977\u20132986","journal-title":"IEEE Transactions on Image Processing"},{"key":"15713_CR33","first-page":"4724","volume":"2016","author":"SE Wei","year":"2016","unstructured":"Wei SE, Ramakrishna V, Kanade T et al (2016) Convolutional pose machines. IEEE Conference on Computer Vision and Pattern Recognition (CVPR) 2016:4724\u20134732","journal-title":"IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"15713_CR34","doi-asserted-by":"crossref","unstructured":"Xiang D, Joo H, Sheikh Y (2019) Monocular total capture: Posing face, body, and hands in the wild. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 10,957\u201310,966","DOI":"10.1109\/CVPR.2019.01122"},{"key":"15713_CR35","doi-asserted-by":"publisher","first-page":"2335","DOI":"10.1109\/ICCV.2019.00242","volume":"2019","author":"L Yang","year":"2019","unstructured":"Yang L, Li S, Lee D et al (2019) Aligning latent spaces for 3d hand pose estimation. IEEE\/CVF International Conference on Computer Vision (ICCV) 2019:2335\u20132343","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"key":"15713_CR36","doi-asserted-by":"crossref","unstructured":"Yan S, Xiong Y, Lin D (2018) Spatial temporal graph convolutional networks for skeleton-based action recognition. Thirty-second AAAI conference on artificial intelligence","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"15713_CR37","unstructured":"Yuan Y, Wang J (2018) Ocnet: Object context network for scene parsing. arXiv:1809.00916"},{"key":"15713_CR38","doi-asserted-by":"publisher","first-page":"2354","DOI":"10.1109\/ICCV.2019.00244","volume":"2019","author":"X Zhang","year":"2019","unstructured":"Zhang X, Li Q, Zhang W et al (2019) End-to-end hand mesh recovery from a monocular rgb image. IEEE\/CVF International Conference on Computer Vision (ICCV) 2019:2354\u20132364","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"key":"15713_CR39","unstructured":"Zhang J, Jiao J, Chen M, et\u00a0al (2016) 3d hand pose tracking and estimation using stereo matching. arXiv:1610.07214"},{"key":"15713_CR40","doi-asserted-by":"crossref","unstructured":"Zhang H, Zhang H, Wang C, et\u00a0al (2019) Co-occurrent features in semantic segmentation. 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 548\u2013557","DOI":"10.1109\/CVPR.2019.00064"},{"key":"15713_CR41","first-page":"3420","volume":"2019","author":"L Zhao","year":"2019","unstructured":"Zhao L, Peng X, Tian Y et al (2019) Semantic graph convolutional networks for 3d human pose regression. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 2019:3420\u20133430","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"15713_CR42","first-page":"5345","volume":"2020","author":"Y Zhou","year":"2020","unstructured":"Zhou Y, Habermann M, Xu W et al (2020) Monocular real-time hand shape and motion capture using multi-modal data. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) 2020:5345\u20135354","journal-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"15713_CR43","doi-asserted-by":"crossref","unstructured":"Zhu JY, Park T, Isola P, et\u00a0al (2017) Unpaired image-to-image translation using cycle-consistent adversarial networks. 2017 IEEE International Conference on Computer Vision (ICCV) 2242\u20132251","DOI":"10.1109\/ICCV.2017.244"},{"key":"15713_CR44","doi-asserted-by":"publisher","first-page":"4913","DOI":"10.1109\/ICCV.2017.525","volume":"2017","author":"C Zimmermann","year":"2017","unstructured":"Zimmermann C, Brox T (2017) Learning to estimate 3d hand pose from single rgb images. IEEE International Conference on Computer Vision (ICCV) 2017:4913\u20134921","journal-title":"IEEE International Conference on Computer Vision (ICCV)"},{"key":"15713_CR45","doi-asserted-by":"publisher","first-page":"813","DOI":"10.1109\/ICCV.2019.00090","volume":"2019","author":"C Zimmermann","year":"2019","unstructured":"Zimmermann C, Ceylan D, Yang J et al (2019) Freihand: A dataset for markerless capture of hand pose and shape from single rgb images. IEEE\/CVF International Conference on Computer Vision (ICCV) 2019:813\u2013822","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15713-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-15713-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-15713-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,10]],"date-time":"2024-03-10T10:08:05Z","timestamp":1710065285000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-15713-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,14]]},"references-count":45,"journal-issue":{"issue":"10","published-online":{"date-parts":[[2024,3]]}},"alternative-id":["15713"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-15713-2","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9,14]]},"assertion":[{"value":"17 August 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 January 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 April 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 September 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}