{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T13:03:32Z","timestamp":1784552612466,"version":"3.55.0"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1007\/s00371-025-04330-9","type":"journal-article","created":{"date-parts":[[2026,1,4]],"date-time":"2026-01-04T14:31:11Z","timestamp":1767537071000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["ER: Extract-regress network for precise 3D reconstruction of interacting hands from monocular images"],"prefix":"10.1007","volume":"42","author":[{"given":"Rui","family":"Feng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fangni","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongsheng","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linzhan","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,1,4]]},"reference":[{"issue":"1","key":"4330_CR1","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1007\/s00371-023-02777-2","volume":"40","author":"L Baker","year":"2024","unstructured":"Baker, L., Ventura, J., Langlotz, T., Gul, S., Mills, S., Zollmann, S.: Localization and tracking of stationary users for augmented reality. Vis. Comput. 40(1), 227\u2013244 (2024)","journal-title":"Vis. Comput."},{"key":"4330_CR2","doi-asserted-by":"crossref","unstructured":"Hampali, S., Rad, M., Oberweger, M., Lepetit, V.: Honnotate,: A method for 3d annotation of hand and object poses. Presented at the (2020)","DOI":"10.1109\/CVPR42600.2020.00326"},{"key":"4330_CR3","doi-asserted-by":"crossref","unstructured":"Moon, G., Yu, S.-I., Wen, H., Shiratori, T., Lee, K.M.: Interhand2. 6m: A dataset and baseline for 3d interacting hand pose estimation from a single rgb image. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XX 16, pp. 548\u2013564 (2020). Springer","DOI":"10.1007\/978-3-030-58565-5_33"},{"key":"4330_CR4","doi-asserted-by":"crossref","unstructured":"Zimmermann, C., Ceylan, D., Yang, J., Russell, B., Argus, M., Brox, T.: Freihand: A dataset for markerless capture of hand pose and shape from single rgb images. Presented at the (2019)","DOI":"10.1109\/ICCV.2019.00090"},{"key":"4330_CR5","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Habermann, M., Xu, W., Habibie, I., Theobalt, C., Xu, F.: Monocular real-time hand shape and motion capture using multi-modal data. Presented at the (2020)","DOI":"10.1109\/CVPR42600.2020.00539"},{"key":"4330_CR6","doi-asserted-by":"crossref","unstructured":"Chen, X., Liu, Y., Dong, Y., Zhang, X., Ma, C., Xiong, Y., Zhang, Y., Guo,: X.: Mobrecon: Mobile-friendly hand mesh reconstruction from monocular image. Presented at the (2022)","DOI":"10.1109\/CVPR52688.2022.01989"},{"issue":"1","key":"4330_CR7","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1007\/s00371-022-02762-1","volume":"40","author":"H Mahmud","year":"2024","unstructured":"Mahmud, H., Morshed, M.M., Hasan, M.K.: Quantized depth image and skeleton-based multimodal dynamic hand gesture recognition. Vis. Comput. 40(1), 11\u201325 (2024)","journal-title":"Vis. Comput."},{"key":"4330_CR8","doi-asserted-by":"crossref","unstructured":"Zuo, B., Zhao, Z., Sun, W., Xie, W., Xue, Z., Wang, Y.: Reconstructing interacting hands with interaction prior from monocular images. Presented at the (2023)","DOI":"10.1109\/ICCV51070.2023.00831"},{"key":"4330_CR9","doi-asserted-by":"crossref","unstructured":"Li, M., An, L., Zhang, H., Wu, L., Chen, F., Yu, T., Liu, Y.: Interacting attention graph for single image two-hand reconstruction. Presented at the (2022)","DOI":"10.1109\/CVPR52688.2022.00278"},{"key":"4330_CR10","doi-asserted-by":"crossref","unstructured":"Yu, Z., Huang, S., Fang, C., Breckon, T.P., Wang, J.: Acr: Attention collaboration-based regressor for arbitrary two-hand reconstruction. Presented at the (2023)","DOI":"10.1109\/CVPR52729.2023.01245"},{"key":"4330_CR11","doi-asserted-by":"crossref","unstructured":"Kulon, D., Guler, R.A., Kokkinos, I., Bronstein, M.M., Zafeiriou, S.: Weakly-supervised mesh-convolutional hand reconstruction in the wild. Presented at the (2020)","DOI":"10.1109\/CVPR42600.2020.00504"},{"key":"4330_CR12","doi-asserted-by":"crossref","unstructured":"Ren, P., Wen, C., Zheng, X., Xue, Z., Sun, H., Qi, Q., Wang, J., Liao, J.: Decoupled iterative refinement framework for interacting hands reconstruction from a single rgb image. Presented at the (2023)","DOI":"10.1109\/ICCV51070.2023.00736"},{"key":"4330_CR13","doi-asserted-by":"crossref","unstructured":"Hampali, S., Sarkar, S.D., Rad, M., Lepetit, V.: Keypoint transformer: Solving joint identification in challenging hands and object interactions for accurate 3d pose estimation. Presented at the (2022)","DOI":"10.1109\/CVPR52688.2022.01081"},{"key":"4330_CR14","doi-asserted-by":"crossref","unstructured":"Zimmermann, C., Brox, T.: Learning to estimate 3d hand pose from single rgb images. Presented at the (2017)","DOI":"10.1109\/ICCV.2017.525"},{"key":"4330_CR15","doi-asserted-by":"crossref","unstructured":"Huang, W., Ren, P., Wang, J., Qi, Q., Sun, H.: Awr: Adaptive weighting regression for 3d hand pose estimation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11061\u201311068 (2020)","DOI":"10.1609\/aaai.v34i07.6761"},{"key":"4330_CR16","doi-asserted-by":"crossref","unstructured":"Ren, P., Sun, H., Hao, J., Wang, J., Qi, Q., Liao, J.: Mining multi-view information: a strong self-supervised framework for depth-based 3d hand pose and mesh estimation. Presented at the (2022)","DOI":"10.1109\/CVPR52688.2022.01990"},{"key":"4330_CR17","doi-asserted-by":"crossref","unstructured":"Cai, Y., Ge, L., Cai, J., Yuan, J.: Weakly-supervised 3d hand pose estimation from monocular rgb images. Presented at the (2018)","DOI":"10.1007\/978-3-030-01231-1_41"},{"key":"4330_CR18","doi-asserted-by":"crossref","unstructured":"Iqbal, U., Molchanov, P., Gall, T.B.J., Kautz, J.: Hand pose estimation via latent 2.5 d heatmap regression. Presented at the (2018)","DOI":"10.1007\/978-3-030-01252-6_8"},{"key":"4330_CR19","doi-asserted-by":"crossref","unstructured":"Tang, X., Wang, T., Fu, C.-W.: Towards accurate alignment in real-time 3d hand-mesh reconstruction. Presented at the (2021)","DOI":"10.1109\/ICCV48922.2021.01149"},{"key":"4330_CR20","doi-asserted-by":"crossref","unstructured":"Wan, C., Probst, T., Van\u00a0Gool, L., Yao, A.: Dual grid net: Hand mesh vertex regression from single depth maps. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXX 16, pp. 442\u2013459 (2020). Springer","DOI":"10.1007\/978-3-030-58577-8_27"},{"key":"4330_CR21","doi-asserted-by":"crossref","unstructured":"Moon, G., Choi, H., Lee, K.M.: Accurate 3d hand pose estimation for whole-body 3d human mesh estimation. Presented at the (2022)","DOI":"10.1109\/CVPRW56347.2022.00257"},{"key":"4330_CR22","doi-asserted-by":"crossref","unstructured":"Kong, D., Zhang, L., Chen, L., Ma, H., Yan, X., Sun, S., Liu, X., Han, K., Xie,: In: X.: Identity-aware hand mesh estimation and personalization from rgb images, pp. 536\u2013553. Springer (2022)","DOI":"10.1007\/978-3-031-20065-6_31"},{"key":"4330_CR23","doi-asserted-by":"crossref","unstructured":"Boukhayma, A., Bem, R.d., Torr, P.H.: 3d hand shape and pose from images in the wild. Presented at the (2019)","DOI":"10.1109\/CVPR.2019.01110"},{"key":"4330_CR24","doi-asserted-by":"crossref","unstructured":"Lin, K., Wang, L., Liu, Z.: End-to-end human pose and mesh reconstruction with transformers. Presented at the (2021)","DOI":"10.1109\/CVPR46437.2021.00199"},{"key":"4330_CR25","doi-asserted-by":"crossref","unstructured":"Chen, X., Liu, Y., Ma, C., Chang, J., Wang, H., Chen, T., Guo, X., Wan, P., Zheng, W.: Camera-space hand mesh recovery via semantic aggregation and adaptive 2d\u20131d registration. Presented at the (2021)","DOI":"10.1109\/CVPR46437.2021.01307"},{"key":"4330_CR26","doi-asserted-by":"crossref","unstructured":"Chen, Y., Tu, Z., Kang, D., Bao, L., Zhang, Y., Zhe, X., Chen, R., Yuan, J.: Model-based 3d hand reconstruction via self-supervised learning. Presented at the (2021)","DOI":"10.1109\/CVPR46437.2021.01031"},{"key":"4330_CR27","unstructured":"Romero, J., Tzionas, D., Black, M.J.: Embodied hands: Modeling and capturing hands and bodies together. arXiv preprint arXiv:2201.02610 (2022)"},{"key":"4330_CR28","doi-asserted-by":"crossref","unstructured":"Zhao, L., Peng, X., Tian, Y., Kapadia, M., Metaxas, D.N.: Semantic graph convolutional networks for 3d human pose regression. Presented at the (2019)","DOI":"10.1109\/CVPR.2019.00354"},{"key":"4330_CR29","doi-asserted-by":"crossref","unstructured":"Zheng, C., Zhu, S., Mendieta, M., Yang, T., Chen, C., Ding, Z.: 3d human pose estimation with spatial and temporal transformers. Presented at the (2021)","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"4330_CR30","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: Action-conditioned 3d human motion synthesis with transformer vae. Presented at the (2021)","DOI":"10.1109\/ICCV48922.2021.01080"},{"issue":"1","key":"4330_CR31","doi-asserted-by":"publisher","first-page":"2201","DOI":"10.1002\/cav.2201","volume":"35","author":"X Zhu","year":"2024","unstructured":"Zhu, X., Yao, X., Zhang, J., Zhu, M., You, L., Yang, X., Zhang, J., Zhao, H., Zeng, D.: Tmsdnet: Transformer with multi-scale dense network for single and multi-view 3d reconstruction. Comput. Anim. Virtual Worlds 35(1), 2201 (2024)","journal-title":"Comput. Anim. Virtual Worlds"},{"key":"4330_CR32","doi-asserted-by":"crossref","unstructured":"Wen, Y., Luo, B., Shi, W., Ji, J., Cao, W., Yang, X., Sheng, B.: Sat-net: structure-aware transformer-based attention fusion network for low-quality retinal fundus images enhancement. IEEE Transactions on Multimedia (2025)","DOI":"10.1109\/TMM.2025.3565935"},{"key":"4330_CR33","doi-asserted-by":"crossref","unstructured":"Wang, Y., Shao, M., Wang, C., Xu, K., Lu, X.: Interacthand: Robust 3d hand mesh reconstruction via interaction-aware segmentation and refinement. The Visual Computer, 1\u201313 (2025)","DOI":"10.1007\/s00371-025-04059-5"},{"issue":"4","key":"4330_CR34","first-page":"366","volume":"5","author":"X Hu","year":"2023","unstructured":"Hu, X., Bao, X., Wei, G., Li, Z.: Human-pose estimation based on weak supervision. Virt. Real. Intel. Hardw. 5(4), 366\u2013377 (2023)","journal-title":"Virt. Real. Intel. Hardw."},{"issue":"1","key":"4330_CR35","doi-asserted-by":"publisher","first-page":"2207","DOI":"10.1002\/cav.2207","volume":"35","author":"U Aiman","year":"2024","unstructured":"Aiman, U., Ahmad, T.: Angle based hand gesture recognition using graph convolutional network. Comput. Anim. Virt. Worlds 35(1), 2207 (2024)","journal-title":"Comput. Anim. Virt. Worlds"},{"key":"4330_CR36","doi-asserted-by":"crossref","unstructured":"Yang, W., Miao, X.: Enhancing hand-object interaction pose reconstruction through semantic-enhanced and reconstruction modules. The Visual Computer, 1\u201318 (2025)","DOI":"10.1007\/s00371-025-04020-6"},{"issue":"6","key":"4330_CR37","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3414685.3417768","volume":"39","author":"B Smith","year":"2020","unstructured":"Smith, B., Wu, C., Wen, H., Peluse, P., Sheikh, Y., Hodgins, J.K., Shiratori, T.: Constraining dense hand surface tracking with elasticity. ACM Trans. Grap. (ToG) 39(6), 1\u201314 (2020)","journal-title":"ACM Trans. Grap. (ToG)"},{"issue":"6","key":"4330_CR38","first-page":"1","volume":"39","author":"J Wang","year":"2020","unstructured":"Wang, J., Mueller, F., Bernard, F., Sorli, S., Sotnychenko, O., Qian, N., Otaduy, M.A., Casas, D., Theobalt, C.: Rgb2hands: real-time tracking of 3d hand interactions from monocular rgb video. ACM Trans. Grap. (ToG) 39(6), 1\u201316 (2020)","journal-title":"ACM Trans. Grap. (ToG)"},{"issue":"4","key":"4330_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3306346.3322958","volume":"38","author":"F Mueller","year":"2019","unstructured":"Mueller, F., Davis, M., Bernard, F., Sotnychenko, O., Verschoor, M., Otaduy, M.A., Casas, D., Theobalt, C.: Real-time pose and shape reconstruction of two interacting hands with a single depth camera. ACM Trans. Grap. (ToG) 38(4), 1\u201313 (2019)","journal-title":"ACM Trans. Grap. (ToG)"},{"issue":"6","key":"4330_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3130800.3130853","volume":"36","author":"J Taylor","year":"2017","unstructured":"Taylor, J., Tankovich, V., Tang, D., Keskin, C., Kim, D., Davidson, P., Kowdle, A., Izadi, S.: Articulated distance fields for ultra-fast tracking of hands interacting. ACM Trans. Grap. (TOG) 36(6), 1\u201312 (2017)","journal-title":"ACM Trans. Grap. (TOG)"},{"key":"4330_CR41","doi-asserted-by":"crossref","unstructured":"Choutas, V., Pavlakos, G., Bolkart, T., Tzionas, D., Black, M.J.: Monocular expressive body regression through body-driven attention. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part X 16, pp. 20\u201340 (2020). Springer","DOI":"10.1007\/978-3-030-58607-2_2"},{"key":"4330_CR42","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Li, Z., An, L., Li, M., Yu, T., Liu, Y.: Lightweight multi-person total motion capture using sparse multi-view cameras. Presented at the (2021)","DOI":"10.1109\/ICCV48922.2021.00551"},{"key":"4330_CR43","doi-asserted-by":"crossref","unstructured":"Kim, D.U., Kim, K.I., Baek, S.: End-to-end detection and pose estimation of two interacting hands. Presented at the (2021)","DOI":"10.1109\/ICCV48922.2021.01100"},{"key":"4330_CR44","doi-asserted-by":"crossref","unstructured":"Fan, Z., Spurr, A., Kocabas, M., Tang, S., Black, M.J., Hilliges, O.: In: Learning to disambiguate strongly interacting hands via probabilistic per-pixel part segmentation, pp. 1\u201310. IEEE (2021)","DOI":"10.1109\/3DV53792.2021.00011"},{"key":"4330_CR45","doi-asserted-by":"crossref","unstructured":"Zhang, B., Wang, Y., Deng, X., Zhang, Y., Tan, P., Ma, C., Wang, H.: Interacting two-hand 3d pose and shape reconstruction from single color image. Presented at the (2021)","DOI":"10.1109\/ICCV48922.2021.01116"},{"key":"4330_CR46","doi-asserted-by":"crossref","unstructured":"Yan, H., Chen, J., Zhang, X., Zhang, S., Jiao, N., Liang, X., Zheng, T.: Ultrapose: Synthesizing dense pose with 1 billion points by human-body decoupling 3d model. Presented at the (2021)","DOI":"10.1109\/ICCV48922.2021.01071"},{"issue":"4","key":"4330_CR47","first-page":"1","volume":"41","author":"Y Li","year":"2022","unstructured":"Li, Y., Zhang, L., Qiu, Z., Jiang, Y., Li, N., Ma, Y., Zhang, Y., Xu, L., Yu, J.: Nimble: a non-rigid hand model with bones and muscles. ACM Trans. Grap. (TOG) 41(4), 1\u201316 (2022)","journal-title":"ACM Trans. Grap. (TOG)"},{"key":"4330_CR48","doi-asserted-by":"publisher","first-page":"37055","DOI":"10.52202\/068431-2685","volume":"35","author":"D Gao","year":"2022","unstructured":"Gao, D., Xiu, Y., Li, K., Yang, L., Wang, F., Zhang, P., Zhang, B., Lu, C., Tan, P.: Dart: Articulated hand model with diverse accessories and rich textures. Adv. Neural. Inf. Process. Syst. 35, 37055\u201337067 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4330_CR49","doi-asserted-by":"crossref","unstructured":"Jiang, C., Xiao, Y., Wu, C., Zhang, M., Zheng, J., Cao, Z., Zhou, J.T.: A2j-transformer: Anchor-to-joint transformer network for 3d interacting hand pose estimation from a single rgb image. Presented at the (2023)","DOI":"10.1109\/CVPR52729.2023.00854"},{"key":"4330_CR50","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. Presented at the (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"4330_CR51","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., Fei-Fei, L.: In: Imagenet: A large-scale hierarchical image database, pp. 248\u2013255. Ieee (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"4330_CR52","doi-asserted-by":"crossref","unstructured":"Moon, G.: Bringing inputs to shared domains for 3d interacting hands recovery in the wild. Presented at the (2023)","DOI":"10.1109\/CVPR52729.2023.01633"},{"key":"4330_CR53","doi-asserted-by":"crossref","unstructured":"Meng, H., Jin, S., Liu, W., Qian, C., Lin, M., Ouyang, W., Luo, P.: In: 3d interacting hand pose estimation by hand de-occlusion and removal, pp. 380\u2013397. Springer (2022)","DOI":"10.1007\/978-3-031-20068-7_22"},{"key":"4330_CR54","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Shan, D., Radosavovic, I., Kanazawa, A., Fouhey, D., Malik, J.: Reconstructing hands in 3d with transformers. Presented at the (2024)","DOI":"10.1109\/CVPR52733.2024.00938"},{"key":"4330_CR55","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., Antiga, L., et\u00a0al.: Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems 32 (2019)"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04330-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-04330-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04330-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T13:05:20Z","timestamp":1772629520000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-04330-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1]]},"references-count":55,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["4330"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-04330-9","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1]]},"assertion":[{"value":"1 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"107"}}