{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:06:00Z","timestamp":1765343160990,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":65,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62332019, 62406039"],"award-info":[{"award-number":["62332019, 62406039"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Program of China","award":["2023YFF-1203900, 2023YFF1203903"],"award-info":[{"award-number":["2023YFF-1203900, 2023YFF1203903"]}]},{"name":"China Postdoctoral Science Foundatiion","award":["2023TQ0039, 2024M750257, GZC20230320"],"award-info":[{"award-number":["2023TQ0039, 2024M750257, GZC20230320"]}]},{"name":"Beijing Nova Program","award":["20240484513"],"award-info":[{"award-number":["20240484513"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758215","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:37:21Z","timestamp":1761377841000},"page":"12753-12760","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["VIHand: Enhancing 3D Hand Pose Estimation with Visual-Inertial Benchmark"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-3010-1032","authenticated-orcid":false,"given":"Xinyi","family":"Wang","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1691-6457","authenticated-orcid":false,"given":"Pengfei","family":"Ren","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9253-5824","authenticated-orcid":false,"given":"Haoyang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Defense Innovation Institute, Academy of Military Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4799-4112","authenticated-orcid":false,"given":"Xin","family":"Sheng","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3320-8830","authenticated-orcid":false,"given":"Da","family":"Li","sequence":"additional","affiliation":[{"name":"Nankai University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8286-6785","authenticated-orcid":false,"given":"Liang","family":"Xie","sequence":"additional","affiliation":[{"name":"Defense Innovation Institute, Academy of Military Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8901-9335","authenticated-orcid":false,"given":"Yue","family":"Gao","sequence":"additional","affiliation":[{"name":"MoE Key Lab of Artificial Intelligence, AI Institute, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2147-9888","authenticated-orcid":false,"given":"Erwei","family":"Yin","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China and Defense Innovation Institute, Academy of Military Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/THMS.2017.2720667"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01110"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings, Part XIII 16","author":"Brahmbhatt Samarth","year":"2020","unstructured":"Samarth Brahmbhatt, Chengcheng Tang, Christopher D Twigg, Charles C Kemp, and James Hays. 2020. ContactPose: A dataset of grasps with object contact and hand pose. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XIII 16. Springer, 361--378."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIE.2019.2912765"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00893"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01989"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01307"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings, Part VII 16","author":"Choi Hongsuk","year":"2020","unstructured":"Hongsuk Choi, Gyeongsik Moon, and Kyoung Mu Lee. 2020. Pose2mesh: Graph convolutional network for 3d human pose and mesh recovery from a 2d human pose. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part VII 16. Springer, 769--787."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSENS.2024.3501586"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1155\/2017\/7594763"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.112532"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Kun Gao Haoyang Zhang Xiaolong Liu Xinyi Wang Liang Xie Bowen Ji Ye Yan and Erwei Yin. 2024. Challenges and solutions for vision-based hand gesture interpretation: A review. Computer vision and image understanding Nov. (2024) 248.","DOI":"10.1016\/j.cviu.2024.104095"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00050"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.391"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.602"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01109"},{"volume-title":"So Kweon, and Yaser Sheikh. 2017. Deltille Grids for Geometric Camera Calibration. In 2017 IEEE International Conference on Computer Vision (ICCV).","author":"Ha Hyowon","key":"e_1_3_2_1_17_1","unstructured":"Hyowon Ha, Michal Perdoch, Hatem Alismail, In So Kweon, and Yaser Sheikh. 2017. Deltille Grids for Geometric Camera Calibration. In 2017 IEEE International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00326"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1080\/10739149.2020.1789657"},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the European conference on computer vision (ECCV). 118--134","author":"Iqbal Umar","year":"2018","unstructured":"Umar Iqbal, Pavlo Molchanov, Thomas Breuel Juergen Gall, and Jan Kautz. 2018. Hand pose estimation via latent 2.5 d heatmap regression. In Proceedings of the European conference on computer vision (ECCV). 118--134."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TII.2017.2779814"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNSRE.2014.2357579"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00998"},{"key":"e_1_3_2_1_24_1","volume-title":"Visual-inertial hand motion tracking with robustness against occlusion, interference, and contact. Science Robotics 6, 58","author":"Lee Yongseok","year":"2021","unstructured":"Yongseok Lee, Wonkyung Do, Hanbyeol Yoon, Jinuk Heo, WonHa Lee, and Dongjun Lee. 2021. Visual-inertial hand motion tracking with robustness against occlusion, interference, and contact. Science Robotics 6, 58 (2021), eabe1315."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMECH.2018.2872570"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i3.16287"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.3390\/s18051545"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00199"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01270"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","unstructured":"Francesca Mongelli and Matteo Menolotto. 2023. HANDMI4 (HAND Motion capture data for Industry 4.0). doi:10.21227\/c6t8-ge47","DOI":"10.21227\/c6t8-ge47"},{"key":"e_1_3_2_1_31_1","first-page":"17689","article-title":"A dataset of relighted 3D interacting hands","volume":"36","author":"Moon Gyeongsik","year":"2023","unstructured":"Gyeongsik Moon, Shunsuke Saito, Weipeng Xu, Rohan Joshi, Julia Buffalini, Harley Bellan, Nicholas Rosen, Jesse Richardson, Mallorie Mize, Philippe De Bree, et al. 2023. A dataset of relighted 3D interacting hands. Advances in Neural Information Processing Systems 36 (2023), 17689--17701.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings, Part XX 16","author":"Moon Gyeongsik","year":"2020","unstructured":"Gyeongsik Moon, Shoou-I Yu, He Wen, Takaaki Shiratori, and Kyoung Mu Lee. 2020. Interhand2. 6m: A dataset and baseline for 3d interacting hand pose estimation from a single rgb image. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XX 16. Springer, 548--564."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00013"},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the IEEE international conference on computer vision. 1154--1163","author":"Mueller Franziska","year":"2017","unstructured":"Franziska Mueller, Dushyant Mehta, Oleksandr Sotnychenko, Srinath Sridhar, Dan Casas, and Christian Theobalt. 2017. Real-time hand tracking under occlusion from an egocentric rgb-d sensor. In Proceedings of the IEEE international conference on computer vision. 1154--1163."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618145"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00938"},{"key":"e_1_3_2_1_37_1","volume-title":"Wilor: End-to-end 3d hand localization and reconstruction in-the-wild. arXiv preprint arXiv:2409.12259","author":"Potamias Rolandos Alexandros","year":"2024","unstructured":"Rolandos Alexandros Potamias, Jinglei Zhang, Jiankang Deng, and Stefanos Zafeiriou. 2024. Wilor: End-to-end 3d hand localization and reconstruction in-the-wild. arXiv preprint arXiv:2409.12259 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.mejo.2018.01.014"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3192708"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01990"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICHR.2009.5379596"},{"key":"e_1_3_2_1_42_1","first-page":"6","article-title":"Embodied Hands: Modeling and Capturing Hands and Bodies Together","volume":"36","author":"Romero Javier","year":"2017","unstructured":"Javier Romero, Dimitrios Tzionas, and Michael J. Black. 2017. Embodied Hands: Modeling and Capturing Hands and Bodies Together. ACM Transactions on Graphics (Proc. SIGGRAPH Asia) 36, 6 (Nov. 2017).","journal-title":"ACM Transactions on Graphics (Proc. SIGGRAPH Asia)"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/2906388.2906407"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00017"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.305"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581468"},{"key":"e_1_3_2_1_47_1","volume-title":"EPMF: Efficient perception-aware multi-sensor fusion for 3D semantic segmentation","author":"Tan Mingkui","year":"2024","unstructured":"Mingkui Tan, Zhuangwei Zhuang, Sitao Chen, Rong Li, Kui Jia, Qicheng Wang, and Yuanqing Li. 2024. EPMF: Efficient perception-aware multi-sensor fusion for 3D semantic segmentation. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.490"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/2629500"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3414685.3417852","article-title":"Rgb2hands: real-time tracking of 3d hand interactions from monocular rgb video","volume":"39","author":"Wang Jiayi","year":"2020","unstructured":"Jiayi Wang, Franziska Mueller, Florian Bernard, Suzanne Sorli, Oleksandr Sotnychenko, Neng Qian, Miguel A Otaduy, Dan Casas, and Christian Theobalt. 2020. Rgb2hands: real-time tracking of 3d hand interactions from monocular rgb video. ACM Transactions on Graphics (ToG) 39, 6 (2020), 1--16.","journal-title":"ACM Transactions on Graphics (ToG)"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/BSN51625.2021.9507012"},{"key":"e_1_3_2_1_52_1","volume-title":"andWei Fan","author":"Wu Zhenyu","year":"2020","unstructured":"Zhenyu Wu, Duc Hoang, Shihyao Lin, Yusheng Xie, Liangjian Chen, Yenyu Lin, ZhangyangWang, andWei Fan. 2020. MM-Hand: 3D-Aware Multi-Modal Guided Hand Generation for 3D Hand Pose Synthesis. (2020)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01117"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02028"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00242"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01011"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.3390\/s20144008"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2012.48"},{"key":"e_1_3_2_1_59_1","first-page":"1","article-title":"Transpose: Real-time 3d human translation and pose estimation with six inertial sensors","volume":"40","author":"Yi Xinyu","year":"2021","unstructured":"Xinyu Yi, Yuxiao Zhou, and Feng Xu. 2021. Transpose: Real-time 3d human translation and pose estimation with six inertial sensors. ACM Transactions On Graphics (TOG) 40, 4 (2021), 1--13.","journal-title":"ACM Transactions On Graphics (TOG)"},{"key":"e_1_3_2_1_60_1","volume-title":"Fine-grained and real-time gesture recognition by using IMU sensors","author":"Zhang Dian","year":"2021","unstructured":"Dian Zhang, Zexiong Liao, Wen Xie, Xiaofeng Wu, Haoran Xie, Jiang Xiao, and Landu Jiang. 2021. Fine-grained and real-time gesture recognition by using IMU sensors. IEEE Transactions on Mobile Computing (2021)."},{"key":"e_1_3_2_1_61_1","volume-title":"3d hand pose tracking and estimation using stereo matching. arXiv preprint arXiv:1610.07214","author":"Zhang Jiawei","year":"2016","unstructured":"Jiawei Zhang, Jianbo Jiao, Mingliang Chen, Liangqiong Qu, Xiaobin Xu, and Qingxiong Yang. 2016. 3d hand pose tracking and estimation using stereo matching. arXiv preprint arXiv:1610.07214 (2016)."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00185"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3369836"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.525"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00090"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758215","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:02:47Z","timestamp":1765342967000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758215"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":65,"alternative-id":["10.1145\/3746027.3758215","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758215","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}