{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:22:01Z","timestamp":1778048521880,"version":"3.51.4"},"reference-count":85,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00519","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"5352-5363","source":"Crossref","is-referenced-by-count":0,"title":["DF-Mamba: Deformable State Space Modeling for 3D Hand Pose Estimation in Interactions"],"prefix":"10.1109","author":[{"given":"Yifan","family":"Zhou","sequence":"first","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takehiko","family":"Ohkawa","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guwenxiao","family":"Zhou","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kanoko","family":"Goto","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takumi","family":"Hirose","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yusuke","family":"Sekikawa","sequence":"additional","affiliation":[{"name":"Denso IT Laboratory"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nakamasa","family":"Inoue","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Assemblyhands-x: Modeling 3d hand-body coordination for understanding bimanual human activities","author":"Banno","year":"2025"},{"key":"ref2","volume-title":"Modern Control Theory","author":"Brogan","year":"1974"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00893"},{"key":"ref5","article-title":"Generating realistic training images based on tonality-alignment generative adversarial networks for hand pose estimation","author":"Chen","year":"2018"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01989"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.195"},{"key":"ref8","article-title":"Twins: Revisiting the design of spatial attention in vision transformers","volume-title":"Proc. Annual Conference on Neural Information Processing Systems (NeurIPS)","author":"Chu"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.89"},{"key":"ref10","first-page":"933","article-title":"Language modeling with gated convolutional networks","volume-title":"Proc. International Conference on Machine Learning (ICML)","author":"Dauphin"},{"key":"ref11","first-page":"2127","article-title":"Hamba: Single-view 3d hand reconstruction with graph-guided bi-scanning mamba","volume-title":"Proc. Annual Conference on Neural Information Processing Systems (NeurIPS)","volume":"37","author":"Dong"},{"key":"ref12","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. International Conference on Learning Representations (ICLR)","author":"Dosovitskiy"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2006.10.012"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/3DV53792.2021.00011"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01244"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72698-9_25"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58539-6_8"},{"key":"ref18","article-title":"Hungry hungry hippos: Towards language modeling with state space models","volume-title":"Proc. International Conference on Learning Representations (ICLR)","author":"Fu"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01109"},{"key":"ref20","article-title":"Mamba: Linear-time sequence modeling with selective state spaces","volume-title":"Proc. Conference on Language Modeling (COLM)","author":"Gu"},{"key":"ref21","first-page":"1474","article-title":"Hippo: Recurrent memory with optimal polynomial projections","volume-title":"Proc. Annual Conference on Neural Information Processing Systems (NeurIPS)","author":"Gu"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612390"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01081"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3550469.3555378"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02352"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref27","article-title":"Mobilenets: Efficient convolutional neural networks for mobile vision applications","author":"Howard","year":"2017"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413775"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-91979-4_2"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i4.32401"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_8"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00854"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1115\/1.3662552"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00102"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00278"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-63820-7_52"},{"key":"ref37","article-title":"SiM-Hand: Pre-training for 3d hand pose estimation with contrastive learning on large-scale hand images in the wild","volume-title":"Proc. International Conference on Learning Representations (ICLR)","author":"Lin"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00071"},{"key":"ref39","article-title":"Leveraging rgb images for pre-training of event-based hand pose estimation","author":"Liu","year":"2025"},{"key":"ref40","article-title":"VMamba: Visual state space model","volume-title":"Proc. Annual Conference on Neural Information Processing Systems (NeurIPS)","author":"Liu"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00746"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20068-7_22"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00533"},{"key":"ref46","first-page":"548","article-title":"Interhand 2.6M: A dataset and baseline for 3d interacting hand pose estimation from a single rgb image","volume-title":"Proc. European Conference on Computer Vision (ECCV)","author":"Moon"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_5"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01856-0"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01249"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00515"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00155"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00938"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i6.32690"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681709"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73229-4_11"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00736"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00017"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58520-4_13"},{"key":"ref59","article-title":"Affordance-guided diffusion prior for 3d hand reconstruction","author":"Suzuki","year":"2025"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1145\/2629500"},{"key":"ref61","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"Proc. International Conference on Machine Learning (ICML)","author":"Touvron"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00171"},{"key":"ref63","article-title":"Attention is all you need","volume-title":"Proc. Annual Conference on Neural Information Processing Systems (NeurIPS)","author":"Vaswani"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2983686"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.00061"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02035"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-91578-9_3"},{"key":"ref68","article-title":"Spatial-mamba: Effective visual state space models via structure-aware state fusion","volume-title":"Proc. International Conference on Learning Representations (ICLR)","author":"Xiao"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00088"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611762"},{"key":"ref71","article-title":"Plainmamba: Improving non-hierarchical mamba in visual recognition","volume-title":"Proc. British Machine Vision Conference (BMVC)","author":"Yang"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01011"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681068"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00423"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01245"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01116"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i10.33112"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413651"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00539"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00136"},{"key":"ref81","article-title":"Vision Mamba: Efficient visual representation learning with bidirectional state space model","volume-title":"Proc. International Conference on Machine Learning (ICML)","author":"Zhu"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00953"},{"key":"ref83","article-title":"Deformable detr: Deformable transformers for end-to-end object detection","volume-title":"Proc. International Conference on Learning Representations (ICLR)","author":"Zhu"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.525"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00090"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492348.pdf?arnumber=11492348","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:06:14Z","timestamp":1778047574000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492348\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":85,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00519","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}