{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T23:32:34Z","timestamp":1766791954860,"version":"3.48.0"},"reference-count":39,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,2,1]],"date-time":"2026-02-01T00:00:00Z","timestamp":1769904000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Signal Processing: Image Communication"],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1016\/j.image.2025.117440","type":"journal-article","created":{"date-parts":[[2025,11,25]],"date-time":"2025-11-25T07:44:28Z","timestamp":1764056668000},"page":"117440","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Tri-modal fusion for dynamic hand gesture recognition: Integrating RGB, depth, and skeleton data"],"prefix":"10.1016","volume":"141","author":[{"given":"Reena","family":"Tripathi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3534-3364","authenticated-orcid":false,"given":"Bindu","family":"Verma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.image.2025.117440_b1","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2022.103454","article-title":"A novel dynamic gesture understanding algorithm fusing convolutional neural networks with hand-crafted features","volume":"83","author":"Liu","year":"2022","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.image.2025.117440_b2","doi-asserted-by":"crossref","first-page":"2213","DOI":"10.1007\/s11042-019-08266-w","article-title":"Grassmann manifold based dynamic hand gesture recognition using depth data","volume":"79","author":"Verma","year":"2020","journal-title":"Multimedia Tools Appl."},{"key":"10.1016\/j.image.2025.117440_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2022.103554","article-title":"A two stream convolutional neural network with bi-directional GRU model to classify dynamic hand gesture","volume":"87","author":"Verma","year":"2022","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.image.2025.117440_b4","doi-asserted-by":"crossref","first-page":"617","DOI":"10.1007\/s11760-020-01783-4","article-title":"Fast and robust key frame extraction method for gesture video based on high-level feature representation","volume":"15","author":"Yang","year":"2021","journal-title":"Signal, Image Video Process."},{"key":"10.1016\/j.image.2025.117440_b5","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.image.2025.117440_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2023.120735","article-title":"SBI-DHGR: Skeleton-based intelligent dynamic hand gestures recognition","volume":"232","author":"Narayan","year":"2023","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.image.2025.117440_b7","doi-asserted-by":"crossref","DOI":"10.1109\/TITS.2023.3266113","article-title":"CAT-CapsNet: A convolutional and attention based capsule network to detect the driver\u2019s distraction","author":"Mittal","year":"2023","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.image.2025.117440_b8","series-title":"2023 IEEE 20th India Council International Conference","first-page":"926","article-title":"CLIP-LSTM: Fused model for dynamic hand gesture recognition","author":"Tripathi","year":"2023"},{"issue":"2","key":"10.1016\/j.image.2025.117440_b9","doi-asserted-by":"crossref","first-page":"1377","DOI":"10.1007\/s40747-022-00858-8","article-title":"Parallel temporal feature selection based on improved attention mechanism for dynamic gesture recognition","volume":"9","author":"Chen","year":"2023","journal-title":"Complex & Intell. Syst."},{"key":"10.1016\/j.image.2025.117440_b10","doi-asserted-by":"crossref","unstructured":"X. Xiaoyan, C. Panyu, Z. Zhaozhe, A Dynamic Gesture Recognition Method Based on Encoded Video, in: Proceedings of the 2022 5th International Conference on Artificial Intelligence and Pattern Recognition, 2022, pp. 711\u2013716.","DOI":"10.1145\/3573942.3574084"},{"key":"10.1016\/j.image.2025.117440_b11","doi-asserted-by":"crossref","unstructured":"T. Do, K. Vuong, H.S. Park, Egocentric scene understanding via multimodal spatial rectifier, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022, pp. 2832\u20132841.","DOI":"10.1109\/CVPR52688.2022.00285"},{"key":"10.1016\/j.image.2025.117440_b12","doi-asserted-by":"crossref","DOI":"10.1145\/3656044","article-title":"Multimodal score fusion with sparse low rank bilinear pooling for egocentric hand action recognition","author":"Roy","year":"2024","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.image.2025.117440_b13","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part III 16","first-page":"769","article-title":"Collaborative learning of gesture recognition and 3d hand pose estimation with multi-order feature analysis","author":"Yang","year":"2020"},{"key":"10.1016\/j.image.2025.117440_b14","doi-asserted-by":"crossref","unstructured":"Y. Wen, H. Pan, L. Yang, J. Pan, T. Komura, W. Wang, Hierarchical temporal transformer for 3d hand pose estimation and action recognition from egocentric rgb videos, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 21243\u201321253.","DOI":"10.1109\/CVPR52729.2023.02035"},{"key":"10.1016\/j.image.2025.117440_b15","first-page":"1","article-title":"Motion feature estimation using bi-directional GRU for skeleton-based dynamic hand gesture recognition","author":"Tripathi","year":"2024","journal-title":"Signal, Image Video Process."},{"key":"10.1016\/j.image.2025.117440_b16","series-title":"2018 24th International Conference on Pattern Recognition","first-page":"3365","article-title":"Multimodal gesture recognition using densely connected convolution and blstm","author":"Li","year":"2018"},{"key":"10.1016\/j.image.2025.117440_b17","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2020.114499","article-title":"Selective spatiotemporal features learning for dynamic gesture recognition","volume":"169","author":"Tang","year":"2021","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.image.2025.117440_b18","first-page":"1","article-title":"Survey on vision-based dynamic hand gesture recognition","author":"Tripathi","year":"2023","journal-title":"Vis. Comput."},{"key":"10.1016\/j.image.2025.117440_b19","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2019.107040","article-title":"Learning binary code for fast nearest subspace search","volume":"98","author":"Zhou","year":"2020","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.image.2025.117440_b20","doi-asserted-by":"crossref","unstructured":"X.S. Nguyen, L. Brun, O. L\u00e9zoray, S. Bougleux, A neural network based on SPD manifold learning for skeleton-based hand gesture recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 12036\u201312045.","DOI":"10.1109\/CVPR.2019.01231"},{"issue":"4","key":"10.1016\/j.image.2025.117440_b21","doi-asserted-by":"crossref","first-page":"7823","DOI":"10.1109\/LRA.2021.3101822","article-title":"Domain and view-point agnostic hand action recognition","volume":"6","author":"Sabater","year":"2021","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.image.2025.117440_b22","doi-asserted-by":"crossref","unstructured":"J. Liu, Y. Liu, Y. Wang, V. Prinet, S. Xiang, C. Pan, Decoupled representation learning for skeleton-based gesture recognition, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 5751\u20135760.","DOI":"10.1109\/CVPR42600.2020.00579"},{"key":"10.1016\/j.image.2025.117440_b23","doi-asserted-by":"crossref","DOI":"10.1109\/TCDS.2023.3242988","article-title":"An efficient graph convolution network for skeleton-based dynamic hand gesture recognition","author":"Peng","year":"2023","journal-title":"IEEE Trans. Cogn. Dev. Syst."},{"key":"10.1016\/j.image.2025.117440_b24","first-page":"1","article-title":"MVHANet: multi-view hierarchical aggregation network for skeleton-based hand gesture recognition","author":"Li","year":"2023","journal-title":"Signal, Image Video Process."},{"key":"10.1016\/j.image.2025.117440_b25","doi-asserted-by":"crossref","first-page":"4433","DOI":"10.1109\/TMM.2021.3117124","article-title":"Graph-based multimodal sequential embedding for sign language translation","volume":"24","author":"Tang","year":"2021","journal-title":"IEEE Trans. Multimed."},{"issue":"4","key":"10.1016\/j.image.2025.117440_b26","first-page":"1","article-title":"Semi-supervised RGB-D hand gesture recognition via mutual learning of self-supervised models","volume":"21","author":"Zhang","year":"2025","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"issue":"4","key":"10.1016\/j.image.2025.117440_b27","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3663572","article-title":"Gloss-driven conditional diffusion models for sign language production","volume":"21","author":"Tang","year":"2025","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.image.2025.117440_b28","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"7266","article-title":"Sign-idd: Iconicity disentangled diffusion for sign language production","volume":"39","author":"Tang","year":"2025"},{"key":"10.1016\/j.image.2025.117440_b29","doi-asserted-by":"crossref","unstructured":"S. Wang, D. Guo, W.-g. Zhou, Z.-J. Zha, M. Wang, Connectionist temporal fusion for sign language translation, in: Proceedings of the 26th ACM International Conference on Multimedia, 2018, pp. 1483\u20131491.","DOI":"10.1145\/3240508.3240671"},{"key":"10.1016\/j.image.2025.117440_b30","series-title":"IJCAI","first-page":"8","article-title":"Dense temporal convolution network for sign language translation.","volume":"vol. 2","author":"Guo","year":"2019"},{"key":"10.1016\/j.image.2025.117440_b31","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Linguistics-vision monotonic consistent network for sign language production","author":"Wang","year":"2025"},{"key":"10.1016\/j.image.2025.117440_b32","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"issue":"18","key":"10.1016\/j.image.2025.117440_b33","doi-asserted-by":"crossref","first-page":"8972","DOI":"10.3390\/app12188972","article-title":"Deep residual learning for image recognition: A survey","volume":"12","author":"Shafiq","year":"2022","journal-title":"Appl. Sci."},{"issue":"3","key":"10.1016\/j.image.2025.117440_b34","article-title":"Dynamic hand gesture recognition using 3d-cnn and lstm networks","volume":"70","author":"Ur Rehman","year":"2021","journal-title":"Comput. Mater. Contin."},{"key":"10.1016\/j.image.2025.117440_b35","doi-asserted-by":"crossref","unstructured":"G. Garcia-Hernando, S. Yuan, S. Baek, T.-K. Kim, First-person hand action benchmark with rgb-d videos and 3d hand pose annotations, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 409\u2013419.","DOI":"10.1109\/CVPR.2018.00050"},{"key":"10.1016\/j.image.2025.117440_b36","unstructured":"L. Liu, L. Shao, Learning discriminative representations from RGB-D video data, in: Twenty-Third International Joint Conference on Artificial Intelligence, 2013."},{"year":"2024","series-title":"In my perspective, in my hands: Accurate egocentric 2D hand pose and action recognition","author":"Mucha","key":"10.1016\/j.image.2025.117440_b37"},{"key":"10.1016\/j.image.2025.117440_b38","doi-asserted-by":"crossref","unstructured":"H. Cho, C. Kim, J. Kim, S. Lee, E. Ismayilzada, S. Baek, Transformer-based unified recognition of two hands manipulating objects, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 4769\u20134778.","DOI":"10.1109\/CVPR52729.2023.00462"},{"key":"10.1016\/j.image.2025.117440_b39","doi-asserted-by":"crossref","unstructured":"B. Verma, A. Choudhary, Dynamic hand gesture recognition using convolutional neural network with RGB-D fusion, in: Proceedings of the 11th Indian Conference on Computer Vision, Graphics and Image Processing, 2018, pp. 1\u20138.","DOI":"10.1145\/3293353.3293421"}],"container-title":["Signal Processing: Image Communication"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0923596525001869?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0923596525001869?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2025,12,26]],"date-time":"2025-12-26T23:31:41Z","timestamp":1766791901000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0923596525001869"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2]]},"references-count":39,"alternative-id":["S0923596525001869"],"URL":"https:\/\/doi.org\/10.1016\/j.image.2025.117440","relation":{},"ISSN":["0923-5965"],"issn-type":[{"type":"print","value":"0923-5965"}],"subject":[],"published":{"date-parts":[[2026,2]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Tri-modal fusion for dynamic hand gesture recognition: Integrating RGB, depth, and skeleton data","name":"articletitle","label":"Article Title"},{"value":"Signal Processing: Image Communication","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.image.2025.117440","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2025 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"117440"}}