{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T01:14:54Z","timestamp":1783214094535,"version":"3.54.6"},"reference-count":60,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013804","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013804","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neucom.2026.134382","type":"journal-article","created":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T20:15:34Z","timestamp":1782850534000},"page":"134382","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["HFAT-HMR: Empowering ViT for human mesh recovery via high-frequency enhancement and auxiliary tokens"],"prefix":"10.1016","volume":"699","author":[{"given":"Jinming","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingjie","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6752-0252","authenticated-orcid":false,"given":"Linlin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Boshu","family":"Jia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baochang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3481-7820","authenticated-orcid":false,"given":"Xiaoyu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Libiao","family":"Jin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.134382_bib0005","series-title":"2024 International Conference on Engineering & Computing Technologies (ICECT)","first-page":"1","article-title":"Drone-based human action recognition for surveillance: a multi-feature approach","author":"Abbas","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0010","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.103079","article-title":"TSCMamba: Mamba meets multi-view learning for time series classification","volume":"120","author":"Ahamed","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.neucom.2026.134382_bib0015","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3686","article-title":"2D human pose estimation: new benchmark and state of the art analysis","author":"Andriluka","year":"2014"},{"key":"10.1016\/j.neucom.2026.134382_bib0020","series-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention","first-page":"736","article-title":"Laplacian-former: overcoming the limitations of vision transformers in local texture detection","author":"Azad","year":"2023"},{"key":"10.1016\/j.neucom.2026.134382_bib0025","series-title":"European Conference on Computer Vision","first-page":"1","article-title":"Improving vision transformers by revisiting high-frequency components","author":"Bai","year":"2022"},{"key":"10.1016\/j.neucom.2026.134382_bib0030","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8726","article-title":"Bedlam: a synthetic dataset of bodies exhibiting detailed lifelike animated motion","author":"Black","year":"2023"},{"key":"10.1016\/j.neucom.2026.134382_bib0035","series-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention","first-page":"266","article-title":"Em-net: efficient channel and frequency learning with Mamba for 3D medical image segmentation","author":"Chang","year":"2024"},{"issue":"12","key":"10.1016\/j.neucom.2026.134382_bib0040","doi-asserted-by":"crossref","first-page":"10763","DOI":"10.1109\/TPAMI.2024.3449959","article-title":"Frequency-aware feature fusion for dense image prediction","volume":"46","author":"Chen","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134382_bib0045","series-title":"Forty-Second International Conference on Machine Learning","article-title":"ExtPose: robust and coherent pose estimation by extending ViTs","author":"Chen","year":"2025"},{"key":"10.1016\/j.neucom.2026.134382_bib0050","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6208","article-title":"ShadowRefiner: towards mask-free shadow removal via fast Fourier transformer","author":"Dong","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0055","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1323","article-title":"Tokenhmr: advancing human mesh recovery with a tokenized pose representation","author":"Dwivedi","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0060","series-title":"European Conference on Computer Vision","first-page":"19","article-title":"Spiking wavelet transformer","author":"Fang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0065","series-title":"European Conference on Computer Vision","first-page":"363","article-title":"Wavelet convolutions for large receptive fields","author":"Finder","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0070","series-title":"European Conference on Computer Vision","first-page":"37","article-title":"Weconvene: learned image compression with wavelet-domain convolution and entropy model","author":"Fu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0075","author":"Gao"},{"key":"10.1016\/j.neucom.2026.134382_bib0080","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"14783","article-title":"Humans in 4D: reconstructing and tracking humans with transformers","author":"Goel","year":"2023"},{"key":"10.1016\/j.neucom.2026.134382_bib0085","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6047","article-title":"Ava: a video dataset of spatio-temporally localized atomic visual actions","author":"Gu","year":"2018"},{"key":"10.1016\/j.neucom.2026.134382_bib0090","doi-asserted-by":"crossref","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","article-title":"Human3. 6m: large scale datasets and predictive methods for 3D human sensing in natural environments","volume":"36","author":"Ionescu","year":"2013","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134382_bib0095","series-title":"European Conference on Computer Vision","first-page":"381","article-title":"When fast Fourier transform meets transformer for image restoration","author":"Jiang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0100","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"5614","article-title":"Learning 3D human dynamics from video","author":"Kanazawa","year":"2019"},{"key":"10.1016\/j.neucom.2026.134382_bib0105","doi-asserted-by":"crossref","first-page":"36372","DOI":"10.1109\/ACCESS.2024.3373199","article-title":"Human action recognition systems: a review of the trends and state-of-the-art","volume":"12","author":"Karim","year":"2024","journal-title":"IEEE Access"},{"key":"10.1016\/j.neucom.2026.134382_bib0110","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"14632","article-title":"Emdb: the electromagnetic database of global 3D human pose and shape in the wild","author":"Kaufmann","year":"2023"},{"key":"10.1016\/j.neucom.2026.134382_bib0115","author":"Li"},{"key":"10.1016\/j.neucom.2026.134382_bib0120","series-title":"European Conference on Computer Vision","first-page":"590","article-title":"Cliff: carrying location information in full frames into human pose and shape estimation","author":"Li","year":"2022"},{"key":"10.1016\/j.neucom.2026.134382_bib0125","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"17184","article-title":"Cross-domain 3D hand pose estimation with dual modalities","author":"Lin","year":"2023"},{"key":"10.1016\/j.neucom.2026.134382_bib0130","series-title":"European Conference on Computer Vision","first-page":"740","article-title":"Microsoft COCO: common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.neucom.2026.134382_bib0135","doi-asserted-by":"crossref","first-page":"4483","DOI":"10.1007\/s11263-025-02393-8","article-title":"Part-whole relational fusion towards multi-modal scene understanding","volume":"133","author":"Liu","year":"2025","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.134382_bib0140","first-page":"3688","article-title":"Part-object relational visual saliency","volume":"44","author":"Liu","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134382_bib0145","doi-asserted-by":"crossref","DOI":"10.1016\/j.engstruct.2024.118390","article-title":"Target-free measurement of cable forces based on computer vision and equivalent frequency difference","volume":"314","author":"Luo","year":"2024","journal-title":"Eng. Struct."},{"key":"10.1016\/j.neucom.2026.134382_bib0150","doi-asserted-by":"crossref","first-page":"7354","DOI":"10.1109\/TCSVT.2023.3281462","article-title":"Multi-modal image fusion via deep laplacian pyramid hybrid network","volume":"33","author":"Luo","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.neucom.2026.134382_bib0155","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision","first-page":"6920","article-title":"Motionagformer: enhancing 3D human pose estimation with a transformer-gcnformer network","author":"Mehraban","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0160","series-title":"2017 International Conference on 3D Vision (3DV)","first-page":"506","article-title":"Monocular 3D human pose estimation in the wild using improved CNN supervision","author":"Mehta","year":"2017"},{"key":"10.1016\/j.neucom.2026.134382_bib0165","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1186","article-title":"Eventego3d: 3D human motion capture from egocentric event streams","author":"Millerdurai","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0170","doi-asserted-by":"crossref","first-page":"11340","DOI":"10.1109\/TCSVT.2024.3423411","article-title":"From methods to applications: a review of deep 3D human motion capture","volume":"34","author":"Niu","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.neucom.2026.134382_bib0175","doi-asserted-by":"crossref","first-page":"14541","DOI":"10.52202\/068431-1057","article-title":"Fast vision transformers with Hilo attention","volume":"35","author":"Pan","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134382_bib0180","doi-asserted-by":"crossref","first-page":"26034","DOI":"10.52202\/068431-1888","article-title":"Benchmarking and analyzing 3D human pose and shape estimation beyond algorithms","volume":"35","author":"Pang","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134382_bib0185","series-title":"2025 International Conference on 3D Vision (3DV)","first-page":"1562","article-title":"Camerahmr: aligning people with perspective","author":"Patel","year":"2025"},{"key":"10.1016\/j.neucom.2026.134382_bib0190","series-title":"2025 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","first-page":"9543","article-title":"Spectformer: frequency and attention is what you need in a vision transformer","author":"Patro","year":"2025"},{"key":"10.1016\/j.neucom.2026.134382_bib0195","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"12242","article-title":"Wilor: end-to-end 3D hand localization and reconstruction in-the-wild","author":"Potamias","year":"2025"},{"key":"10.1016\/j.neucom.2026.134382_bib0200","doi-asserted-by":"crossref","first-page":"0100","DOI":"10.34133\/cbsystems.0100","article-title":"A survey on 3D skeleton-based action recognition using learning method","volume":"5","author":"Ren","year":"2024","journal-title":"Cyborg and Bionic Systems"},{"key":"10.1016\/j.neucom.2026.134382_bib0205","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"2070","article-title":"Wham: reconstructing world-grounded humans with accurate 3D motion","author":"Shin","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0210","doi-asserted-by":"crossref","first-page":"23495","DOI":"10.52202\/068431-1707","article-title":"Inception transformer","volume":"35","author":"Si","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134382_bib0215","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops","first-page":"138","article-title":"Training sparse neural networks","author":"Srinivas","year":"2017"},{"key":"10.1016\/j.neucom.2026.134382_bib0220","author":"Sun"},{"issue":"12","key":"10.1016\/j.neucom.2026.134382_bib0225","doi-asserted-by":"crossref","first-page":"8000","DOI":"10.1109\/TIV.2024.3405990","article-title":"Driver distraction behavior recognition for autonomous driving: approaches, datasets and challenges","volume":"9","author":"Tan","year":"2024","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.neucom.2026.134382_bib0230","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"17763","article-title":"Fourier-basis functions to bridge augmentation gap: rethinking frequency augmentation in image classification","author":"Vaish","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0235","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134382_bib0240","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"601","article-title":"Recovering accurate 3D human pose in the wild using IMUs and a moving camera","author":"Von Marcard","year":"2018"},{"key":"10.1016\/j.neucom.2026.134382_bib0245","author":"Wang"},{"key":"10.1016\/j.neucom.2026.134382_bib0250","series-title":"Proceedings of the Computer Vision and Pattern Recognition Conference","first-page":"1148","article-title":"PromptHMR: promptable human mesh recovery","author":"Wang","year":"2025"},{"key":"10.1016\/j.neucom.2026.134382_bib0255","author":"Wu"},{"key":"10.1016\/j.neucom.2026.134382_bib0260","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"4794","article-title":"Vision transformer with deformable attention","author":"Xia","year":"2022"},{"key":"10.1016\/j.neucom.2026.134382_bib0265","doi-asserted-by":"crossref","first-page":"1783","DOI":"10.1109\/TMM.2024.3521798","article-title":"Frequency-assisted Mamba for remote sensing image super-resolution","volume":"27","author":"Xiao","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.neucom.2026.134382_bib0270","series-title":"Proceedings of the Asian Conference on Computer Vision","first-page":"4334","article-title":"Window-based channel attention for wavelet-enhanced learned image compression","author":"Xu","year":"2024"},{"key":"10.1016\/j.neucom.2026.134382_bib0275","author":"Yang"},{"key":"10.1016\/j.neucom.2026.134382_bib0280","series-title":"European Conference on Computer Vision","first-page":"328","article-title":"Wave-vit: unifying wavelet and transformers for visual representation learning","author":"Yao","year":"2022"},{"key":"10.1016\/j.neucom.2026.134382_bib0285","doi-asserted-by":"crossref","first-page":"1053","DOI":"10.1080\/10589759.2024.2338921","article-title":"GBC-BCD: an improved bridge crack detection method based on bidirectional laplacian pyramid structure with lightweight attention mechanism convolution","volume":"40","author":"Zhang","year":"2025","journal-title":"Nondestruct. Test. Eval."},{"key":"10.1016\/j.neucom.2026.134382_bib0290","series-title":"European Conference on Computer Vision","first-page":"467","article-title":"Lapose: laplacian mixture shape modeling for RGB-based category-level object pose estimation","author":"Zhang","year":"2024"},{"issue":"7","key":"10.1016\/j.neucom.2026.134382_bib0295","doi-asserted-by":"crossref","first-page":"5847","DOI":"10.1109\/TPAMI.2025.3555485","article-title":"Cross-modality distillation for multi-modal tracking","volume":"47","author":"Zhang","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134382_bib0300","doi-asserted-by":"crossref","DOI":"10.1016\/j.robot.2023.104580","article-title":"Towards comprehensive understanding of pedestrians for autonomous driving: efficient multi-task-learning-based pedestrian detection, tracking and attribute recognition","volume":"171","author":"Zhou","year":"2024","journal-title":"Robot. Auton. Syst."}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017807?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017807?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:18:10Z","timestamp":1783210690000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226017807"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":60,"alternative-id":["S0925231226017807"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134382","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"HFAT-HMR: Empowering ViT for human mesh recovery via high-frequency enhancement and auxiliary tokens","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134382","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"134382"}}