{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T04:15:10Z","timestamp":1783656910122,"version":"3.55.0"},"reference-count":47,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003787","name":"Hebei Provincial Natural Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003787","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.knosys.2026.116592","type":"journal-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T07:00:52Z","timestamp":1782975652000},"page":"116592","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MeshSyne: State-space guided synergistic spatio-temporal optimization for human pose and mesh"],"prefix":"10.1016","volume":"350","author":[{"given":"Hehao","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0300-6144","authenticated-orcid":false,"given":"Zhengping","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jirui","family":"Di","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiming","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhe","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116592_bib0001","first-page":"114","article-title":"Multi-view human pose and shape estimation via mesh-aligned voxel interpolation","author":"Zhang","year":"2025","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.knosys.2026.116592_bib0002","first-page":"316","article-title":"KCM-Net: kinematic continuity-aware multi-relational cross-attention interaction network for video-based human pose and mesh reconstruction","author":"Zhang","year":"2025","journal-title":"Knowl. Based Syst."},{"key":"10.1016\/j.knosys.2026.116592_bib0003","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.109809","article-title":"Human fall detection using pose estimation: from traditional machine learning to vision transformers","volume":"143","author":"Raza","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.knosys.2026.116592_bib0004","first-page":"164","article-title":"Online signature verification based on the Lagrange formulation with 2D and 3D robotic models","author":"Diaz","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0005","first-page":"286","article-title":"CKTN: commonsense knowledge transfer network for human activity understanding","author":"Liu","year":"2024","journal-title":"Knowl. Based Syst."},{"key":"10.1016\/j.knosys.2026.116592_bib0006","first-page":"5852","article-title":"Utilizing uncertainty in 2D pose detectors for probabilistic 3D Human mesh recovery","author":"Wehrbein","year":"2025","journal-title":"IEEE Winter Confer. Applic. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0007","first-page":"1323","article-title":"TokenHMR: advancing human mesh recovery with a tokenized pose representation","author":"Dwivedi","year":"2024","journal-title":"IEEE Confer. Comput. Vis. Pattern Recogn."},{"key":"10.1016\/j.knosys.2026.116592_bib0008","first-page":"906","article-title":"Score-guided diffusion for 3D Human recovery","author":"Stathopoulos","year":"2024","journal-title":"IEEE Confer. Comput. Vis. Pattern Recogn."},{"key":"10.1016\/j.knosys.2026.116592_bib0009","first-page":"3403","article-title":"MPT: mesh pre-training with transformers for Human pose and mesh reconstruction","author":"Lin","year":"2024","journal-title":"IEEE Confer. Comput. Vis. Pattern Recogn."},{"key":"10.1016\/j.knosys.2026.116592_bib0010","first-page":"13211","article-title":"Capturing humans in motion: temporal-attentive 3D human pose and shape estimation from monocular video","author":"Wei","year":"2022","journal-title":"IEEE Confer. Comput. Vis. Pattern Recogn."},{"key":"10.1016\/j.knosys.2026.116592_bib0011","first-page":"2835","article-title":"POSE-HMR: heuristic transformer with postural prior constraints for 3D Human mesh reconstruction","author":"Pan","year":"2024","journal-title":"IEEE Int. Confer. Acoust. Speech Signal Process."},{"key":"10.1016\/j.knosys.2026.116592_bib0012","doi-asserted-by":"crossref","first-page":"3285","DOI":"10.1109\/TIP.2024.3393716","article-title":"LEAPSE: learning environment affordances for 3D Human pose and shape estimation","volume":"33","author":"Tian","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.knosys.2026.116592_bib0013","first-page":"769","article-title":"Pose2Mesh: graph convolutional network for 3D Human pose and mesh recovery from a 2D Human pose","author":"Choi","year":"2020","journal-title":"Eur. Confer. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0014","first-page":"5496","article-title":"A lightweight graph transformer network for Human mesh reconstruction from 2D Human pose","author":"Zheng","year":"2022","journal-title":"ACM Int. Confer. Multim."},{"key":"10.1016\/j.knosys.2026.116592_bib0015","first-page":"6311","article-title":"Progressive hypothesis transformer for 3D Human mesh recovery","author":"Liao","year":"2024","journal-title":"IEEE Winter Confer. Applic. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0016","first-page":"17006","article-title":"Deformable mesh transformer for 3D Human mesh recovery","author":"Yoshiyasu","year":"2023","journal-title":"IEEE Confer. Comput. Vis. Pattern Recogn."},{"key":"10.1016\/j.knosys.2026.116592_bib0017","first-page":"1","article-title":"Linear-time sequence modeling with selective State spaces","author":"Gu","year":"2024","journal-title":"Confer. Lang. Model."},{"key":"10.1016\/j.knosys.2026.116592_bib0018","first-page":"10041","article-title":"Transformers are SSMs: generalized models and efficient algorithms through structured state space duality","volume":"235","author":"Dao","year":"2024","journal-title":"Int. Confer. Mach. Learn."},{"key":"10.1016\/j.knosys.2026.116592_bib0019","first-page":"1","article-title":"AVS-Mamba: exploring temporal and multi-modal mamba for audio-visual segmentation","author":"Gong","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.knosys.2026.116592_bib0020","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/2816795.2818013","article-title":"SMPL: a skinned multiperson linear model","volume":"34","author":"Loper","year":"2015","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.knosys.2026.116592_bib0021","first-page":"1964","article-title":"Beyond static features for temporally consistent 3D human pose and shape from a video","author":"Choi","year":"2021","journal-title":"IEEE Confer. Comput. Vis. Pattern Recogn."},{"key":"10.1016\/j.knosys.2026.116592_bib0022","first-page":"249","article-title":"Action-conditioned contrastive learning for 3D human pose and shape estimation in videos","author":"Song","year":"2024","journal-title":"Comput. Vis. Image Underst."},{"key":"10.1016\/j.knosys.2026.116592_bib0023","first-page":"3004","article-title":"UNSPAT: uncertainty-guided spatiotemporal transformer for 3D human pose and shape estimation on videos","author":"Lee","year":"2024","journal-title":"IEEE\/CVF Winter Conf. Appl. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0024","first-page":"8887","article-title":"Global-to-local modeling for video-based 3D Human pose and shape estimation","author":"Shen","year":"2023","journal-title":"IEEE Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0025","first-page":"14917","article-title":"Co-evolution of pose and mesh for 3D Human body estimation from video","author":"You","year":"2023","journal-title":"IEEE Int. Conf. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0026","first-page":"2318","article-title":"HumMUSS: human motion understanding using State space models","author":"Mondal","year":"2024","journal-title":"IEEE Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0027","first-page":"3842","article-title":"PoseMamba: monocular 3D Human pose estimation with bidirectional global-local spatio","volume":"39","author":"Huang","year":"2025","journal-title":"Temporal State Space Model AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.knosys.2026.116592_bib0028","article-title":"State space model meets transformer: a new paradigm for 3D object detection","author":"Wang","year":"2025","journal-title":"Int. Conf. Learn. Represent."},{"key":"10.1016\/j.knosys.2026.116592_bib0029","first-page":"8150","article-title":"MambaPro: multi-modal object re-identification with Mamba aggregation and synergistic prompt","volume":"39","author":"Wang","year":"2025","journal-title":"AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.knosys.2026.116592_bib0030","first-page":"2252","article-title":"Learning to reconstruct 3D human pose and shape via model-fitting in the loop","author":"Kolotouros","year":"2019","journal-title":"IEEE Int. Conf. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0031","first-page":"13222","article-title":"MixSTE: seq2seq mixed spatio-temporal encoder for 3D Human pose estimation in video","author":"Zhang","year":"2022","journal-title":"IEEE Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0032","first-page":"601","article-title":"Recovering accurate 3d human pose in the wild using imus and a moving camera","author":"Marcard","year":"2018","journal-title":"Eur. Conf. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0033","doi-asserted-by":"crossref","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","article-title":"Human3.6m: large scale datasets and predictive methods for 3d human sensing in natural environments","volume":"36","author":"Ionescu","year":"2013","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.knosys.2026.116592_bib0034","first-page":"506","article-title":"Monocular 3d human pose estimation in the wild using improved cnn supervision","author":"Mehta","year":"2017","journal-title":"Int. Conf. 3D Vis. (3DV)"},{"key":"10.1016\/j.knosys.2026.116592_bib0035","first-page":"105","article-title":"Clip fusion with bi-level optimization for human mesh reconstruction from monocular videos","author":"Wu","year":"2023","journal-title":"ACM Int. Conf. Multimed."},{"key":"10.1016\/j.knosys.2026.116592_bib0036","first-page":"5662","article-title":"Two-stage co-segmentation network based on discriminative representation for recovering human mesh from videos","author":"Zhang","year":"2023","journal-title":"IEEE Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0037","doi-asserted-by":"crossref","first-page":"10564","DOI":"10.1109\/TCSVT.2024.3410400","article-title":"STAF: 3D Human mesh recovery from video with spatio-temporal alignment fusion","volume":"34","author":"Yao","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.knosys.2026.116592_bib0038","first-page":"740","article-title":"Microsoft COCO: common objects in context","author":"Lin","year":"2014","journal-title":"Eur. Conf. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0039","first-page":"3686","article-title":"2D human pose estimation: new benchmark and state of the art analysis","author":"Andriluka","year":"2014","journal-title":"IEEE Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0040","doi-asserted-by":"crossref","first-page":"38571","DOI":"10.52202\/068431-2795","article-title":"ViTPose: simple vision transformer baselines for Human pose estimation","volume":"35","author":"Xu","year":"2022","journal-title":"Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116592_bib0041","first-page":"7103","article-title":"Cascaded pyramid network for multi-person pose estimation","author":"Chen","year":"2018","journal-title":"IEEE Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0042","first-page":"13013","article-title":"Encoder-decoder with multi-level attention for 3D Human shape and pose estimation","author":"Wan","year":"2021","journal-title":"IEEE Int. Conf. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0043","first-page":"752","article-title":"I2L-MeshNet: image-to-lixel prediction network for accurate 3D Human pose and mesh estimation from a single RGB image","author":"Moon","year":"2020","journal-title":"Eur. Conf. Comput. Vis."},{"key":"10.1016\/j.knosys.2026.116592_bib0044","article-title":"Capturing the motion of every joint: 3D human pose and shape estimation with independent tokens","author":"Yang","year":"2023","journal-title":"Int. Conf. Learn. Represent."},{"key":"10.1016\/j.knosys.2026.116592_bib0045","first-page":"2070","article-title":"Wham: reconstructing world-grounded humans with accurate 3d motion","author":"Shin","year":"2024","journal-title":"IEEE Conf. Comput. Vis. Pattern Recognit."},{"key":"10.1016\/j.knosys.2026.116592_bib0046","first-page":"1","article-title":"World-grounded human motion recovery via gravity-view coordinates","volume":"144","author":"Shen","year":"2024","journal-title":"SIGGR. Asia 2024 Conf. Pap."},{"key":"10.1016\/j.knosys.2026.116592_bib0047","first-page":"10248","article-title":"Pose magic: efficient and temporally consistent human pose estimation with a hybrid mamba-gcn network","volume":"39","author":"Zhang","year":"2025","journal-title":"AAAI Conf. Artif. Intell."}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126013183?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126013183?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T03:33:11Z","timestamp":1783654391000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126013183"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":47,"alternative-id":["S0950705126013183"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116592","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MeshSyne: State-space guided synergistic spatio-temporal optimization for human pose and mesh","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116592","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116592"}}