{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T01:47:26Z","timestamp":1777945646920,"version":"3.51.4"},"reference-count":26,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022YFC3003002-03"],"award-info":[{"award-number":["2022YFC3003002-03"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["32572044"],"award-info":[{"award-number":["32572044"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62571130"],"award-info":[{"award-number":["62571130"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition Letters"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.patrec.2026.03.015","type":"journal-article","created":{"date-parts":[[2026,3,18]],"date-time":"2026-03-18T09:25:33Z","timestamp":1773825933000},"page":"8-14","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["From coarse to fine:Clip-cross hierarchical refinement network for 3D human pose estimation from monocular videos"],"prefix":"10.1016","volume":"204","author":[{"given":"Xinxin","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weitian","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianwei","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.patrec.2026.03.015_bib0001","series-title":"Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)","first-page":"1724","article-title":"Learning phrase representations using RNN encoder\u2013decoder for statistical machine translation","author":"Cho","year":"2014"},{"key":"10.1016\/j.patrec.2026.03.015_bib0002","series-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","first-page":"6000","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.patrec.2026.03.015_bib0003","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"7122","article-title":"End-to-End recovery of Human shape and pose","author":"Kanazawa","year":"2018"},{"key":"10.1016\/j.patrec.2026.03.015_bib0004","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/2816795.2818013","article-title":"SMPL: a skinned multi-person linear model","volume":"34","author":"Loper","year":"2015","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.patrec.2026.03.015_bib0005","first-page":"2252","author":"Kolotouros","year":"2019","journal-title":"Learn. Reconstruct. 3D Human Pose Shape via Model-Fitting Loop"},{"key":"10.1016\/j.patrec.2026.03.015_bib0006","series-title":"Keep It SMPL: Automatic Estimation of 3D Human Pose and Shape from a Single Image","first-page":"561","author":"Bogo","year":"2016"},{"key":"10.1016\/j.patrec.2026.03.015_bib0007","first-page":"4501","author":"Kolotouros","year":"2019","journal-title":"Convol. Mesh Regress. Single-Image Hum. Shape Reconstruct."},{"key":"10.1016\/j.patrec.2026.03.015_bib0008","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"1954","article-title":"End-to-end human pose and mesh reconstruction with transformers","author":"Lin","year":"2021"},{"key":"10.1016\/j.patrec.2026.03.015_bib0009","series-title":"2021 IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"12919","article-title":"Mesh graphormer","author":"Lin","year":"2021"},{"key":"10.1016\/j.patrec.2026.03.015_bib0010","series-title":"2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"3390","article-title":"Exploiting temporal context for 3D Human pose estimation in the wild,","author":"Arnab","year":"2019"},{"key":"10.1016\/j.patrec.2026.03.015_bib0011","series-title":"2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5607","article-title":"Learning 3D Human dynamics from video","author":"Kanazawa","year":"2019"},{"key":"10.1016\/j.patrec.2026.03.015_bib0012","series-title":"IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5252","article-title":"VIBE: video inference for human body pose and shape estimation, in: 2020","author":"Kocabas","year":"2020"},{"key":"10.1016\/j.patrec.2026.03.015_bib0013","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), IEEE","first-page":"1964","article-title":"Beyond static features for temporally consistent 3D Human pose and shape from a video","author":"Choi","year":"2021"},{"key":"10.1016\/j.patrec.2026.03.015_bib0014","doi-asserted-by":"crossref","unstructured":"W. Wei, J. Lin, T. Liu, H.M. Liao, Capturing humans in motion: Temporal-attentive 3D human pose and shape estimation from monocular video, in: 2022: pp. 13211\u201313220.","DOI":"10.1109\/CVPR52688.2022.01286"},{"key":"10.1016\/j.patrec.2026.03.015_bib0015","doi-asserted-by":"crossref","unstructured":"Z. Wan, Z. Li, M. Tian, J. Liu, S. Yi, H. Li, Encoder-decoder with multi-level attention for 3D human shape and pose estimation, in: 2021: pp. 13033\u201313042.","DOI":"10.1109\/ICCV48922.2021.01279"},{"key":"10.1016\/j.patrec.2026.03.015_bib0016","series-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), IEEE","first-page":"8887","article-title":"Global-to-local modeling for video-based 3D Human pose and shape estimation","author":"Shen","year":"2023"},{"key":"10.1016\/j.patrec.2026.03.015_bib0017","series-title":"Proceedings of the 31st ACM International Conference on Multimedia","first-page":"105","article-title":"Clip fusion with bi-level optimization for human mesh reconstruction from monocular videos","author":"Wu","year":"2023"},{"key":"10.1016\/j.patrec.2026.03.015_bib0018","series-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.patrec.2026.03.015_bib0019","doi-asserted-by":"crossref","first-page":"614","DOI":"10.1007\/978-3-030-01249-6_37","article-title":"Recovering accurate 3D Human pose in the wild using IMUs and a moving camera","author":"von Marcard","year":"2018","journal-title":"Comput. Vision \u2013 ECCV 2018"},{"key":"10.1016\/j.patrec.2026.03.015_bib0020","series-title":"2017 International Conference on 3D Vision (3DV)","first-page":"506","article-title":"Monocular 3D Human pose estimation in the wild using improved CNN supervision","author":"Mehta","year":"2017"},{"key":"10.1016\/j.patrec.2026.03.015_bib0021","doi-asserted-by":"crossref","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","article-title":"Human3.6M: large scale datasets and predictive methods for 3D Human sensing in natural environments","volume":"36","author":"Ionescu","year":"2014","journal-title":"IEEE Trans. Pattern. Anal. Mach. Intell."},{"key":"10.1016\/j.patrec.2026.03.015_bib0022","series-title":"Computer Vision \u2013 ACCV 2020: 15th Asian Conference On Computer Vision, Kyoto, Japan, November 30 \u2013 December 4, 2020, Revised Selected Papers, Part V","first-page":"324","article-title":"3D human motion estimation via motion compression and refinement","author":"Luo","year":"2020"},{"key":"10.1016\/j.patrec.2026.03.015_bib0023","series-title":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"1","article-title":"Monocular 3D Human pose estimation based on global temporal-attentive and joints-attention In video","author":"He","year":"2023"},{"key":"10.1016\/j.patrec.2026.03.015_bib0024","series-title":"Artificial Intelligence","first-page":"207","article-title":"An efficient graph transformer network for video-based Human mesh reconstruction","author":"Tang","year":"2024"},{"key":"10.1016\/j.patrec.2026.03.015_bib0025","series-title":"2024 IEEE International Conference on Multimedia and Expo (ICME)","first-page":"1","article-title":"Multi-candidate motion modeling for 3D Human pose and shape estimation from monocular video","author":"Wei","year":"2024"},{"key":"10.1016\/j.patrec.2026.03.015_bib0026","doi-asserted-by":"crossref","first-page":"10564","DOI":"10.1109\/TCSVT.2024.3410400","article-title":"STAF: 3D Human mesh recovery from video with spatio-temporal alignment fusion","volume":"34","author":"Yao","year":"2024","journal-title":"IEEE Trans. Circu. Syst. Video Technol."}],"container-title":["Pattern Recognition Letters"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001054?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001054?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,2]],"date-time":"2026-05-02T12:40:22Z","timestamp":1777725622000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167865526001054"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":26,"alternative-id":["S0167865526001054"],"URL":"https:\/\/doi.org\/10.1016\/j.patrec.2026.03.015","relation":{},"ISSN":["0167-8655"],"issn-type":[{"value":"0167-8655","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"From coarse to fine:Clip-cross hierarchical refinement network for 3D human pose estimation from monocular videos","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition Letters","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patrec.2026.03.015","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}]}}