{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T09:54:59Z","timestamp":1772790899665,"version":"3.50.1"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,2,1]],"date-time":"2023-02-01T00:00:00Z","timestamp":1675209600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,2,1]],"date-time":"2023-02-01T00:00:00Z","timestamp":1675209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100003787","name":"Natural Science Foundation of Hebei Province","doi-asserted-by":"publisher","award":["F2019201451"],"award-info":[{"award-number":["F2019201451"]}],"id":[{"id":"10.13039\/501100003787","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Science and Technology Project of Hebei Education Department","award":["QN2018214"],"award-info":[{"award-number":["QN2018214"]}]},{"name":"Science and Technology Project of Hebei Education Department","award":["ZD2019131"],"award-info":[{"award-number":["ZD2019131"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Pattern Anal Applic"],"published-print":{"date-parts":[[2023,5]]},"DOI":"10.1007\/s10044-023-01130-6","type":"journal-article","created":{"date-parts":[[2023,2,1]],"date-time":"2023-02-01T16:05:46Z","timestamp":1675267546000},"page":"591-603","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["MSRT: multi-scale representation transformer for regression-based human pose estimation"],"prefix":"10.1007","volume":"26","author":[{"given":"Beiguang","family":"Shan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingxuan","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fang","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,2,1]]},"reference":[{"key":"1130_CR1","doi-asserted-by":"crossref","unstructured":"Geng Z, Sun K, Xiao B, Zhang Z, Wang J (2021) Bottom-up human pose estimation via disentangled keypoint regression. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 14676\u201314686","DOI":"10.1109\/CVPR46437.2021.01444"},{"key":"1130_CR2","doi-asserted-by":"crossref","unstructured":"Su C, Li J, Zhang S, Xing J, Gao W, Tian Q (2017) Pose-driven deep convolutional model for person re-identification. In: Proceedings of the IEEE international conference on computer vision, pp. 3960\u20133969","DOI":"10.1109\/ICCV.2017.427"},{"issue":"4","key":"1130_CR3","doi-asserted-by":"publisher","first-page":"1307","DOI":"10.1007\/s10044-018-0727-y","volume":"22","author":"M Farrajota","year":"2019","unstructured":"Farrajota M, Rodrigues JM, du Buf JH (2019) Human action recognition in videos with articulated pose information by deep networks. Pattern Anal Appl 22(4):1307\u20131318","journal-title":"Pattern Anal Appl"},{"key":"1130_CR4","doi-asserted-by":"crossref","unstructured":"Xiao B, Wu H, Wei Y (2018) Simple baselines for human pose estimation and tracking. In: Proceedings of the European conference on computer vision (ECCV), pp. 466\u2013481","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"1130_CR5","doi-asserted-by":"crossref","unstructured":"Sun K, Xiao B, Liu D, Wang J (2019) Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 5693\u20135703","DOI":"10.1109\/CVPR.2019.00584"},{"key":"1130_CR6","doi-asserted-by":"crossref","unstructured":"Sun X, Xiao B, Wei F, Liang S, Wei Y (2018) Integral human pose regression. In: Proceedings of the European conference on computer vision (ECCV), pp. 529\u2013545","DOI":"10.1007\/978-3-030-01231-1_33"},{"key":"1130_CR7","doi-asserted-by":"crossref","unstructured":"Wei F, Sun X, Li H, Wang J, Lin S (2020) Point-set anchors for object detection, instance segmentation and pose estimation. In: European conference on computer vision, pp. 527\u2013544","DOI":"10.1007\/978-3-030-58607-2_31"},{"key":"1130_CR8","doi-asserted-by":"crossref","unstructured":"Fang H.-S, Xie S, Tai Y.-W, Lu C (2017) Rmpe: regional multi-person pose estimation. In: Proceedings of the IEEE international conference on computer vision, pp. 2334\u20132343","DOI":"10.1109\/ICCV.2017.256"},{"key":"1130_CR9","doi-asserted-by":"crossref","unstructured":"Li J, Wang C, Zhu H, Mao Y, Fang H-S, Lu C (2019) Crowdpose: efficient crowded scenes pose estimation and a new benchmark. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 10863\u201310872","DOI":"10.1109\/CVPR.2019.01112"},{"key":"1130_CR10","unstructured":"Hidalgo G, Raaj Y, Idrees H, Xiang D, Joo H, Simon T, Sheikh Y (2019) Single-network whole-body pose estimation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 6982\u20136991"},{"key":"1130_CR11","doi-asserted-by":"publisher","first-page":"269","DOI":"10.1016\/j.neucom.2016.09.033","volume":"219","author":"Q Shi","year":"2017","unstructured":"Shi Q, Di H, Lu Y, Lv F, Tian X (2017) Video pose estimation with global motion cues. Neurocomputing 219:269\u2013279","journal-title":"Neurocomputing"},{"key":"1130_CR12","doi-asserted-by":"crossref","unstructured":"Zhou T, Wang W, Liu S, Yang Y, Van\u00a0Gool L (2021) Differentiable multi-granularity human representation learning for instance-aware human semantic parsing. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 1622\u20131631","DOI":"10.1109\/CVPR46437.2021.00167"},{"key":"1130_CR13","doi-asserted-by":"crossref","unstructured":"Zhou L, Chen Y, Gao Y, Wang J, Lu H (2020) Occlusion-aware Siamese network for human pose estimation. In: European conference on computer vision, pp. 396\u2013412","DOI":"10.1007\/978-3-030-58565-5_24"},{"key":"1130_CR14","first-page":"5998","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30:5998\u20136008","journal-title":"Adv Neural Inf Process Syst"},{"key":"1130_CR15","doi-asserted-by":"crossref","unstructured":"Sun X, Shang J, Liang S, Wei Y (2017) Compositional human pose regression. In: Proceedings of the IEEE international conference on computer vision, pp. 2602\u20132611","DOI":"10.1109\/ICCV.2017.284"},{"key":"1130_CR16","doi-asserted-by":"crossref","unstructured":"Li K, Wang S, Zhang X, Xu Y, Xu W, Tu Z (2021) Pose recognition with cascade transformers. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 1944\u20131953","DOI":"10.1109\/CVPR46437.2021.00198"},{"key":"1130_CR17","doi-asserted-by":"crossref","unstructured":"Papandreou G, Zhu T, Kanazawa N, Toshev A, Tompson J, Bregler C, Murphy K(2017) Towards accurate multi-person pose estimation in the wild. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 4903\u20134911","DOI":"10.1109\/CVPR.2017.395"},{"key":"1130_CR18","doi-asserted-by":"crossref","unstructured":"Su K, Yu D, Xu Z, Geng X, Wang C (2019) Multi-person pose estimation with enhanced channel-wise and spatial information. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 5674\u20135682","DOI":"10.1109\/CVPR.2019.00582"},{"key":"1130_CR19","unstructured":"Li W, Wang Z, Yin B, Peng Q, Du Y, Xiao T, Yu G, Lu H, Wei Y, Sun J (2019) Rethinking on multi-stage networks for human pose estimation. arXiv preprint arXiv:1901.00148"},{"key":"1130_CR20","doi-asserted-by":"crossref","unstructured":"Wang J, Long X, Gao Y, Ding E, Wen S (2020) Graph-PCNN: two stage human pose estimation with graph pose refinement. In: European conference on computer vision, pp. 492\u2013508","DOI":"10.1007\/978-3-030-58621-8_29"},{"key":"1130_CR21","doi-asserted-by":"crossref","unstructured":"Toshev A, Szegedy C (2014) Human pose estimation via deep neural networks. CVPR.(Columbus, Ohio, 2014), pp. 1653\u20131660","DOI":"10.1109\/CVPR.2014.214"},{"key":"1130_CR22","doi-asserted-by":"crossref","unstructured":"Carreira J, Agrawal P, Fragkiadaki K, Malik J (2016) Human pose estimation with iterative error feedback. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 4733\u20134742","DOI":"10.1109\/CVPR.2016.512"},{"key":"1130_CR23","unstructured":"Tian Z, Chen H, Shen C (2019) Directpose: direct end-to-end multi-person pose estimation. arXiv preprint arXiv:1911.07451"},{"key":"1130_CR24","unstructured":"Zhou X, Wang D, Kr\u00e4henb\u00fchl P (2019) Objects as points. arXiv preprint arXiv:1904.07850"},{"key":"1130_CR25","doi-asserted-by":"crossref","unstructured":"Nie X, Feng J, Zhang J, Yan S (2019) Single-stage multi-person pose machines. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 6951\u20136960","DOI":"10.1109\/ICCV.2019.00705"},{"key":"1130_CR26","doi-asserted-by":"crossref","unstructured":"Li J, Bian S, Zeng A, Wang C, Pang B, Liu W, Lu C (2021) Human pose regression with residual log-likelihood estimation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 11025\u201311034","DOI":"10.1109\/ICCV48922.2021.01084"},{"key":"1130_CR27","doi-asserted-by":"crossref","unstructured":"Mao W, Ge Y, Shen C, Tian Z, Wang X, Wang Z, Hengel A.V.D (2022) Poseur: direct human pose regression with transformers. arXiv preprint arXiv:2201.07412","DOI":"10.1007\/978-3-031-20068-7_5"},{"key":"1130_CR28","doi-asserted-by":"crossref","unstructured":"Wang W, Song H, Zhao S, Shen J, Zhao S, Hoi S.C, Ling H (2019) Learning unsupervised video object segmentation through visual attention. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 3064\u20133074","DOI":"10.1109\/CVPR.2019.00318"},{"key":"1130_CR29","doi-asserted-by":"publisher","first-page":"8326","DOI":"10.1109\/TIP.2020.3013162","volume":"29","author":"T Zhou","year":"2020","unstructured":"Zhou T, Li J, Wang S, Tao R, Shen J (2020) Matnet: motion-attentive transition network for zero-shot video object segmentation. IEEE Trans Image Process 29:8326\u20138338","journal-title":"IEEE Trans Image Process"},{"key":"1130_CR30","doi-asserted-by":"crossref","unstructured":"Wang W, Zhao S, Shen J, Hoi S.C, Borji A (2019) Salient object detection with pyramid attention and salient edges. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 1448\u20131457","DOI":"10.1109\/CVPR.2019.00154"},{"key":"1130_CR31","doi-asserted-by":"crossref","unstructured":"Fan D.-P, Wang W, Cheng M.-M, Shen J (2019) Shifting more attention to video salient object detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 8554\u20138564","DOI":"10.1109\/CVPR.2019.00875"},{"issue":"5","key":"1130_CR32","doi-asserted-by":"publisher","first-page":"2368","DOI":"10.1109\/TIP.2017.2787612","volume":"27","author":"W Wang","year":"2017","unstructured":"Wang W, Shen J (2017) Deep visual attention prediction. IEEE Trans Image Process 27(5):2368\u20132378","journal-title":"IEEE Trans Image Process"},{"key":"1130_CR33","doi-asserted-by":"crossref","unstructured":"Wang W, Shen J (2017) Deep cropping via attention box prediction and aesthetics assessment. In: Proceedings of the IEEE international conference on computer vision, pp. 2186\u20132194","DOI":"10.1109\/ICCV.2017.240"},{"key":"1130_CR34","unstructured":"Zhu X, Su W, Lu L, Li B, Wang X, Dai J (2020) Deformable DETR: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159"},{"key":"1130_CR35","unstructured":"Yang S, Quan Z, Nie M, Yang W (2020) Transpose: towards explainable human pose estimation by transformer. arXiv preprint arXiv:2012.14214"},{"key":"1130_CR36","doi-asserted-by":"crossref","unstructured":"Khan S, Naseer M, Hayat M, Zamir S.W, Khan F.S, Shah M (2021) Transformers in vision: a survey. arXiv preprint arXiv:2101.01169","DOI":"10.1145\/3505244"},{"key":"1130_CR37","doi-asserted-by":"crossref","unstructured":"Zheng C, Zhu S, Mendieta M, Yang T, Chen C, Ding Z (2021) 3d human pose estimation with spatial and temporal transformers. arXiv preprint arXiv:2103.10455","DOI":"10.1109\/ICCV48922.2021.01145"},{"key":"1130_CR38","unstructured":"Han K, Wang Y, Chen H, Chen X, Guo J, Liu Z, Tang Y, Xiao A, Xu C, Xu Y, et al. (2020) A survey on visual transformer. arXiv preprint arXiv:2012.12556"},{"key":"1130_CR39","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers. In: European conference on computer vision, pp. 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1130_CR40","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A, Weissenborn D, Zhai X, Unterthiner T, Dehghani M, Minderer M, Heigold G, Gelly S, et al. (2020) An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929"},{"key":"1130_CR41","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: hierarchical vision transformer using shifted windows. arXiv preprint arXiv:2103.14030","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1130_CR42","doi-asserted-by":"crossref","unstructured":"Li Y, Zhang S, Wang Z, Yang S, Yang W, Xia S.-T, Zhou E (2021) Tokenpose: learning keypoint tokens for human pose estimation. arXiv preprint arXiv:2104.03516","DOI":"10.1109\/ICCV48922.2021.01112"},{"key":"1130_CR43","doi-asserted-by":"crossref","unstructured":"Mao W, Ge Y, Shen C, Tian Z, Wang X, Wang Z (2021) Tfpose: direct human pose estimation with transformers. arXiv preprint arXiv:2103.15320","DOI":"10.1007\/978-3-031-20068-7_5"},{"key":"1130_CR44","doi-asserted-by":"crossref","unstructured":"Yang Y, Ramanan D (2011) Articulated pose estimation with flexible mixtures-of-parts. In: CVPR 2011, pp. 1385\u20131392. IEEE","DOI":"10.1109\/CVPR.2011.5995741"},{"key":"1130_CR45","doi-asserted-by":"crossref","unstructured":"Chen X, Yuille AL (2015) Parsing occluded people by flexible compositions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3945\u20133954","DOI":"10.1109\/CVPR.2015.7299020"},{"issue":"2","key":"1130_CR46","doi-asserted-by":"publisher","first-page":"927","DOI":"10.1109\/TIP.2016.2639441","volume":"26","author":"L Fu","year":"2016","unstructured":"Fu L, Zhang J, Huang K (2016) ORGM: occlusion relational graphical model for human pose estimation. IEEE Trans Image Process 26(2):927\u2013941","journal-title":"IEEE Trans Image Process"},{"key":"1130_CR47","unstructured":"Islam M.A, Jia S, Bruce N.D (2020) How much position information do convolutional neural networks encode? arXiv preprint arXiv:2001.08248"},{"key":"1130_CR48","doi-asserted-by":"crossref","unstructured":"Wu K, Peng H, Chen M, Fu J, Chao H (2021) Rethinking and improving relative position encoding for vision transformer. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 10033\u201310041","DOI":"10.1109\/ICCV48922.2021.00988"},{"key":"1130_CR49","doi-asserted-by":"crossref","unstructured":"Lin T.-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick C.L (2014) Microsoft coco: common objects in context. In: European conference on computer vision, pp. 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1130_CR50","doi-asserted-by":"crossref","unstructured":"Andriluka M, Pishchulin L, Gehler P, Schiele B (2014) 2d human pose estimation: new benchmark and state of the art analysis. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3686\u20133693","DOI":"10.1109\/CVPR.2014.471"},{"key":"1130_CR51","doi-asserted-by":"crossref","unstructured":"Chen Y, Wang Z, Peng Y, Zhang Z, Yu G, Sun J (2018) Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 7103\u20137112","DOI":"10.1109\/CVPR.2018.00742"},{"key":"1130_CR52","doi-asserted-by":"crossref","unstructured":"Li Z, Ye J, Song M, Huang Y, Pan Z (2021) Online knowledge distillation for efficient pose estimation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 11740\u201311750","DOI":"10.1109\/ICCV48922.2021.01153"},{"key":"1130_CR53","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"1130_CR54","doi-asserted-by":"crossref","unstructured":"Tang W, Yu P, Wu Y (2018) Deeply learned compositional models for human pose estimation. In: Proceedings of the European conference on computer vision (ECCV), pp. 190\u2013206","DOI":"10.1007\/978-3-030-01219-9_12"},{"key":"1130_CR55","unstructured":"Nibali A, He Z, Morganc S, Prendergast L (2018) Numerical coordinate regression with convolutional neural networks. arXiv preprint arXiv:1801.07372"}],"container-title":["Pattern Analysis and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10044-023-01130-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10044-023-01130-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10044-023-01130-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,4,15]],"date-time":"2023-04-15T04:38:44Z","timestamp":1681533524000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10044-023-01130-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,1]]},"references-count":55,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,5]]}},"alternative-id":["1130"],"URL":"https:\/\/doi.org\/10.1007\/s10044-023-01130-6","relation":{},"ISSN":["1433-7541","1433-755X"],"issn-type":[{"value":"1433-7541","type":"print"},{"value":"1433-755X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,2,1]]},"assertion":[{"value":"17 April 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 January 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 February 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}