{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T07:52:42Z","timestamp":1782978762219,"version":"3.54.5"},"reference-count":57,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2023,10,18]],"date-time":"2023-10-18T00:00:00Z","timestamp":1697587200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,10,18]],"date-time":"2023-10-18T00:00:00Z","timestamp":1697587200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2022YFB2503405"],"award-info":[{"award-number":["2022YFB2503405"]}]},{"DOI":"10.13039\/100007847","name":"Natural Science Foundation of Jilin Province","doi-asserted-by":"publisher","award":["20210101061JC"],"award-info":[{"award-number":["20210101061JC"]}],"id":[{"id":"10.13039\/100007847","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s11227-023-05691-5","type":"journal-article","created":{"date-parts":[[2023,10,18]],"date-time":"2023-10-18T09:01:38Z","timestamp":1697619698000},"page":"6169-6191","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["IDPNet: a light-weight network and its variants for human pose estimation"],"prefix":"10.1007","volume":"80","author":[{"given":"Huan","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jian","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rui","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,10,18]]},"reference":[{"key":"5691_CR1","doi-asserted-by":"crossref","unstructured":"Reddy ND, Vo M, Narasimhan SG (2018) CarFusion: combining point tracking and part detection for dynamic 3d reconstruction of vehicle. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 18\u201323","DOI":"10.1109\/CVPR.2018.00204"},{"key":"5691_CR2","doi-asserted-by":"crossref","unstructured":"Fernando T, Denman S, Sridharan S, Fookes C (2018) Tracking by prediction: a deep generative model for mutli-person localisation and tracking. In: Proceedings of the IEEE Winter Conference on Applications of Computer Vision (WACV), pp 12\u201315","DOI":"10.1109\/WACV.2018.00128"},{"key":"5691_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3412384","volume":"16","author":"MP Li","year":"2020","unstructured":"Li MP, Zhou Z, Liu X (2020) Cross refinement techniques for markerless human motion capture. ACM Trans Multimed Comput 16:1\u201318","journal-title":"ACM Trans Multimed Comput"},{"key":"5691_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3491228","volume":"18","author":"X Zhu","year":"2022","unstructured":"Zhu X, Zhu Y, Wang H, Wen H, Yan Y, Liu P (2022) Skeleton sequence and RGB frame based multi-modality feature fusion network for action recognition. ACM Trans Multimed Comput 18:1\u201324","journal-title":"ACM Trans Multimed Comput"},{"key":"5691_CR5","doi-asserted-by":"crossref","unstructured":"Krizhevsky A, Sutskever I, Hinton G (2017) Imagenet classification with deep convolutional neural networks. In: Proceedings of the Conference and Workshop on Neural Information Processing Systems, pp 4\u20139","DOI":"10.1145\/3065386"},{"key":"5691_CR6","unstructured":"Simonyan K, Zisserman A (2015) Very deep convolutional networks for large-scale image recognition. In: Proceedings of the International Conference on Learning Representations, pp 7\u20139"},{"key":"5691_CR7","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 7\u201312","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"5691_CR8","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"5691_CR9","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y Lecun","year":"1998","unstructured":"Lecun Y, Bottou L, Bengio Y, Haffner P (1998) Gradient-based learning applied to document recognition. IEEE 86:2278\u20132324","journal-title":"IEEE"},{"key":"5691_CR10","doi-asserted-by":"crossref","unstructured":"Newell A, Yang K, Deng J (2016) Stacked hourglass networks for human pose estimation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 10\u201316","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"5691_CR11","doi-asserted-by":"publisher","first-page":"2481","DOI":"10.1109\/TPAMI.2016.2644615","volume":"39","author":"V Badrinarayanan","year":"2017","unstructured":"Badrinarayanan V, Kendall A, Cipolla R (2017) SegNet: a deep convolutional encoder-decoder architecture for image segmentation. IEEE Trans Pattern Anal 39:2481\u20132495","journal-title":"IEEE Trans Pattern Anal"},{"key":"5691_CR12","doi-asserted-by":"crossref","unstructured":"Noh H, Hong S, Han B (2015) Learning Deconvolution network for semantic segmentation. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp 7\u201313","DOI":"10.1109\/ICCV.2015.178"},{"key":"5691_CR13","unstructured":"Weng W, Zhu X (2015) U-net: convolutional networks for biomedical image segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Boston, pp 7\u201312"},{"key":"5691_CR14","doi-asserted-by":"crossref","unstructured":"Xiao B, Wu H, Wei Y (2018) Simple baselines for human pose estimation and tracking. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 8\u201314","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"5691_CR15","doi-asserted-by":"crossref","unstructured":"Insafutdinov E, Pishchulin L, Andres B, Andriluka M, Schiele B (2016) DeeperCut: a deeper, stronger, and faster multi-person pose estimation model. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 10\u201316","DOI":"10.1007\/978-3-319-46466-4_3"},{"key":"5691_CR16","doi-asserted-by":"crossref","unstructured":"Yang W, Li S, Ouyang W Li H, Wang X (2017) Learning feature pyramids for human pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp 22\u201329","DOI":"10.1109\/ICCV.2017.144"},{"key":"5691_CR17","doi-asserted-by":"crossref","unstructured":"Chen Y, Wang Z, Peng Y, Zhang Z, Yu G, Sun J (2018) Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 18\u201323","DOI":"10.1109\/CVPR.2018.00742"},{"key":"5691_CR18","doi-asserted-by":"crossref","unstructured":"Sun K, Xiao B, Liu D, Wang J (2019) Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 5686\u20135696","DOI":"10.1109\/CVPR.2019.00584"},{"key":"5691_CR19","doi-asserted-by":"crossref","unstructured":"Dantone M, Gall J, Leistner C, VanGool L (2013) Human pose estimation using body parts dependent joint regressors. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 23\u201327","DOI":"10.1109\/CVPR.2013.391"},{"key":"5691_CR20","doi-asserted-by":"crossref","unstructured":"Gkioxari G, Hariharan B, Girshick R, Malik J (2014) Using k-poselets for detecting people and localizing their keypoints. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 23\u201328","DOI":"10.1109\/CVPR.2014.458"},{"key":"5691_CR21","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1023\/B:VISI.0000042934.15159.49","volume":"61","author":"PF Felzenszwalb","year":"2005","unstructured":"Felzenszwalb PF, Huttenlocher DP (2005) Pictorial structures for object recognition. Int J Comput Vis 61:55\u201379","journal-title":"Int J Comput Vis"},{"key":"5691_CR22","doi-asserted-by":"crossref","unstructured":"Andriluka M, Roth S, Schiele B (2009) Pictorial structures revisited: people detection and articulated pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 1014\u20131021","DOI":"10.1109\/CVPR.2009.5206754"},{"key":"5691_CR23","doi-asserted-by":"crossref","unstructured":"Yang Y, Ramanan D (2011) Articulated pose estimation with flexible mixtures-of-parts. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), Colorado Springs, pp 20\u201325","DOI":"10.1109\/CVPR.2011.5995741"},{"key":"5691_CR24","doi-asserted-by":"crossref","unstructured":"Pishchulin L, Andriluka M, Gehler P, Schiele B (2013) Poselet conditioned pictorial structures. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 23\u201327","DOI":"10.1109\/CVPR.2013.82"},{"key":"5691_CR25","doi-asserted-by":"crossref","unstructured":"Toshev A, Szegedy C (2014) DeepPose: human pose estimation via deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 23\u201328","DOI":"10.1109\/CVPR.2014.214"},{"key":"5691_CR26","doi-asserted-by":"crossref","unstructured":"Toshev A, Gkioxari G, Jaitly N (2016) Chained predictions using convolutional neural networks. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 10\u201316","DOI":"10.1007\/978-3-319-46493-0_44"},{"key":"5691_CR27","doi-asserted-by":"crossref","unstructured":"Tang W, Yu P, Wu Y (2018) Deeply learned compositional models for human pose estimation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 8\u201314","DOI":"10.1007\/978-3-030-01219-9_12"},{"key":"5691_CR28","doi-asserted-by":"crossref","unstructured":"Sun K, Lan C, Xing J, Zeng W, Liu D, Wang J (2017) Human pose estimation using global and local normalization. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp 22\u201329","DOI":"10.1109\/ICCV.2017.597"},{"key":"5691_CR29","unstructured":"Fan X, Zheng K, Lin Y, Wang S (2015) Combining local appearance and holistic view: Dual-Source Deep Neural Networks for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 7\u201312"},{"key":"5691_CR30","doi-asserted-by":"crossref","unstructured":"Peng X, Tang ZQ, Yang F, Feris R, Metaxas DN (2018) Jointly optimize data augmentation and network training: adversarial data augmentation in human pose estimation. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18\u201323","DOI":"10.1109\/CVPR.2018.00237"},{"key":"5691_CR31","doi-asserted-by":"crossref","unstructured":"Carreira J, Agrawal P, Fragkiadaki K, Malik J (2016) Human pose estimation with iterative error feedback. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 4733\u20134742","DOI":"10.1109\/CVPR.2016.512"},{"key":"5691_CR32","doi-asserted-by":"crossref","unstructured":"Chu X, Ouyang W, Li H, Wang X (2016) Structured feature learning for pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 4715\u20134723","DOI":"10.1109\/CVPR.2016.510"},{"key":"5691_CR33","doi-asserted-by":"crossref","unstructured":"Chu X, Yang W, Ouyang W, Ma C, Yuille A, Wang X (2017) Multi-context attention for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 21\u201326","DOI":"10.1109\/CVPR.2017.601"},{"key":"5691_CR34","doi-asserted-by":"crossref","unstructured":"Yang W, Ouyang W, Li H, Wang X (2016) End-to-end learning of deformable mixture of parts and deep convolutional neural networks for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 3073\u20133082","DOI":"10.1109\/CVPR.2016.335"},{"key":"5691_CR35","doi-asserted-by":"crossref","unstructured":"Zhou Y, Hu X, Zhang B (2018) Interlinked convolutional neural networks for face parsing. In: Proceedings of the International Symposium on Neural Networks, pp 25\u201328","DOI":"10.1007\/978-3-319-25393-0_56"},{"key":"5691_CR36","unstructured":"Saxena S, Verbeek J (2016) Convolutional neural fabrics. In: Proceedings of the Advances in Neural Information Processing Systems. Montreal, Canada, pp 4060\u20134068"},{"key":"5691_CR37","unstructured":"Huang G, Chen D, Li T, Wu F, Laurens V, Weinberger K (2017) Multi-scale dense convolutional networks for efficient prediction. CoRR (ACM)"},{"key":"5691_CR38","doi-asserted-by":"crossref","unstructured":"Bulat A, Tzimiropoulos G (2017) Binarized convolutional landmark localizers for human pose estimation and face alignment with limited resources. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV), pp 22\u201329","DOI":"10.1109\/ICCV.2017.400"},{"key":"5691_CR39","doi-asserted-by":"crossref","unstructured":"Zhang F, Zhu X, Ye M (2019) Fast human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 15\u201320","DOI":"10.1109\/CVPR.2019.00363"},{"key":"5691_CR40","unstructured":"Howard AG, Zhu M, Chen B, Kalenichenko D, Wang W, Weyand T, Andreetto M, Adam H (2017) MobileNets: efficient convolutional neural networks for mobile vision applications. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 21\u201326"},{"key":"5691_CR41","doi-asserted-by":"crossref","unstructured":"Sandler M, Howard A, Zhu M, Zhmoginov A, Wang W, Weyand T, Andreetto M, Adam H (2018) MobileNetV2: inverted residuals and linear bottlenecks. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18\u201323","DOI":"10.1109\/CVPR.2018.00474"},{"key":"5691_CR42","doi-asserted-by":"crossref","unstructured":"Zhang X, Zhou X, Lin M, Sun J (2018) ShuffleNet: an extremely efficient convolutional neural network for mobile devices. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 18\u201323","DOI":"10.1109\/CVPR.2018.00716"},{"key":"5691_CR43","doi-asserted-by":"crossref","unstructured":"Ma NN, Zhang XY, Zheng HT, Sun J (2018) ShuffleNet V2: practical guidelines for efficient CNN architecture design. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 8\u201314","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"5691_CR44","unstructured":"Iandola FN, Han S, Moskewicz MW, Ashraf K, Dally W, Keutzer K (2016) Squeezenet: Alexnet-level accuracy with 50\u00d7 fewer parameters and <0.5 mb model size. In: Proceedings of the International Conference on Learning Representations, pp 2\u20134"},{"key":"5691_CR45","doi-asserted-by":"crossref","unstructured":"Shen X, Yuan G, Niu W, Ma X, Wang Y (2021) Towards fast and accurate multi-person pose estimation on mobile devices. In: Proceedings of the International Joint Conferences on Artificial Intelligence Organization, pp 19\u201326","DOI":"10.24963\/ijcai.2021\/715"},{"key":"5691_CR46","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1109\/TPAMI.2019.2938758","volume":"43","author":"SH Gao","year":"2021","unstructured":"Gao SH, Cheng MM, Zhao K, Zhang XY, Yang M, Torr P (2021) Res2net: a new multi-scale backbone architecture. IEEE Trans Pattern Anal 43:652\u2013662","journal-title":"IEEE Trans Pattern Anal"},{"key":"5691_CR47","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3503464","volume":"18","author":"HB Dai","year":"2022","unstructured":"Dai HB, Shi HL, Liu W, Wang L, Liu Y, Mei T (2022) FasterPose: a faster simple baseline for human pose estimation. ACM Trans Multimed Comput 18:1\u201316","journal-title":"ACM Trans Multimed Comput"},{"key":"5691_CR48","doi-asserted-by":"crossref","unstructured":"Ding X, Guo Y, Ding G, Han J (2019) ACNet: strengthening the kernel skeletons for powerful CNN via asymmetric convolution blocks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp 1911\u20131920","DOI":"10.1109\/ICCV.2019.00200"},{"key":"5691_CR49","doi-asserted-by":"crossref","unstructured":"Cai YH, Wang ZC, Luo ZX, Yin B, Du A, Wang H, Zhang X, Zhou X, Zhou E, Sun J (2020) Learning delicate local representations for multi-person pose estimation. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 23\u201328","DOI":"10.1007\/978-3-030-58580-8_27"},{"key":"5691_CR50","doi-asserted-by":"crossref","unstructured":"Lin TY, Maire M, Belongie S, Girshick R, Bourdev L, Hays J, Perona P, Ramanan D, Zitnick CL, Dollar P (2014) Microsoft COCO: common objects in context. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 5\u201312","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"5691_CR51","unstructured":"Wang Z, Li W, Yin B et al (2018) Mscoco keypoints challenge. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 8\u201314"},{"key":"5691_CR52","unstructured":"Kingma D, Ba J (2015) Adam: a method for stochastic optimization. CoRR (ACM). 2015, abs\/1412.6980"},{"key":"5691_CR53","doi-asserted-by":"crossref","unstructured":"Andriluka M, Pishchulin L, Gehler P, Schiele B (2014) 2D human pose estimation: new benchmark and state of the art analysis. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 23\u201328","DOI":"10.1109\/CVPR.2014.471"},{"key":"5691_CR54","doi-asserted-by":"crossref","unstructured":"Yu C, Xiao B, Gao C, Yuan L, Zhang L, Sang N, Wang J (2021) Lite-HRNet: a lightweight high-resolution network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp 10435\u201310445","DOI":"10.1109\/CVPR46437.2021.01030"},{"key":"5691_CR55","doi-asserted-by":"crossref","unstructured":"Bulat A, Tzimiropoulos G (2016) Human pose estimation via convolutional part heatmap regression. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 10\u201316","DOI":"10.1007\/978-3-319-46478-7_44"},{"key":"5691_CR56","doi-asserted-by":"crossref","unstructured":"He K, Girshick R, Dollar P (2019) Rethinking ImageNet pre-training. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp 3059\u20133062","DOI":"10.1109\/ICCV.2019.00502"},{"key":"5691_CR57","doi-asserted-by":"crossref","unstructured":"Huang J, Zhu Z, Guo F, Huang G (2020) The devil is in the details: delving into unbiased data processing for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp 5699\u20135708","DOI":"10.1109\/CVPR42600.2020.00574"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05691-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-023-05691-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-023-05691-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,12]],"date-time":"2024-03-12T20:11:50Z","timestamp":1710274310000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-023-05691-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,18]]},"references-count":57,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["5691"],"URL":"https:\/\/doi.org\/10.1007\/s11227-023-05691-5","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,10,18]]},"assertion":[{"value":"28 September 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 October 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"The datasets used during the current study are available from  and .","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Data availability"}}]}}