{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T15:50:04Z","timestamp":1783007404380,"version":"3.54.5"},"reference-count":206,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2022,11,11]],"date-time":"2022-11-11T00:00:00Z","timestamp":1668124800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,11,11]],"date-time":"2022-11-11T00:00:00Z","timestamp":1668124800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2023,10]]},"DOI":"10.1007\/s00530-022-01019-0","type":"journal-article","created":{"date-parts":[[2022,11,11]],"date-time":"2022-11-11T19:03:06Z","timestamp":1668193386000},"page":"3115-3138","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":96,"title":["2D Human pose estimation: a survey"],"prefix":"10.1007","volume":"29","author":[{"given":"Haoming","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Runyang","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sifan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fengcheng","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhenguang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,11]]},"reference":[{"key":"1019_CR1","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., Schiele, B.: 2d human pose estimation: New benchmark and state of the art analysis. In: Proceedings of the IEEE Conference on computer Vision and Pattern Recognition, pp. 3686\u20133693 (2014)","DOI":"10.1109\/CVPR.2014.471"},{"key":"1019_CR2","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Iqbal, U., Insafutdinov, E., Pishchulin, L., Milan, A., Gall, J., Schiele, B.: Posetrack: A benchmark for human pose estimation and tracking. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 5167\u20135176 (2018)","DOI":"10.1109\/CVPR.2018.00542"},{"key":"1019_CR3","doi-asserted-by":"crossref","unstructured":"Artacho, B., Savakis, A.: Unipose: Unified human pose estimation in single images and videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7035\u20137044 (2020)","DOI":"10.1109\/CVPR42600.2020.00706"},{"key":"1019_CR4","doi-asserted-by":"crossref","unstructured":"Baccouche, M., Mamalet, F., Wolf, C., Garcia, C., Baskurt, A.: Sequential deep learning for human action recognition. In: International workshop on human behavior understanding, Springer, pp. 29\u201339 (2011)","DOI":"10.1007\/978-3-642-25446-8_4"},{"key":"1019_CR5","unstructured":"Bertasius, G., Feichtenhofer, C., Tran, D., Shi, J., Torresani, L.: Learning temporal pose estimation from sparsely-labeled videos. In: Advances in Neural Information Processing Systems, pp. 3027\u20133038 (2019)"},{"key":"1019_CR6","doi-asserted-by":"crossref","unstructured":"Bin, Y., Cao, X., Chen, X., Ge, Y., Tai, Y., Wang, C., Li, J., Huang, F., Gao, C., Sang, N.: Adversarial semantic data augmentation for human pose estimation. In: European Conference on Computer Vision, Springer, pp. 606\u2013622 (2020)","DOI":"10.1007\/978-3-030-58529-7_36"},{"key":"1019_CR7","doi-asserted-by":"crossref","unstructured":"Bourdev, L., Malik, J.: Poselets: Body part detectors trained using 3d human pose annotations. In: 2009 IEEE 12th International Conference on Computer Vision, IEEE, pp. 1365\u20131372 (2009)","DOI":"10.1109\/ICCV.2009.5459303"},{"key":"1019_CR8","doi-asserted-by":"crossref","unstructured":"Cai, Y., Wang, Z., Luo, Z., Yin, B., Du, A., Wang, H., Zhou, X., Zhou, E., Zhang, X., Sun, J.: Learning delicate local representations for multi-person pose estimation. arXiv preprint: arXiv:2003.04030 (2020)","DOI":"10.1007\/978-3-030-58580-8_27"},{"key":"1019_CR9","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S.E., Sheikh, Y.: Realtime multi-person 2d pose estimation using part affinity fields. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017a)","DOI":"10.1109\/CVPR.2017.143"},{"key":"1019_CR10","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S.E., Sheikh, Y.: Realtime multi-person 2d pose estimation using part affinity fields. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 7291\u20137299 (2017b)","DOI":"10.1109\/CVPR.2017.143"},{"key":"1019_CR11","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, Springer, pp. 213\u2013229 (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1019_CR12","doi-asserted-by":"crossref","unstructured":"Carreira, J., Agrawal, P., Fragkiadaki, K., Malik, J.: Human pose estimation with iterative error feedback. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 4733\u20134742 (2016)","DOI":"10.1109\/CVPR.2016.512"},{"key":"1019_CR13","doi-asserted-by":"crossref","unstructured":"Chan, C., Ginosar, S., Zhou, T., Efros, A.A.: Everybody dance now. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5933\u20135942 (2019)","DOI":"10.1109\/ICCV.2019.00603"},{"key":"1019_CR14","doi-asserted-by":"crossref","unstructured":"Chang, S., Yuan, L., Nie, X., Huang, Z., Zhou, Y., Chen, Y., Feng, J., Yan, S.: Towards accurate human pose estimation in videos of crowded scenes. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4630\u20134634 (2020)","DOI":"10.1145\/3394171.3416299"},{"key":"1019_CR15","doi-asserted-by":"crossref","unstructured":"Charles, J., Pfister, T., Magee, D., Hogg, D., Zisserman, A.: Personalizing human video pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3063\u20133072 (2016)","DOI":"10.1109\/CVPR.2016.334"},{"key":"1019_CR16","doi-asserted-by":"crossref","unstructured":"Chen, C.H., Ramanan, D.: 3d human pose estimation= 2d pose estimation+ matching. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7035\u20137043 (2017)","DOI":"10.1109\/CVPR.2017.610"},{"key":"1019_CR17","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, Z., Peng, Y., Zhang, Z., Yu, G., Sun, J.: Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 7103\u20137112 (2018)","DOI":"10.1109\/CVPR.2018.00742"},{"key":"1019_CR18","doi-asserted-by":"crossref","unstructured":"Chen, Y., Tian, Y., He, M.: Monocular human pose estimation: A survey of deep learning-based methods. Comput. Vis. Image Underst. 192, (2020)","DOI":"10.1016\/j.cviu.2019.102897"},{"key":"1019_CR19","doi-asserted-by":"crossref","unstructured":"Cheng, B., Xiao, B., Wang, J., Shi, H., Huang, T.S., Zhang, L.: Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5386\u20135395 (2020)","DOI":"10.1109\/CVPR42600.2020.00543"},{"key":"1019_CR20","doi-asserted-by":"crossref","unstructured":"Chu, X., Yang, W., Ouyang, W., Ma, C., Yuille, A.L., Wang, X.: Multi-context attention for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1831\u20131840 (2017)","DOI":"10.1109\/CVPR.2017.601"},{"issue":"5","key":"1019_CR21","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1109\/34.1000236","volume":"24","author":"D Comaniciu","year":"2002","unstructured":"Comaniciu, D., Meer, P.: Mean shift: A robust approach toward feature space analysis. IEEE Trans. Pattern Anal. Mach. Intell. 24(5), 603\u2013619 (2002)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1019_CR22","doi-asserted-by":"crossref","unstructured":"Datta, S., Sikka, K., Roy, A., Ahuja, K., Parikh, D., Divakaran, A.: Align2ground: Weakly supervised phrase grounding guided by image-caption alignment. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019)","DOI":"10.1109\/ICCV.2019.00269"},{"issue":"1","key":"1019_CR23","doi-asserted-by":"publisher","first-page":"269","DOI":"10.1007\/BF01386390","volume":"1","author":"EW Dijkstra","year":"1959","unstructured":"Dijkstra, E.W., et al.: A note on two problems in connexion with graphs. Numer. Math. 1(1), 269\u2013271 (1959)","journal-title":"Numer. Math."},{"key":"1019_CR24","unstructured":"Doering, A., Iqbal, U., Gall, J.: Joint flow: Temporal flow fields for multi person tracking. arXiv preprint: arXiv:1805.04596 (2018)"},{"key":"1019_CR25","doi-asserted-by":"crossref","unstructured":"Dong, J., Chen, Q., Shen, X., Yang, J., Yan, S.: Towards unified human parsing and pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 843\u2013850 (2014)","DOI":"10.1109\/CVPR.2014.113"},{"key":"1019_CR26","doi-asserted-by":"crossref","unstructured":"Dosovitskiy, A., Fischer, P., Ilg, E., Hausser, P., Hazirbas, C., Golkov, V., van\u00a0der Smagt, P., Cremers, D., Brox, T.: Flownet: Learning optical flow with convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV) (2015)","DOI":"10.1109\/ICCV.2015.316"},{"key":"1019_CR27","doi-asserted-by":"crossref","unstructured":"Duan, H., Lin, K.Y., Jin, S., Liu, W., Qian, C., Ouyang, W.: Trb: a novel triplet representation for understanding 2d human body. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9479\u20139488 (2019)","DOI":"10.1109\/ICCV.2019.00957"},{"key":"1019_CR28","doi-asserted-by":"crossref","unstructured":"Eichner, M., Ferrarim, V.: We are family: Joint pose estimation of multiple persons. In: European conference on computer vision, Springer, pp. 228\u2013242 (2010)","DOI":"10.1007\/978-3-642-15549-9_17"},{"issue":"11","key":"1019_CR29","doi-asserted-by":"publisher","first-page":"2282","DOI":"10.1109\/TPAMI.2012.85","volume":"34","author":"M Eichner","year":"2012","unstructured":"Eichner, M., Ferrari, V.: Human pose co-estimation and applications. IEEE Trans. Pattern Anal. Mach. Intell. 34(11), 2282\u20132288 (2012)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1019_CR30","doi-asserted-by":"crossref","unstructured":"Eichner, M., Ferrari, V., Zurich, S.: Better appearance models for pictorial structures. In: Bmvc, Citeseer, vol\u00a02, p\u00a05 (2009)","DOI":"10.5244\/C.23.3"},{"issue":"2","key":"1019_CR31","doi-asserted-by":"publisher","first-page":"190","DOI":"10.1007\/s11263-012-0524-9","volume":"99","author":"M Eichner","year":"2012","unstructured":"Eichner, M., Marin-Jimenez, M., Zisserman, A., Ferrari, V.: 2d articulated human pose estimation and retrieval in (almost) unconstrained still images. Int. J. Comput. Vis. 99(2), 190\u2013214 (2012)","journal-title":"Int. J. Comput. Vis."},{"issue":"2","key":"1019_CR32","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C.K., Winn, J., Zisserman, A.: The pascal visual object classes (voc) challenge. Int. J. Comput. Vis. 88(2), 303\u2013338 (2010)","journal-title":"Int. J. Comput. Vis."},{"key":"1019_CR33","unstructured":"Fan, X., Zheng, K., Lin, Y., Wang, S.: Combining local appearance and holistic view: Dual-source deep neural networks for human pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 1347\u20131355 (2015)"},{"key":"1019_CR34","doi-asserted-by":"crossref","unstructured":"Fang, H.S., Xie, S., Tai, Y.W., Lu, C.: Rmpe: Regional multi-person pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2334\u20132343 (2017)","DOI":"10.1109\/ICCV.2017.256"},{"issue":"1","key":"1019_CR35","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1023\/B:VISI.0000042934.15159.49","volume":"61","author":"PF Felzenszwalb","year":"2005","unstructured":"Felzenszwalb, P.F., Huttenlocher, D.P.: Pictorial structures for object recognition. Int. J. Comput. Vis. 61(1), 55\u201379 (2005)","journal-title":"Int. J. Comput. Vis."},{"key":"1019_CR36","doi-asserted-by":"crossref","unstructured":"Fieraru M, Khoreva A, Pishchulin L, Schiele B (2018) Learning to refine human pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition workshops, pp. 205\u2013214","DOI":"10.1109\/CVPRW.2018.00058"},{"key":"1019_CR37","unstructured":"Gao, Y., Chang, H.J., Demiris, Y.: User modelling for personalised dressing assistance by humanoid robots. In: 2015 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), IEEE, pp. 1840\u20131845 (2015)"},{"key":"1019_CR38","doi-asserted-by":"crossref","unstructured":"Gao, Y., Chang, H.J., Demiris, Y.: Iterative path optimisation for personalised dressing assistance using vision and force information. In: 2016 IEEE\/RSJ international conference on intelligent robots and systems (IROS), IEEE, pp. 4398\u20134403 (2016)","DOI":"10.1109\/IROS.2016.7759647"},{"key":"1019_CR39","doi-asserted-by":"crossref","unstructured":"Garau, N., Bisagno, N., Br\u00f3dka, P., Conci, N.: Deca: Deep viewpoint-equivariant human pose estimation using capsule autoencoders. arXiv preprint: arXiv:2108.08557 (2021)","DOI":"10.1109\/ICCV48922.2021.01147"},{"key":"1019_CR40","doi-asserted-by":"crossref","unstructured":"Geng, Z., Sun, K., Xiao, B., Zhang, Z., Wang, J.: Bottom-up human pose estimation via disentangled keypoint regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14676\u201314686 (2021)","DOI":"10.1109\/CVPR46437.2021.01444"},{"key":"1019_CR41","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Gkioxari, G., Torresani, L., Paluri, M., Tran, D.: Detect-and-track: Efficient pose estimation in videos. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 350\u2013359 (2018)","DOI":"10.1109\/CVPR.2018.00044"},{"key":"1019_CR42","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Toshev, A., Jaitly, N.: Chained predictions using convolutional neural networks. In: European Conference on Computer Vision, Springer, pp. 728\u2013743 (2016)","DOI":"10.1007\/978-3-319-46493-0_44"},{"key":"1019_CR43","doi-asserted-by":"crossref","unstructured":"Gong, K., Liang, X., Zhang, D., Shen, X., Lin, L.: Look into person: Self-supervised structure-sensitive learning and a new benchmark for human parsing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 932\u2013940 (2017)","DOI":"10.1109\/CVPR.2017.715"},{"key":"1019_CR44","doi-asserted-by":"crossref","unstructured":"Gong, K., Liang, X., Li, Y., Chen, Y., Yang, M., Lin, L.: Instance-level human parsing via part grouping network. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 770\u2013785 (2018)","DOI":"10.1007\/978-3-030-01225-0_47"},{"issue":"12","key":"1019_CR45","doi-asserted-by":"publisher","first-page":"1966","DOI":"10.3390\/s16121966","volume":"16","author":"W Gong","year":"2016","unstructured":"Gong, W., Zhang, X., Gonz\u00e0lez, J., Sobral, A., Bouwmans, T., Tu, C., Zahzah, E.H.: Human pose estimation from monocular images: a comprehensive survey. Sensors 16(12), 1966 (2016)","journal-title":"Sensors"},{"key":"1019_CR46","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., Bengio, Y.: Generative adversarial nets. Adv. Neural Inf. Process Syst. 27 (2014)"},{"key":"1019_CR47","doi-asserted-by":"crossref","unstructured":"Guo, H., Tang, T., Luo, G., Chen, R., Lu, Y., Wen, L.: Multi-domain pose network for multi-person pose estimation and tracking. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 0\u20130 (2018)","DOI":"10.1007\/978-3-030-11012-3_17"},{"issue":"4","key":"1019_CR48","doi-asserted-by":"publisher","first-page":"802","DOI":"10.1109\/LCOMM.2019.2961890","volume":"24","author":"L Guo","year":"2019","unstructured":"Guo, L., Lu, Z., Wen, X., Zhou, S., Han, Z.: From signal to image: Capturing fine-grained human poses with commodity wi-fi. IEEE Commun. Lett. 24(4), 802\u2013806 (2019)","journal-title":"IEEE Commun. Lett."},{"key":"1019_CR49","doi-asserted-by":"crossref","unstructured":"Guo, Y., Cheng, Z., Nie, L., Liu, Y., Wang, Y., Kankanhalli, M.S.: Quantifying and alleviating the language prior problem in visual question answering. In: SIGIR, ACM, pp. 75\u201384 (2019b)","DOI":"10.1145\/3331184.3331186"},{"key":"1019_CR50","doi-asserted-by":"crossref","unstructured":"Guo, Y., Nie, L., Cheng, Z., Ji, F., Zhang, J., Bimbo, A.D.: Adavqa: Overcoming language priors with adapted margin cosine loss. In: IJCAI, ijcai.org, pp. 708\u2013714 (2021a)","DOI":"10.24963\/ijcai.2021\/98"},{"key":"1019_CR51","doi-asserted-by":"crossref","unstructured":"Guo, Y., Nie, L., Cheng, Z., Ji, F., Zhang, J., Del\u00a0Bimbo, A.: Adavqa: Overcoming language priors with adapted margin cosine loss. arXiv preprint: arXiv:2105.01993 (2021b)","DOI":"10.24963\/ijcai.2021\/98"},{"key":"1019_CR52","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1019_CR53","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Dollar, P., Girshick, R.: Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV) (2017a)","DOI":"10.1109\/ICCV.2017.322"},{"key":"1019_CR54","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask r-cnn. In: Proceedings of the IEEE international conference on computer vision, pp. 2961\u20132969 (2017b)","DOI":"10.1109\/ICCV.2017.322"},{"key":"1019_CR55","unstructured":"Hidalgo, G., Raaj, Y., Idrees, H., Xiang, D., Joo, H., Simon, T., Sheikh, Y.: Single-network whole-body pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6982\u20136991 (2019)"},{"key":"1019_CR56","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. arXiv preprint: arXiv:1503.02531 (2015)"},{"issue":"5","key":"1019_CR57","doi-asserted-by":"publisher","first-page":"538","DOI":"10.1109\/JSTSP.2012.2196975","volume":"6","author":"MB Holte","year":"2012","unstructured":"Holte, M.B., Tran, C., Trivedi, M.M., Moeslund, T.B.: Human pose estimation and activity recognition from multi-view videos: Comparative explorations of recent developments. IEEE J. Select. Topic Signal Proces 6(5), 538\u2013552 (2012)","journal-title":"IEEE J. Select. Topic Signal Proces"},{"key":"1019_CR58","doi-asserted-by":"crossref","unstructured":"Huang, J., Zhu, Z., Guo, F., Huang, G.: The devil is in the details: Delving into unbiased data processing for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5700\u20135709 (2020a)","DOI":"10.1109\/CVPR42600.2020.00574"},{"key":"1019_CR59","unstructured":"Huang, J., Zhu, Z., Huang, G., Du, D.: Aid: Pushing the performance boundary of human pose estimation with information dropping augmentation. arXiv preprint: arXiv:2008.07139"},{"key":"1019_CR60","doi-asserted-by":"crossref","unstructured":"Huang, S., Gong, M., Tao, D.: A coarse-fine network for keypoint localization. In: Proceedings of the IEEE international conference on computer vision, pp. 3028\u20133037 (2017)","DOI":"10.1109\/ICCV.2017.329"},{"key":"1019_CR61","doi-asserted-by":"crossref","unstructured":"Ilg, E., Mayer, N., Saikia, T., Keuper, M., Dosovitskiy, A., Brox, T.: Flownet 2.0: Evolution of optical flow estimation with deep networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.179"},{"key":"1019_CR62","doi-asserted-by":"crossref","unstructured":"Insafutdinov, E., Pishchulin, L., Andres, B., Andriluka, M., Schiele, B.: Deepercut: A deeper, stronger, and faster multi-person pose estimation model. In: European Conference on Computer Vision, Springer, pp. 34\u201350 (2016)","DOI":"10.1007\/978-3-319-46466-4_3"},{"key":"1019_CR63","doi-asserted-by":"crossref","unstructured":"Iqbal, U., Garbade, M., Gall, J.: Pose for action-action for pose. In: 2017 12th IEEE International Conference on Automatic Face & Gesture Recognition (FG 2017), IEEE, pp. 438\u2013445 (2017)","DOI":"10.1109\/FG.2017.61"},{"key":"1019_CR64","first-page":"2017","volume":"28","author":"M Jaderberg","year":"2015","unstructured":"Jaderberg, M., Simonyan, K., Zisserman, A., et al.: Spatial transformer networks. Adv. Neural Inf. Process System 28, 2017\u20132025 (2015)","journal-title":"Adv. Neural Inf. Process System"},{"key":"1019_CR65","doi-asserted-by":"crossref","unstructured":"Jhuang, H., Gall, J., Zuffi, S., Schmid, C., Black, M.J.: Towards understanding action recognition. In: Proceedings of the IEEE international conference on computer vision, pp. 3192\u20133199 (2013)","DOI":"10.1109\/ICCV.2013.396"},{"issue":"1","key":"1019_CR66","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2012","unstructured":"Ji, S., Xu, W., Yang, M., Yu, K.: 3d convolutional neural networks for human action recognition. IEEE Trans. Pattern Anal. Mach. Intell. 35(1), 221\u2013231 (2012)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"1","key":"1019_CR67","first-page":"13","volume":"40","author":"X Ji","year":"2009","unstructured":"Ji, X., Liu, H.: Advances in view-invariant human motion analysis: a review. IEEE Trans. Syst. Man Cybern. 40(1), 13\u201324 (2009)","journal-title":"IEEE Trans. Syst. Man Cybern."},{"key":"1019_CR68","doi-asserted-by":"crossref","unstructured":"Jiang, C., Huang, K., Zhang, S., Wang, X., Xiao, J.: Pay attention selectively and comprehensively: Pyramid gating network for human pose estimation without pre-training. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 2364\u20132371 (2020)","DOI":"10.1145\/3394171.3414041"},{"key":"1019_CR69","doi-asserted-by":"crossref","unstructured":"Jin, S., Liu, W., Ouyang, W., Qian, C.: Multi-person articulated tracking with spatial and temporal embeddings. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5664\u20135673 (2019)","DOI":"10.1109\/CVPR.2019.00581"},{"key":"1019_CR70","doi-asserted-by":"crossref","unstructured":"Jin, S., Liu, W., Xie, E., Wang, W., Qian, C., Ouyang, W., Luo, P.: Differentiable hierarchical graph grouping for multi-person pose estimation. In: European Conference on Computer Vision, Springer, pp. 718\u2013734 (2020)","DOI":"10.1007\/978-3-030-58571-6_42"},{"key":"1019_CR71","doi-asserted-by":"crossref","unstructured":"Johnson, S., Everingham, M.: Clustered pose and nonlinear appearance models for human pose estimation. In: bmvc, Citeseer, vol\u00a02, p\u00a05 (2010)","DOI":"10.5244\/C.24.12"},{"key":"1019_CR72","doi-asserted-by":"crossref","unstructured":"Johnson, S., Everingham, M.: Learning effective human pose estimation from inaccurate annotation. In: CVPR 2011, IEEE, pp. 1465\u20131472 (2011)","DOI":"10.1109\/CVPR.2011.5995318"},{"key":"1019_CR73","doi-asserted-by":"crossref","unstructured":"Ju, S.X., Black, M.J., Yacoob, Y.: Cardboard people: A parameterized model of articulated image motion. In: Proceedings of the Second International Conference on Automatic Face and Gesture Recognition, IEEE, pp. 38\u201344 (1996)","DOI":"10.1109\/AFGR.1996.557241"},{"key":"1019_CR74","doi-asserted-by":"crossref","unstructured":"Kappel, M., Golyanik, V., Elgharib, M., Henningson, J.O., Seidel, H.P., Castillo, S., Theobalt, C., Magnor, M.: High-fidelity neural human motion transfer from monocular video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1541\u20131550 (2021)","DOI":"10.1109\/CVPR46437.2021.00159"},{"key":"1019_CR75","doi-asserted-by":"crossref","unstructured":"Ke, L., Chang, M.C., Qi, H., Lyu, S.: Multi-scale structure-aware network for human pose estimation. In: Proceedings of the european conference on computer vision (ECCV), pp. 713\u2013728 (2018)","DOI":"10.1007\/978-3-030-01216-8_44"},{"key":"1019_CR76","unstructured":"Kipf, T.N., Welling, M.: Semi-supervised classification with graph convolutional networks. arXiv preprint: arXiv:1609.02907 (2016)"},{"key":"1019_CR77","doi-asserted-by":"crossref","unstructured":"Kocabas, M., Karagoz, S., Akbas, E.: Multiposenet: Fast multi-person pose estimation using pose residual network. In: Proceedings of the European conference on computer vision (ECCV), pp. 417\u2013433 (2018)","DOI":"10.1007\/978-3-030-01252-6_26"},{"key":"1019_CR78","doi-asserted-by":"crossref","unstructured":"Kreiss, S., Bertoni, L., Alahi, A.: Pifpaf: Composite fields for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11977\u201311986 (2019)","DOI":"10.1109\/CVPR.2019.01225"},{"key":"1019_CR79","first-page":"1097","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Adv. Neural Inf. Process System 25, 1097\u20131105 (2012)","journal-title":"Adv. Neural Inf. Process System"},{"key":"1019_CR80","doi-asserted-by":"crossref","unstructured":"Ladicky, L., Torr, P.H., Zisserman, A.: Human pose estimation using a joint pixel-wise and part-wise formulation. In: proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3578\u20133585 (2013)","DOI":"10.1109\/CVPR.2013.459"},{"issue":"11","key":"1019_CR81","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86(11), 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"1019_CR82","doi-asserted-by":"crossref","unstructured":"Li, C., Lee, G.H.: From synthetic to real: Unsupervised domain adaptation for animal pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1482\u20131491 (2021)","DOI":"10.1109\/CVPR46437.2021.00153"},{"key":"1019_CR83","doi-asserted-by":"crossref","unstructured":"Li, G., Zhang, Z., Yang, H., Pan, J., Chen, D., Zhang, J.: Capturing human pose using mmwave radar. In: 2020 IEEE International Conference on Pervasive Computing and Communications Workshops (PerCom Workshops), IEEE, pp. 1\u20136 (2020a)","DOI":"10.1109\/PerComWorkshops48775.2020.9156151"},{"key":"1019_CR84","doi-asserted-by":"crossref","unstructured":"Li, J., Wang, C., Zhu, H., Mao, Y., Fang, H.S., Lu, C.: Crowdpose: Efficient crowded scenes pose estimation and a new benchmark. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10863\u201310872 (2019)","DOI":"10.1109\/CVPR.2019.01112"},{"key":"1019_CR85","doi-asserted-by":"crossref","unstructured":"Li, J., Su, W., Wang, Z.: Simple pose: Rethinking and improving a bottom-up approach for multi-person pose estimation. In: Proceedings of the AAAI conference on artificial intelligence, vol\u00a034, pp. 11354\u201311361 (2020b)","DOI":"10.1609\/aaai.v34i07.6797"},{"key":"1019_CR86","doi-asserted-by":"crossref","unstructured":"Li, J., Bian, S., Zeng, A., Wang, C., Pang, B., Liu, W., Lu, C.: Human pose regression with residual log-likelihood estimation. arXiv preprint arXiv:2107.11291 (2021a)","DOI":"10.1109\/ICCV48922.2021.01084"},{"key":"1019_CR87","doi-asserted-by":"crossref","unstructured":"Li, K., Wang, S., Zhang, X., Xu, Y., Xu, W., Tu, Z.: Pose recognition with cascade transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1944\u20131953 (2021b)","DOI":"10.1109\/CVPR46437.2021.00198"},{"key":"1019_CR88","doi-asserted-by":"crossref","unstructured":"Li, L.J., Fei-Fei, L.: What, where and who? classifying events by scene and object recognition. In: 2007 IEEE 11th international conference on computer vision, IEEE, pp. 1\u20138 (2007)","DOI":"10.1109\/ICCV.2007.4408872"},{"key":"1019_CR89","doi-asserted-by":"crossref","unstructured":"Li, S., Liu, Z.Q., Chan, A.B.: Heterogeneous multi-task learning for human pose estimation with deep convolutional neural network. In: Proceedings of the IEEE conference on computer vision and pattern recognition workshops, pp. 482\u2013489 (2014)","DOI":"10.1109\/CVPRW.2014.78"},{"key":"1019_CR90","doi-asserted-by":"crossref","unstructured":"Li, Y., Yang, X., Shang, X., Chua, T.S.: Interventional video relation detection. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 4091\u20134099 (2021c)","DOI":"10.1145\/3474085.3475540"},{"key":"1019_CR91","doi-asserted-by":"crossref","unstructured":"Li, Z., Ye, J., Song, M., Huang, Y., Pan, Z.: Online knowledge distillation for efficient pose estimation. arXiv preprint arXiv:2108.02092 (2021d)","DOI":"10.1109\/ICCV48922.2021.01153"},{"issue":"4","key":"1019_CR92","doi-asserted-by":"publisher","first-page":"871","DOI":"10.1109\/TPAMI.2018.2820063","volume":"41","author":"X Liang","year":"2018","unstructured":"Liang, X., Gong, K., Shen, X., Lin, L.: Look into person: Joint body parsing & pose estimation network and a new benchmark. IEEE Trans. pattern Anal. Mach. Intell. 41(4), 871\u2013885 (2018)","journal-title":"IEEE Trans. pattern Anal. Mach. Intell."},{"key":"1019_CR93","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: Common objects in context. In: European conference on computer vision, Springer, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1019_CR94","unstructured":"Lin, W., Liu, H., Liu, S., Li, Y., Qian, R., Wang, T., Xu, N., Xiong, H., Qi, G.J., Sebe, N.: Human in events: A large-scale benchmark for human-centric video analysis in complex events. arXiv preprint arXiv:2005.04490 (2020)"},{"key":"1019_CR95","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.Y., Berg, A.C.: Ssd: Single shot multibox detector. In: European conference on computer vision, Springer, pp. 21\u201337 (2016)","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"1019_CR96","doi-asserted-by":"crossref","unstructured":"Liu, W., Chen, J., Li, C., Qian, C., Chu, X., Hu, X.: A cascaded inception of inception network with attention modulated feature fusion for human pose estimation. In: Thirty-Second AAAI Conference on Artificial Intelligence (2018)","DOI":"10.1609\/aaai.v32i1.12334"},{"key":"1019_CR97","doi-asserted-by":"publisher","first-page":"10","DOI":"10.1016\/j.jvcir.2015.06.013","volume":"32","author":"Z Liu","year":"2015","unstructured":"Liu, Z., Zhu, J., Bu, J., Chen, C.: A survey of human pose estimation: the body parts parsing based methods. J. Vis. Commun. Image Represent 32, 10\u201319 (2015)","journal-title":"J. Vis. Commun. Image Represent"},{"key":"1019_CR98","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wu, S., Jin, S., Liu, Q., Lu, S., Zimmermann, R., Cheng, L.: Towards natural and accurate future motion prediction of humans and animals. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10004\u201310012 (2019)","DOI":"10.1109\/CVPR.2019.01024"},{"key":"1019_CR99","doi-asserted-by":"crossref","unstructured":"Liu, Z., Chen, H., Feng, R., Wu, S., Ji, S., Yang, B., Wang, X.: Deep dual consecutive network for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 525\u2013534 (2021a)","DOI":"10.1109\/CVPR46437.2021.00059"},{"key":"1019_CR100","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lyu, K., Wu, S., Chen, H., Hao, Y., Ji, S.: Aggregated multi-gans for controlled 3d human motion prediction. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol\u00a035, pp. 2225\u20132232 (2021b)","DOI":"10.1609\/aaai.v35i3.16321"},{"key":"1019_CR101","doi-asserted-by":"crossref","unstructured":"Liu, Z., Qian, P., Wang, X., Zhuang, Y., Qiu, L., Wang, X.: Combining graph neural networks with expert knowledge for smart contract vulnerability detection. IEEE Transactions on Knowledge and Data Engineering (2021c)","DOI":"10.1109\/TKDE.2021.3095196"},{"key":"1019_CR102","doi-asserted-by":"crossref","unstructured":"Liu, Z., Su, P., Wu, S., Shen, X., Chen, H., Hao, Y., Wang, M.: Motion prediction using trajectory cues. IEEE International Conference on Computer Vision (2021d)","DOI":"10.1109\/ICCV48922.2021.01305"},{"key":"1019_CR103","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.Y., Feichtenhofer, C., Darrell, T., Xie, S.: A convnet for the 2020s. arXiv preprint arXiv:2201.03545 (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"1019_CR104","doi-asserted-by":"crossref","unstructured":"Luo, Y., Ren, J., Wang, Z., Sun, W., Pan, J., Liu, J., Pang, J., Lin, L.: Lstm pose machines. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 5207\u20135215 (2018a)","DOI":"10.1109\/CVPR.2018.00546"},{"issue":"1","key":"1019_CR105","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1109\/TIP.2018.2865666","volume":"28","author":"Y Luo","year":"2018","unstructured":"Luo, Y., Xu, Z., Liu, P., Du, Y., Guo, J.M.: Multi-person pose estimation via multi-layer fractal network and joints kinship pattern. IEEE Trans. Image Process 28(1), 142\u2013155 (2018)","journal-title":"IEEE Trans. Image Process"},{"key":"1019_CR106","doi-asserted-by":"crossref","unstructured":"Luo, Z., Wang, Z., Huang, Y., Wang, L., Tan, T., Zhou, E.: Rethinking the heatmap regression for bottom-up human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13264\u201313273 (2021)","DOI":"10.1109\/CVPR46437.2021.01306"},{"key":"1019_CR107","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1016\/j.cag.2019.09.002","volume":"85","author":"DC Luvizon","year":"2019","unstructured":"Luvizon, D.C., Tabia, H., Picard, D.: Human pose regression by combining indirect part detection and contextual information. Comput. Graphic 85, 15\u201322 (2019)","journal-title":"Comput. Graphic"},{"key":"1019_CR108","doi-asserted-by":"crossref","unstructured":"Ma, N., Zhang, X., Zheng, H.T., Sun, J.: Shufflenet v2: Practical guidelines for efficient cnn architecture design. In: Proceedings of the European conference on computer vision (ECCV), pp. 116\u2013131 (2018)","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"1019_CR109","doi-asserted-by":"crossref","unstructured":"Mao, W., Tian, Z., Wang, X., Shen, C.: Fcpose: Fully convolutional multi-person pose estimation with dynamic instance-aware convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9034\u20139043 (2021)","DOI":"10.1109\/CVPR46437.2021.00892"},{"issue":"3","key":"1019_CR110","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1007\/s11263-013-0655-7","volume":"106","author":"MJ Marin-Jimenez","year":"2014","unstructured":"Marin-Jimenez, M.J., Zisserman, A., Eichner, M., Ferrari, V.: Detecting people looking at each other in videos. Int. J. Comput Vis. 106(3), 282\u2013296 (2014)","journal-title":"Int. J. Comput Vis."},{"key":"1019_CR111","doi-asserted-by":"crossref","unstructured":"Martinez, J., Hossain, R., Romero, J., Little, J.J.: A simple yet effective baseline for 3d human pose estimation. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2640\u20132649 (2017)","DOI":"10.1109\/ICCV.2017.288"},{"issue":"4","key":"1019_CR112","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073596","volume":"36","author":"D Mehta","year":"2017","unstructured":"Mehta, D., Sridhar, S., Sotnychenko, O., Rhodin, H., Shafiei, M., Seidel, H.P., Xu, W., Casas, D., Theobalt, C.: Vnect: Real-time 3d human pose estimation with a single rgb camera. ACM Trans Graphic (TOG) 36(4), 1\u201314 (2017)","journal-title":"ACM Trans Graphic (TOG)"},{"key":"1019_CR113","doi-asserted-by":"crossref","unstructured":"Mirzadeh, S.I., Farajtabar, M., Li, A., Levine, N., Matsukawa, A., Ghasemzadeh, H.: Improved knowledge distillation via teacher assistant. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol\u00a034, pp. 5191\u20135198 (2020)","DOI":"10.1609\/aaai.v34i04.5963"},{"issue":"3","key":"1019_CR114","doi-asserted-by":"publisher","first-page":"231","DOI":"10.1006\/cviu.2000.0897","volume":"81","author":"TB Moeslund","year":"2001","unstructured":"Moeslund, T.B., Granum, E.: A survey of computer vision-based human motion capture. Comput. Vis. Image Understand 81(3), 231\u2013268 (2001)","journal-title":"Comput. Vis. Image Understand"},{"issue":"2\u20133","key":"1019_CR115","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1016\/j.cviu.2006.08.002","volume":"104","author":"TB Moeslund","year":"2006","unstructured":"Moeslund, T.B., Hilton, A., Kr\u00fcger, V.: A survey of advances in vision-based human motion capture and analysis. Comput. Vis. image Understand. 104(2\u20133), 90\u2013126 (2006)","journal-title":"Comput. Vis. image Understand."},{"key":"1019_CR116","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-85729-997-0","volume-title":"Visual analysis of humans","author":"TB Moeslund","year":"2011","unstructured":"Moeslund, T.B., Hilton, A., Kr\u00fcger, V., Sigal, L.: Visual analysis of humans. Springer, NY (2011)"},{"key":"1019_CR117","doi-asserted-by":"crossref","unstructured":"Mogadala, A., Kalimuthu, M., Klakow, D.: Trends in integration of vision and language research: A survey of tasks, datasets, and methods. J. Artif. Intell. Res. (2021)","DOI":"10.1613\/jair.1.11688"},{"key":"1019_CR118","doi-asserted-by":"crossref","unstructured":"Moon, G., Chang, J.Y., Lee, K.M.: Posefix: Model-agnostic general human pose refinement network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7773\u20137781(2019)","DOI":"10.1109\/CVPR.2019.00796"},{"key":"1019_CR119","doi-asserted-by":"publisher","first-page":"133330","DOI":"10.1109\/ACCESS.2020.3010248","volume":"8","author":"TL Munea","year":"2020","unstructured":"Munea, T.L., Jembre, Y.Z., Weldegebriel, H.T., Chen, L., Huang, C., Yang, C.: The progress of human pose estimation: a survey and taxonomy of models applied in 2d human pose estimation. IEEE Access 8, 133330\u2013133348 (2020)","journal-title":"IEEE Access"},{"key":"1019_CR120","unstructured":"Naksuk, N., Lee, C.G., Rietdyk, S.: Whole-body human-to-humanoid motion transfer. In: 5th IEEE-RAS International Conference on Humanoid Robots, 2005., IEEE, pp. 104\u2013109 (2005)"},{"key":"1019_CR121","unstructured":"Newell, A., Huang, Z., Deng, J.: Associative embedding: End-to-end learning for joint detection and grouping. arXiv preprint arXiv:1611.05424 (2016a)"},{"key":"1019_CR122","doi-asserted-by":"crossref","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation. In: European conference on computer vision, Springer, pp. 483\u2013499 (2016b)","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"1019_CR123","doi-asserted-by":"crossref","unstructured":"Nie, X., Feng, J., Xing, J., Yan, S.: Pose partition networks for multi-person pose estimation. In: Proceedings of the european conference on computer vision (eccv), pp. 684\u2013699 (2018a)","DOI":"10.1007\/978-3-030-01228-1_42"},{"key":"1019_CR124","doi-asserted-by":"crossref","unstructured":"Nie, X., Feng, J., Zuo, Y., Yan, S.: Human pose estimation with parsing induced learner. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2100\u20132108 (2018b)","DOI":"10.1109\/CVPR.2018.00224"},{"key":"1019_CR125","doi-asserted-by":"crossref","unstructured":"Nie, X., Feng, J., Zhang, J., Yan, S.: Single-stage multi-person pose machines. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6951\u20136960 (2019a)","DOI":"10.1109\/ICCV.2019.00705"},{"key":"1019_CR126","doi-asserted-by":"crossref","unstructured":"Nie X, Li Y, Luo L, Zhang N, Feng J (2019b) Dynamic kernel distillation for efficient pose estimation in videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6942\u20136950","DOI":"10.1109\/ICCV.2019.00704"},{"key":"1019_CR127","doi-asserted-by":"crossref","unstructured":"Nie, X., Feng, J., Zhang, J., Yan, S.: Single-stage multi-person pose machines. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV) (2020)","DOI":"10.1109\/ICCV.2019.00705"},{"key":"1019_CR128","doi-asserted-by":"crossref","unstructured":"Papandreou, G., Zhu, T., Kanazawa, N., Toshev, A., Tompson, J., Bregler, C., Murphy, K.: Towards accurate multi-person pose estimation in the wild. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4903\u20134911(2017)","DOI":"10.1109\/CVPR.2017.395"},{"key":"1019_CR129","doi-asserted-by":"crossref","unstructured":"Papandreou, G., Zhu, T., Chen, L.C., Gidaris, S., Tompson, J., Murphy, K.: Personlab: Person pose estimation and instance segmentation with a bottom-up, part-based, geometric embedding model. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 269\u2013286 (2018)","DOI":"10.1007\/978-3-030-01264-9_17"},{"key":"1019_CR130","doi-asserted-by":"crossref","unstructured":"Peng, X., Tang, Z., Yang, F., Feris, R.S., Metaxas, D.: Jointly optimize data augmentation and network training: Adversarial data augmentation in human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2226\u20132234(2018)","DOI":"10.1109\/CVPR.2018.00237"},{"key":"1019_CR131","doi-asserted-by":"crossref","unstructured":"Pfister, T., Charles, J., Zisserman, A.: Flowing convnets for human pose estimation in videos. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1913\u20131921 (2015)","DOI":"10.1109\/ICCV.2015.222"},{"key":"1019_CR132","doi-asserted-by":"crossref","unstructured":"Pishchulin, L., Andriluka, M., Gehler, P., Schiele, B.: Poselet conditioned pictorial structures. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 588\u2013595 (2013)","DOI":"10.1109\/CVPR.2013.82"},{"key":"1019_CR133","doi-asserted-by":"crossref","unstructured":"Pishchulin, L., Insafutdinov, E., Tang, S., Andres, B., Andriluka, M., Gehler, P.V., Schiele, B.: Deepcut: Joint subset partition and labeling for multi person pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 4929\u20134937 (2016)","DOI":"10.1109\/CVPR.2016.533"},{"issue":"1\u20132","key":"1019_CR134","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1016\/j.cviu.2006.10.016","volume":"108","author":"R Poppe","year":"2007","unstructured":"Poppe, R.: Vision-based human motion analysis: An overview. Comput. Vis. Image Understand 108(1\u20132), 4\u201318 (2007)","journal-title":"Comput. Vis. Image Understand"},{"key":"1019_CR135","doi-asserted-by":"crossref","unstructured":"Qiu, L., Zhang, X., Li, Y., Li, G., Wu, X., Xiong, Z., Han, X., Cui, S.: Peeking into occluded joints: A novel framework for crowd pose estimation. In: European Conference on Computer Vision, Springer, pp. 488\u2013504 (2020)","DOI":"10.1007\/978-3-030-58529-7_29"},{"key":"1019_CR136","doi-asserted-by":"crossref","unstructured":"Raaj, Y., Idrees, H., Hidalgo, G., Sheikh, Y.: Efficient online multi-person 2d pose tracking with recurrent spatio-temporal affinity fields. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4620\u20134628 (2019)","DOI":"10.1109\/CVPR.2019.00475"},{"key":"1019_CR137","doi-asserted-by":"crossref","unstructured":"Ramakrishna, V., Munoz, D., Hebert, M., Bagnell, J.A., Sheikh, Y.: Pose machines: Articulated pose estimation via inference machines. In: European Conference on Computer Vision, Springer, pp. 33\u201347 (2014)","DOI":"10.1007\/978-3-319-10605-2_3"},{"key":"1019_CR138","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: Towards real-time object detection with region proposal networks. Adv. Neural Inf. Process Syst. 28, 91\u201399 (2015)","journal-title":"Adv. Neural Inf. Process Syst."},{"key":"1019_CR139","unstructured":"Romero, A., Ballas, N., Kahou, S.E., Chassang, A., Gatta, C., Bengio, Y.: Fitnets: Hints for thin deep nets. arXiv preprint arXiv:1412.6550 (2014)"},{"key":"1019_CR140","doi-asserted-by":"crossref","unstructured":"Ruan, T., Liu, T., Huang, Z., Wei, Y., Wei, S., Zhao, Y.: Devil in the details: Towards accurate single and multiple human parsing. In: Proc. AAAI Conf. Artif. Intell. 33, 4814\u20134821 (2019)","DOI":"10.1609\/aaai.v33i01.33014814"},{"key":"1019_CR141","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.C.: Mobilenetv2: Inverted residuals and linear bottlenecks. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"1019_CR142","doi-asserted-by":"crossref","unstructured":"Sapp, B., Taskar, B.: Modec: Multimodal decomposable models for human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3674\u20133681 (2013)","DOI":"10.1109\/CVPR.2013.471"},{"key":"1019_CR143","doi-asserted-by":"crossref","unstructured":"Sapp, B., Toshev, A., Taskar, B.: Cascaded models for articulated pose estimation. In: European conference on computer vision, Springer, pp. 406\u2013420 (2010)","DOI":"10.1007\/978-3-642-15552-9_30"},{"key":"1019_CR144","doi-asserted-by":"crossref","unstructured":"Sapp, B., Weiss, D., Taskar, B.: Parsing human motion with stretchable models. In: CVPR 2011, IEEE, pp. 1281\u20131288 (2011)","DOI":"10.1109\/CVPR.2011.5995607"},{"key":"1019_CR145","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.cviu.2016.09.002","volume":"152","author":"N Sarafianos","year":"2016","unstructured":"Sarafianos, N., Boteanu, B., Ionescu, B., Kakadiaris, I.A.: 3d human pose estimation: A review of the literature and analysis of covariates. Comput. Vis. Image Understand 152, 1\u201320 (2016)","journal-title":"Comput. Vis. Image Understand"},{"key":"1019_CR146","doi-asserted-by":"crossref","unstructured":"Schmidtke, L., Vlontzos, A., Ellershaw, S., Lukens, A., Arichi, T., Kainz, B.: Unsupervised human pose estimation through transforming shape templates. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2484\u20132494 (2021)","DOI":"10.1109\/CVPR46437.2021.00251"},{"key":"1019_CR147","doi-asserted-by":"crossref","unstructured":"Shang, X., Di, D., Xiao, J., Cao, Y., Yang, X., Chua, T.S.: Annotating objects and relations in user-generated videos. In: Proceedings of the 2019 on International Conference on Multimedia Retrieval, pp. 279\u2013287 (2019)","DOI":"10.1145\/3323873.3325056"},{"key":"1019_CR148","doi-asserted-by":"crossref","unstructured":"Sidenbladh, H., De\u00a0la Torre, F., Black, M.J.: A framework for modeling the appearance of 3d articulated figures. In: Proceedings Fourth IEEE International Conference on Automatic Face and Gesture Recognition (Cat. No. PR00580), IEEE, pp. 368\u2013375 (2000)","DOI":"10.1109\/AFGR.2000.840661"},{"key":"1019_CR149","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"1019_CR150","doi-asserted-by":"crossref","unstructured":"Snower, M., Kadav, A., Lai, F., Graf, H.P.: 15 keypoints is all you need. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6738\u20136748 (2020)","DOI":"10.1109\/CVPR42600.2020.00677"},{"key":"1019_CR151","doi-asserted-by":"crossref","unstructured":"Song, J., Wang, L., Van\u00a0Gool, L., Hilliges, O.: Thin-slicing network: A deep structured model for pose estimation in videos. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 4220\u20134229 (2017)","DOI":"10.1109\/CVPR.2017.590"},{"key":"1019_CR152","doi-asserted-by":"crossref","unstructured":"Su, K., Yu, D., Xu, Z., Geng, X., Wang, C.: Multi-person pose estimation with enhanced channel-wise and spatial information. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5674\u20135682 (2019)","DOI":"10.1109\/CVPR.2019.00582"},{"key":"1019_CR153","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 5693\u20135703 (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"key":"1019_CR154","doi-asserted-by":"crossref","unstructured":"Sun, X., Shang, J., Liang, S., Wei, Y.: Compositional human pose regression. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2602\u20132611 (2017)","DOI":"10.1109\/ICCV.2017.284"},{"key":"1019_CR155","doi-asserted-by":"crossref","unstructured":"Sun, X., Xiao, B., Wei, F., Liang, S., Wei, Y.: Integral human pose regression. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 529\u2013545 (2018)","DOI":"10.1007\/978-3-030-01231-1_33"},{"key":"1019_CR156","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., Rabinovich, A.: Going deeper with convolutions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 1\u20139 (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"1019_CR157","doi-asserted-by":"crossref","unstructured":"Tang, W., Wu, Y.: Does learning specific features for related parts help human pose estimation? In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1107\u20131116 (2019)","DOI":"10.1109\/CVPR.2019.00120"},{"key":"1019_CR158","doi-asserted-by":"crossref","unstructured":"Tang, W., Yu, P., Wu, Y.: Deeply learned compositional models for human pose estimation. In: Proceedings of the European conference on computer vision (ECCV), pp. 190\u2013206 (2018)","DOI":"10.1007\/978-3-030-01219-9_12"},{"key":"1019_CR159","unstructured":"Tian, Z., Chen, H., Shen, C.: Directpose: Direct end-to-end multi-person pose estimation. arXiv preprint arXiv:1911.07451 (2019)"},{"key":"1019_CR160","first-page":"1799","volume":"27","author":"JJ Tompson","year":"2014","unstructured":"Tompson, J.J., Jain, A., LeCun, Y., Bregler, C.: Joint training of a convolutional network and a graphical model for human pose estimation. Adv. Neural Inf. Process. Syst. 27, 1799\u20131807 (2014)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1019_CR161","doi-asserted-by":"crossref","unstructured":"Toshev, A., Szegedy, C.: Deeppose: Human pose estimation via deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2014)","DOI":"10.1109\/CVPR.2014.214"},{"key":"1019_CR162","doi-asserted-by":"crossref","unstructured":"Varamesh, A., Tuytelaars, T.: Mixture dense regression for object detection and human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13086\u201313095 (2020)","DOI":"10.1109\/CVPR42600.2020.01310"},{"key":"1019_CR163","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. In: Advances in neural information processing systems, pp. 5998\u20136008 (2017)"},{"key":"1019_CR164","doi-asserted-by":"crossref","unstructured":"Wang, F., Li, Y.: Beyond physical connections: Tree models in human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 596\u2013603 (2013)","DOI":"10.1109\/CVPR.2013.83"},{"key":"1019_CR165","unstructured":"Wang, F., Panev, S., Dai, Z., Han, J., Huang, D.: Can wifi estimate person pose? arXiv preprint arXiv:1904.00277 (2019a)"},{"key":"1019_CR166","doi-asserted-by":"crossref","unstructured":"Wang, F., Zhou, S., Panev, S., Han, J., Huang, D.: Person-in-wifi: Fine-grained person perception using wifi. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5452\u20135461 (2019b)","DOI":"10.1109\/ICCV.2019.00555"},{"key":"1019_CR167","doi-asserted-by":"crossref","unstructured":"Wang, H., Schmid, C.: Action recognition with improved trajectories. In: Proceedings of the IEEE international conference on computer vision, pp. 3551\u20133558 (2013)","DOI":"10.1109\/ICCV.2013.441"},{"issue":"6","key":"1019_CR168","doi-asserted-by":"publisher","first-page":"2168","DOI":"10.1109\/TVCG.2019.2903943","volume":"25","author":"J Wang","year":"2019","unstructured":"Wang, J., Gou, L., Zhang, W., Yang, H., Shen, H.W.: Deepvid: Deep visual interpretation and diagnosis for image classifiers via knowledge distillation. IEEE Trans. Visual. Comput. Graphic 25(6), 2168\u20132180 (2019)","journal-title":"IEEE Trans. Visual. Comput. Graphic"},{"key":"1019_CR169","doi-asserted-by":"crossref","unstructured":"Wang, J., Qiu, K., Peng, H., Fu, J., Zhu, J.: Ai coach: Deep human pose estimation and analysis for personalized athletic training assistance. In: Proceedings of the 27th ACM International Conference on Multimedia, pp. 374\u2013382 (2019d)","DOI":"10.1145\/3343031.3350609"},{"key":"1019_CR170","doi-asserted-by":"crossref","unstructured":"Wang, J., Long, X., Gao, Y., Ding, E., Wen, S.: Graph-pcnn: Two stage human pose estimation with graph pose refinement. In: European Conference on Computer Vision, Springer, pp. 492\u2013508 (2020a)","DOI":"10.1007\/978-3-030-58621-8_29"},{"key":"1019_CR171","doi-asserted-by":"crossref","unstructured":"Wang, J., Jin, S., Liu, W., Liu, W., Qian, C., Luo, P.: When human pose estimation meets robustness: Adversarial algorithms and benchmarks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11855\u201311864 (2021)","DOI":"10.1109\/CVPR46437.2021.01168"},{"key":"1019_CR172","doi-asserted-by":"crossref","unstructured":"Wang, M., Tighe, J., Modolo, D.: Combining detection and tracking for human pose estimation in videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11088\u201311096 (2020b)","DOI":"10.1109\/CVPR42600.2020.01110"},{"key":"1019_CR173","doi-asserted-by":"crossref","unstructured":"Wang, X., Gao, L., Song, J., Shen, H.T.: Ktn: Knowledge transfer network for multi-person densepose estimation. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 3780\u20133788 (2020c)","DOI":"10.1145\/3394171.3414014"},{"key":"1019_CR174","doi-asserted-by":"crossref","unstructured":"Wang, Y., Mori, G.: Multiple tree models for occlusion and spatial constraints in human pose estimation. In: European Conference on Computer Vision, Springer, pp. 710\u2013724 (2008)","DOI":"10.1007\/978-3-540-88690-7_53"},{"key":"1019_CR175","doi-asserted-by":"crossref","unstructured":"Wang, Y., Tran, D., Liao, Z.: Learning hierarchical poselets for human parsing. In: CVPR 2011, IEEE, pp. 1705\u20131712 (2011)","DOI":"10.1109\/CVPR.2011.5995519"},{"key":"1019_CR176","doi-asserted-by":"crossref","unstructured":"Wehrbein, T., Rudolph, M., Rosenhahn, B., Wandt, B.: Probabilistic monocular 3d human pose estimation with normalizing flows. arXiv preprint arXiv:2107.13788 (2021)","DOI":"10.1109\/ICCV48922.2021.01101"},{"key":"1019_CR177","doi-asserted-by":"crossref","unstructured":"Wei, F., Sun, X., Li, H., Wang, J., Lin, S.: Point-set anchors for object detection, instance segmentation and pose estimation. In: European Conference on Computer Vision, Springer, pp. 527\u2013544 (2020)","DOI":"10.1007\/978-3-030-58607-2_31"},{"key":"1019_CR178","doi-asserted-by":"crossref","unstructured":"Wei, S.E., Ramakrishna, V., Kanade, T., Sheikh, Y.: Convolutional pose machines. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.511"},{"key":"1019_CR179","unstructured":"Wu, J., Zheng, H., Zhao, B., Li, Y., Yan, B., Liang, R., Wang, W., Zhou, S., Lin, G., Fu, Y., et\u00a0al.: Ai challenger: A large-scale dataset for going deeper in image understanding. arXiv preprint arXiv:1711.06475 (2017)"},{"key":"1019_CR180","doi-asserted-by":"crossref","unstructured":"Xia, F., Wang, P., Chen, X., Yuille, A.L.: Joint multi-person pose estimation and semantic part segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 6769\u20136778 (2017)","DOI":"10.1109\/CVPR.2017.644"},{"key":"1019_CR181","doi-asserted-by":"crossref","unstructured":"Xiao, B., Wu, H., Wei, Y.: Simple baselines for human pose estimation and tracking. In: Proceedings of the European conference on computer vision (ECCV), pp. 466\u2013481(2018)","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"1019_CR182","unstructured":"Xiu, Y., Li, J., Wang, H., Fang, Y., Lu, C.: Pose flow: Efficient online pose tracking. arXiv preprint arXiv:1802.00977 (2018)"},{"key":"1019_CR183","doi-asserted-by":"crossref","unstructured":"Xu, X., Zou, Q., Lin, X.: Alleviating human-level shift: A robust domain adaptation method for multi-person pose estimation. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 2326\u20132335 (2020)","DOI":"10.1145\/3394171.3414040"},{"key":"1019_CR184","doi-asserted-by":"crossref","unstructured":"Yang, W., Li, S., Ouyang, W., Li, H., Wang, X.: Learning feature pyramids for human pose estimation. In: proceedings of the IEEE international conference on computer vision, pp. 1281\u20131290 (2017)","DOI":"10.1109\/ICCV.2017.144"},{"issue":"12","key":"1019_CR185","doi-asserted-by":"publisher","first-page":"2878","DOI":"10.1109\/TPAMI.2012.261","volume":"35","author":"Y Yang","year":"2012","unstructured":"Yang, Y., Ramanan, D.: Articulated human detection with flexible mixtures of parts. IEEE Trans. Pattern Anal. Mach. Intell. 35(12), 2878\u20132890 (2012)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1019_CR186","doi-asserted-by":"crossref","unstructured":"Yang, Y., Ren, Z., Li, H., Zhou, C., Wang, X., Hua, G.: Learning dynamics via graph neural networks for human pose estimation and tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8074\u20138084 (2021)","DOI":"10.1109\/CVPR46437.2021.00798"},{"key":"1019_CR187","doi-asserted-by":"crossref","unstructured":"Yu, C., Xiao, B., Gao, C., Yuan, L., Zhang, L., Sang, N., Wang, J.: Lite-hrnet: A lightweight high-resolution network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10440\u201310450 (2021)","DOI":"10.1109\/CVPR46437.2021.01030"},{"key":"1019_CR188","doi-asserted-by":"crossref","unstructured":"Yu, D., Su, K., Sun, J., Wang, C.: Multi-person pose estimation for pose tracking with enhanced cascaded pyramid network. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 0\u20130 (2018)","DOI":"10.1007\/978-3-030-11012-3_19"},{"key":"1019_CR189","doi-asserted-by":"crossref","unstructured":"Yuan, L., Zhang, S., Fubiao, F., Wei, N., Pan, H.: Combined distillation pose. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4635\u20134639 (2020)","DOI":"10.1145\/3394171.3416278"},{"key":"1019_CR190","doi-asserted-by":"crossref","unstructured":"Zeng, A., Sun, X., Yang, L., Zhao, N., Liu, M., Xu, Q.: Learning skeletal graph neural networks for hard 3d pose estimation. arXiv preprint: arXiv:2108.07181 (2021)","DOI":"10.1109\/ICCV48922.2021.01124"},{"key":"1019_CR191","doi-asserted-by":"crossref","unstructured":"Zhang, D., Shah, M.: Human pose estimation in videos. In: Proceedings of the IEEE International Conference on Computer Vision (ICCV) (2015)","DOI":"10.1109\/ICCV.2015.233"},{"key":"1019_CR192","doi-asserted-by":"crossref","unstructured":"Zhang, D., Guo, G., Huang, D., Han, J.: Poseflow: A deep motion representation for understanding human behaviors in videos. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6762\u20136770 (2018a)","DOI":"10.1109\/CVPR.2018.00707"},{"key":"1019_CR193","doi-asserted-by":"crossref","unstructured":"Zhang, F., Zhu, X., Dai, H., Ye, M., Zhu, C.: Distribution-aware coordinate representation for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7093\u20137102 (2020a)","DOI":"10.1109\/CVPR42600.2020.00712"},{"key":"1019_CR194","unstructured":"Zhang, J., Zhu, Z., Zou, W., Li, P., Li, Y., Su, H., Huang, G.: Fastpose: Towards real-time pose estimation and tracking via scale-normalized multi-task networks. arXiv preprint: arXiv:1908.05593 (2019)"},{"key":"1019_CR195","doi-asserted-by":"crossref","unstructured":"Zhang, W., Zhu, M., Derpanis, K.G.: From actemes to action: A strongly-supervised representation for detailed action understanding. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2248\u20132255 (2013)","DOI":"10.1109\/ICCV.2013.280"},{"key":"1019_CR196","doi-asserted-by":"crossref","unstructured":"Zhang, X., Li, C., Tong, X., Hu, W., Maybank, S., Zhang, Y.: Efficient human pose estimation via parsing a tree structure based human model. In: 2009 IEEE 12th International Conference on Computer Vision, IEEE, pp. 1349\u20131356 (2009)","DOI":"10.1109\/ICCV.2009.5459306"},{"key":"1019_CR197","doi-asserted-by":"crossref","unstructured":"Zhang, X., Zhou, X., Lin, M., Sun, J.: Shufflenet: An extremely efficient convolutional neural network for mobile devices. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 6848\u20136856 (2018b)","DOI":"10.1109\/CVPR.2018.00716"},{"key":"1019_CR198","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Wang, Y., Camps, O., Sznaier, M.: Key frame proposal network for efficient pose estimation in videos. In: European Conference on Computer Vision, Springer, pp. 609\u2013625 (2020b)","DOI":"10.1007\/978-3-030-58520-4_36"},{"key":"1019_CR199","doi-asserted-by":"crossref","unstructured":"Zhao, M., Li, T., Abu\u00a0Alsheikh, M., Tian, Y., Zhao, H., Torralba, A., Katabi, D.: Through-wall human pose estimation using radio signals. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7356\u20137365 (2018)","DOI":"10.1109\/CVPR.2018.00768"},{"key":"1019_CR200","unstructured":"Zheng, C., Wu, W., Yang, T., Zhu, S., Chen, C., Liu, R., Shen, J., Kehtarnavaz, N., Shah, M.: Deep learning-based human pose estimation: A survey. arXiv preprint: arXiv:2012.13392(2020)"},{"key":"1019_CR201","doi-asserted-by":"crossref","unstructured":"Zhou, C., Ren, Z., Hua, G.: Temporal keypoint matching and refinement network for pose estimation and tracking. In: European Conference on Computer Vision, Springer, pp. 680\u2013695 (2020a)","DOI":"10.1007\/978-3-030-58542-6_41"},{"key":"1019_CR202","doi-asserted-by":"crossref","unstructured":"Zhou, G., Fan, Y., Cui, R., Bian, W., Zhu, X., Gai, K.: Rocket launching: A universal and efficient framework for training well-performing light net. In: Thirty-second AAAI conference on artificial intelligence (2018)","DOI":"10.1609\/aaai.v32i1.11601"},{"key":"1019_CR203","doi-asserted-by":"crossref","unstructured":"Zhou, L., Chen, Y., Gao, Y., Wang, J., Lu, H.: Occlusion-aware siamese network for human pose estimation. In: European Conference on Computer Vision, Springer, pp. 396\u2013412 (2020b)","DOI":"10.1007\/978-3-030-58565-5_24"},{"key":"1019_CR204","doi-asserted-by":"crossref","unstructured":"Zhou, X., Huang, Q., Sun, X., Xue, X., Wei, Y.: Towards 3d human pose estimation in the wild: a weakly-supervised approach. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 398\u2013407 (2017)","DOI":"10.1109\/ICCV.2017.51"},{"key":"1019_CR205","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint: arXiv:2010.04159 (2020)"},{"key":"1019_CR206","doi-asserted-by":"crossref","unstructured":"Zou, S., Guo, C., Zuo, X., Wang, S., Wang, P., Hu, X., Chen, S., Gong, M., Cheng, L.: Eventhpe: Event-based 3d human pose and shape estimation. arXiv preprint: arXiv:2108.06819 (2021)","DOI":"10.1109\/ICCV48922.2021.01081"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01019-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-022-01019-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01019-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,8]],"date-time":"2024-10-08T00:51:16Z","timestamp":1728348676000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-022-01019-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,11]]},"references-count":206,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2023,10]]}},"alternative-id":["1019"],"URL":"https:\/\/doi.org\/10.1007\/s00530-022-01019-0","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,11,11]]},"assertion":[{"value":"6 September 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 March 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}