{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T07:52:32Z","timestamp":1782978752246,"version":"3.54.5"},"reference-count":80,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T00:00:00Z","timestamp":1765238400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T00:00:00Z","timestamp":1765238400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1007\/s00371-025-04276-y","type":"journal-article","created":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T18:49:10Z","timestamp":1765306150000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Enhanced 2D human pose estimation via feature-aligned high-resolution network"],"prefix":"10.1007","volume":"42","author":[{"given":"Yuhe","family":"Zhu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhangwen","family":"Lyu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinwei","family":"Zhan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,12,9]]},"reference":[{"key":"4276_CR1","doi-asserted-by":"publisher","unstructured":"Insafutdinov, E., Pishchulin, L., Andres, B., Andriluka, M., Schiele, B.: DeeperCut: A Deeper, Stronger, and Faster Multi-Person Pose Estimation Model, pp. 34\u201350 (2016). https:\/\/doi.org\/10.1007\/978-3-319-46466-4_3","DOI":"10.1007\/978-3-319-46466-4_3"},{"issue":"1","key":"4276_CR2","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/tpami.2019.2929257","volume":"43","author":"Z Cao","year":"2021","unstructured":"Cao, Z., Hidalgo, G., Simon, T., Wei, S.-E., Sheikh, Y.: Openpose: Realtime multi-person 2d pose estimation using part affinity fields. IEEE Trans. Pattern Anal. Mach. Intell. 43(1), 172\u2013186 (2021). https:\/\/doi.org\/10.1109\/tpami.2019.2929257","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4276_CR3","doi-asserted-by":"publisher","first-page":"346","DOI":"10.1016\/j.patcog.2017.02.030","volume":"68","author":"M Liu","year":"2017","unstructured":"Liu, M., Liu, H., Chen, C.: Enhanced skeleton visualization for view invariant human action recognition. Pattern Rec. 68, 346\u2013362 (2017). https:\/\/doi.org\/10.1016\/j.patcog.2017.02.030","journal-title":"Pattern Rec."},{"key":"4276_CR4","doi-asserted-by":"publisher","unstructured":"Liu, M., Yuan, J.: Recognizing human actions as the evolution of pose estimation maps. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2018). https:\/\/doi.org\/10.1109\/cvpr.2018.00127","DOI":"10.1109\/cvpr.2018.00127"},{"key":"4276_CR5","volume-title":"Depth pooling based large-scale 3d action recognition with convolutional neural networks","author":"P Wang","year":"2018","unstructured":"Wang, P., Li, W., Gao, Z., Tang, C., Ogunbona, P.: Depth pooling based large-scale 3d action recognition with convolutional neural networks. IEEE Transactions on Multimedia, IEEE Transactions on Multimedia (2018)"},{"key":"4276_CR6","doi-asserted-by":"publisher","unstructured":"Moon, G., Lee, K.M.: I2L-MeshNet: Image-to-Lixel Prediction Network for Accurate 3D Human Pose and Mesh Estimation from a Single RGB Image, pp. 752\u2013768 (2020). https:\/\/doi.org\/10.1007\/978-3-030-58571-6_44","DOI":"10.1007\/978-3-030-58571-6_44"},{"key":"4276_CR7","doi-asserted-by":"publisher","unstructured":"Moon, G., Lee, K.M.: I2L-MeshNet: Image-to-Lixel Prediction Network for Accurate 3D Human Pose and Mesh Estimation from a Single RGB Image, pp. 752\u2013768 (2020). https:\/\/doi.org\/10.1007\/978-3-030-58571-6_44","DOI":"10.1007\/978-3-030-58571-6_44"},{"key":"4276_CR8","unstructured":"Choi, H., Moon, G., JoonKyu, P., Lee, K.: 3dcrowdnet: 2d human pose-guided3d crowd human pose and shape estimation in the wild. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (2021)"},{"issue":"4","key":"4276_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3072959.3073596","volume":"36","author":"D Mehta","year":"2017","unstructured":"Mehta, D., Sridhar, S., Sotnychenko, O., Rhodin, H., Shafiei, M., Seidel, H.-P., Xu, W., Casas, D., Theobalt, C.: Vnect. ACM Trans. Graphics 36(4), 1\u201314 (2017). https:\/\/doi.org\/10.1145\/3072959.3073596","journal-title":"ACM Trans. Graphics"},{"key":"4276_CR10","doi-asserted-by":"publisher","unstructured":"Hagbi, N., Bergig, O., El-Sana, J., Billinghurst, M.: Shape recognition and pose estimation for mobile augmented reality. In: 2009 8th IEEE International Symposium on Mixed and Augmented Reality, pp. 65\u201371 (2009). https:\/\/doi.org\/10.1109\/ismar.2009.5336498","DOI":"10.1109\/ismar.2009.5336498"},{"key":"4276_CR11","doi-asserted-by":"publisher","DOI":"10.1007\/0-387-27890-7","author":"B Kisa\u010danin","year":"2005","unstructured":"Kisa\u010danin, B., Pavlovic, V., Huang, T.: Real-Time Vision for Human-Comp. Interact. (2005). https:\/\/doi.org\/10.1007\/0-387-27890-7","journal-title":"Real-Time Vision for Human-Comp. Interact."},{"key":"4276_CR12","doi-asserted-by":"publisher","unstructured":"Svenstrup, M., Tranberg, S., Andersen, H.J., Bak, T.: Pose estimation and adaptive robot behaviour for human-robot interaction. In: 2009 IEEE International Conference on Robotics and Automation (2009). https:\/\/doi.org\/10.1109\/robot.2009.5152690","DOI":"10.1109\/robot.2009.5152690"},{"key":"4276_CR13","doi-asserted-by":"publisher","unstructured":"Chen, Y., Wang, Z., Peng, Y., Zhang, Z., Yu, G., Sun, J.: Cascaded pyramid network for multi-person pose estimation. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2018). https:\/\/doi.org\/10.1109\/cvpr.2018.00742","DOI":"10.1109\/cvpr.2018.00742"},{"key":"4276_CR14","doi-asserted-by":"publisher","unstructured":"Martinez, J., Hossain, R., Romero, J., Little, J.J.: A simple yet effective baseline for 3d human pose estimation. In: 2017 IEEE International Conference on Computer Vision (ICCV) (2017). https:\/\/doi.org\/10.1109\/iccv.2017.288","DOI":"10.1109\/iccv.2017.288"},{"key":"4276_CR15","doi-asserted-by":"publisher","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked Hourglass Networks for Human Pose Estimation, pp. 483\u2013499 (2016). https:\/\/doi.org\/10.1007\/978-3-319-46484-8_29","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"4276_CR16","doi-asserted-by":"publisher","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019). https:\/\/doi.org\/10.1109\/cvpr.2019.00584","DOI":"10.1109\/cvpr.2019.00584"},{"key":"4276_CR17","unstructured":"Tan, M., Le, Q.: Efficientnet: Rethinking model scaling for convolutional neural networks (2019)"},{"key":"4276_CR18","doi-asserted-by":"publisher","unstructured":"Hou, Q., Zhou, D., Feng, J.: Coordinate attention for efficient mobile network design. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2021). https:\/\/doi.org\/10.1109\/cvpr46437.2021.01350","DOI":"10.1109\/cvpr46437.2021.01350"},{"key":"4276_CR19","doi-asserted-by":"publisher","unstructured":"Toshev, A., Szegedy, C.: Deeppose: Human pose estimation via deep neural networks. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition (2014). https:\/\/doi.org\/10.1109\/cvpr.2014.214","DOI":"10.1109\/cvpr.2014.214"},{"key":"4276_CR20","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_16","volume-title":"Human pose estimation using deep consensus voting","author":"I Lifshitz","year":"2016","unstructured":"Lifshitz, I., Fetaya, E., Ullman, S.: Human pose estimation using deep consensus voting. Cornell University - arXiv, Cornell University - arXiv (2016)"},{"key":"4276_CR21","doi-asserted-by":"publisher","unstructured":"Carreira, J., Agrawal, P., Fragkiadaki, K., Malik, J.: Human pose estimation with iterative error feedback. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016). https:\/\/doi.org\/10.1109\/cvpr.2016.512","DOI":"10.1109\/cvpr.2016.512"},{"key":"4276_CR22","doi-asserted-by":"crossref","unstructured":"Li, J., Bian, S., Zeng, A., Wang, C., Pang, B., Li, W., Lu, C.: Human pose regression with residual log-likelihood estimation. Cornell University - arXiv, Cornell University - arXiv (2021)","DOI":"10.1109\/ICCV48922.2021.01084"},{"key":"4276_CR23","doi-asserted-by":"publisher","unstructured":"Yang, W., Ouyang, W., Li, H., Wang, X.: End-to-end learning of deformable mixture of parts and deep convolutional neural networks for human pose estimation. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016). https:\/\/doi.org\/10.1109\/cvpr.2016.335","DOI":"10.1109\/cvpr.2016.335"},{"key":"4276_CR24","unstructured":"Tompson, J., Jain, A., LeCun, Y., Bregler, C.: Joint training of a convolutional network and a graphical model for human pose estimation. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (2014)"},{"key":"4276_CR25","doi-asserted-by":"publisher","unstructured":"Chu, X., Ouyang, W., Li, H., Wang, X.: Structured feature learning for pose estimation. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016). https:\/\/doi.org\/10.1109\/cvpr.2016.510","DOI":"10.1109\/cvpr.2016.510"},{"key":"4276_CR26","doi-asserted-by":"publisher","unstructured":"Chu, X., Yang, W., Ouyang, W., Ma, C., Yuille, A.L., Wang, X.: Multi-context attention for human pose estimation. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017). https:\/\/doi.org\/10.1109\/cvpr.2017.601","DOI":"10.1109\/cvpr.2017.601"},{"key":"4276_CR27","doi-asserted-by":"publisher","unstructured":"Wei, S.-E., Ramakrishna, V., Kanade, T., Sheikh, Y.: Convolutional pose machines. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016). https:\/\/doi.org\/10.1109\/cvpr.2016.511","DOI":"10.1109\/cvpr.2016.511"},{"key":"4276_CR28","doi-asserted-by":"publisher","unstructured":"Chen, Y., Shen, C., Wei, X.-S., Liu, L., Yang, J.: Adversarial posenet: A structure-aware convolutional network for human pose estimation. In: 2017 IEEE International Conference on Computer Vision (ICCV) (2017). https:\/\/doi.org\/10.1109\/iccv.2017.137","DOI":"10.1109\/iccv.2017.137"},{"key":"4276_CR29","doi-asserted-by":"publisher","unstructured":"Artacho, B., Savakis, A.: Unipose: Unified human pose estimation in single images and videos. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020). https:\/\/doi.org\/10.1109\/cvpr42600.2020.00706","DOI":"10.1109\/cvpr42600.2020.00706"},{"key":"4276_CR30","doi-asserted-by":"publisher","first-page":"1330","DOI":"10.1109\/TMM.2020.2999181","volume":"23","author":"A Kamel","year":"2020","unstructured":"Kamel, A., Sheng, B., Li, P., Kim, J., Feng, D.D.: Hybrid refinement-correction heatmaps for human pose estimation. IEEE Trans. Multimedia 23, 1330\u20131342 (2020)","journal-title":"IEEE Trans. Multimedia"},{"issue":"10","key":"4276_CR31","doi-asserted-by":"publisher","first-page":"3349","DOI":"10.1109\/tpami.2020.2983686","volume":"43","author":"J Wang","year":"2021","unstructured":"Wang, J., Sun, K., Cheng, T., Jiang, B., Deng, C., Zhao, Y., Liu, D., Mu, Y., Tan, M., Wang, X., Liu, W., Xiao, B.: Deep high-resolution representation learning for visual recognition. IEEE Tran. Pattern Anal. Mach. Intell. 43(10), 3349\u20133364 (2021). https:\/\/doi.org\/10.1109\/tpami.2020.2983686","journal-title":"IEEE Tran. Pattern Anal. Mach. Intell."},{"key":"4276_CR32","doi-asserted-by":"publisher","unstructured":"Yu, C., Xiao, B., Gao, C., Yuan, L., Zhang, L., Sang, N., Wang, J.: Lite-hrnet: A lightweight high-resolution network. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2021). https:\/\/doi.org\/10.1109\/cvpr46437.2021.01030","DOI":"10.1109\/cvpr46437.2021.01030"},{"key":"4276_CR33","doi-asserted-by":"publisher","unstructured":"Ma, N., Zhang, X., Zheng, H.-T., Sun, J.: ShuffleNet V2: Practical Guidelines for Efficient CNN Architecture Design, pp. 122\u2013138 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01264-9_8","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"4276_CR34","doi-asserted-by":"publisher","unstructured":"Zhou, Y., Wang, X., Xu, X., Zhao, L., Song, J.: X-hrnet: Towards lightweight human pose estimation with spatially unidimensional self-attention. In: 2022 IEEE International Conference on Multimedia and Expo (ICME) (2022). https:\/\/doi.org\/10.1109\/icme52920.2022.9859751","DOI":"10.1109\/icme52920.2022.9859751"},{"key":"4276_CR35","doi-asserted-by":"crossref","unstructured":"Li, Q., Zhang, Z., Xiao, F., Zhang, F., Bhanu, B.: Dite-hrnet: Dynamic lightweight high-resolution network for human pose estimation (2022)","DOI":"10.24963\/ijcai.2022\/153"},{"key":"4276_CR36","unstructured":"Zhang, Z., Sun, X., Dang, Y., Yin, J.: Bihrnet: A binary high-resolution network for human pose estimation"},{"key":"4276_CR37","volume-title":"Binarized neural networks","author":"I Hubara","year":"2016","unstructured":"Hubara, I., Courbariaux, M., Soudry, D., El-Yaniv, R., Bengio, Y.: Binarized neural networks. Neural Information Processing Systems, Neural Information Processing Systems (2016)"},{"key":"4276_CR38","doi-asserted-by":"publisher","unstructured":"Cheng, B., Xiao, B., Wang, J., Shi, H., Huang, T.S., Zhang, L.: Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020). https:\/\/doi.org\/10.1109\/cvpr42600.2020.00543","DOI":"10.1109\/cvpr42600.2020.00543"},{"key":"4276_CR39","doi-asserted-by":"crossref","unstructured":"Neff, C., Sheth, A., Furgurson, S., Tabkhi, H.: Efficienthrnet: Efficient scaling for lightweight high-resolution multi-person pose estimation. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (2020)","DOI":"10.1007\/s11554-021-01132-9"},{"key":"4276_CR40","doi-asserted-by":"publisher","unstructured":"Yang, W., Li, S., Ouyang, W., Li, H., Wang, X.: Learning feature pyramids for human pose estimation. In: 2017 IEEE International Conference on Computer Vision (ICCV) (2017). https:\/\/doi.org\/10.1109\/iccv.2017.144","DOI":"10.1109\/iccv.2017.144"},{"key":"4276_CR41","doi-asserted-by":"publisher","unstructured":"Ke, L., Chang, M.-C., Qi, H., Lyu, S.: Multi-scale structure-aware network for human pose estimation, pp. 731\u2013746 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01216-8_44","DOI":"10.1007\/978-3-030-01216-8_44"},{"key":"4276_CR42","doi-asserted-by":"publisher","unstructured":"Lin, T.-Y., Dollar, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017). https:\/\/doi.org\/10.1109\/cvpr.2017.106","DOI":"10.1109\/cvpr.2017.106"},{"key":"4276_CR43","doi-asserted-by":"publisher","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2016). https:\/\/doi.org\/10.1109\/cvpr.2016.90","DOI":"10.1109\/cvpr.2016.90"},{"issue":"10","key":"4276_CR44","doi-asserted-by":"publisher","first-page":"4751","DOI":"10.1007\/s00371-022-02623-x","volume":"39","author":"R Wang","year":"2023","unstructured":"Wang, R., Wu, W., Wang, X.: Enhancing multi-scale information exchange and feature fusion for human pose estimation. Vis. Comput. 39(10), 4751\u20134765 (2023)","journal-title":"Vis. Comput."},{"issue":"2","key":"4276_CR45","doi-asserted-by":"publisher","first-page":"651","DOI":"10.1007\/s00371-021-02364-3","volume":"39","author":"Q Zhang","year":"2023","unstructured":"Zhang, Q., Chen, Y.: Spatial and contextual aware network based on multi-resolution for human pose estimation. Vis. Comput. 39(2), 651\u2013662 (2023)","journal-title":"Vis. Comput."},{"key":"4276_CR46","doi-asserted-by":"publisher","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.-C.: Mobilenetv2: Inverted residuals and linear bottlenecks. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2018). https:\/\/doi.org\/10.1109\/cvpr.2018.00474","DOI":"10.1109\/cvpr.2018.00474"},{"key":"4276_CR47","doi-asserted-by":"publisher","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft COCO: Common Objects in Context, pp. 740\u2013755 (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"4276_CR48","doi-asserted-by":"publisher","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., Schiele, B.: 2d human pose estimation: New benchmark and state of the art analysis. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition (2014). https:\/\/doi.org\/10.1109\/cvpr.2014.471","DOI":"10.1109\/cvpr.2014.471"},{"key":"4276_CR49","doi-asserted-by":"publisher","unstructured":"Johnson, S., Everingham, M.: Clustered pose and nonlinear appearance models for human pose estimation. In: Procedings of the British Machine Vision Conference 2010 (2010). https:\/\/doi.org\/10.5244\/c.24.12","DOI":"10.5244\/c.24.12"},{"key":"4276_CR50","doi-asserted-by":"crossref","unstructured":"Lu, X., Cao, Y., Liu, S., Long, C., Chen, Z., Zhou, X., Yang, Y., Xiao, C.: Video shadow detection via spatio-temporal interpolation consistency training (2022)","DOI":"10.1109\/CVPR52688.2022.00312"},{"key":"4276_CR51","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102522","volume":"111","author":"T Zhang","year":"2024","unstructured":"Zhang, T., Li, Q., Wen, J., Philip Chen, C.L.: Enhancement and optimisation of human pose estimation with multi-scale spatial attention and adversarial data augmentation. Inf. Fusion 111, 102522 (2024). https:\/\/doi.org\/10.1016\/j.inffus.2024.102522","journal-title":"Inf. Fusion"},{"key":"4276_CR52","doi-asserted-by":"publisher","unstructured":"Papandreou, G., Zhu, T., Chen, L.-C., Gidaris, S., Tompson, J., Murphy, K.: PersonLab: person pose estimation and instance segmentation with a bottom-up, part-based, Geometric Embedding Model, pp. 282\u2013299 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01264-9_17","DOI":"10.1007\/978-3-030-01264-9_17"},{"key":"4276_CR53","unstructured":"Yuan, Y., Fu, R., Huang, L., Lin, W., Zhang, C., Chen, X., Wang, J.: Hrformer: High-resolution transformer for dense prediction"},{"key":"4276_CR54","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhang, S., Wang, Z., Yang, S., Yang, W., Xia, S.-T., Zhou, E.: Tokenpose: Learning keypoint tokens for human pose estimation. In: IEEE\/CVF International Conference on Computer Vision (ICCV) (2021)","DOI":"10.1109\/ICCV48922.2021.01112"},{"key":"4276_CR55","doi-asserted-by":"crossref","unstructured":"Dwivedi, S.K., Sun, Y., Patel, P., Feng, Y., Black, M.J.: TokenHMR: Advancing human mesh recovery with a tokenized pose representation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2024)","DOI":"10.1109\/CVPR52733.2024.00132"},{"key":"4276_CR56","unstructured":"Xu, Y., Zhang, J., Zhang, Q., Tao, D.: Vitpose: Simple vision transformer baselines for human pose estimation (2022)"},{"key":"4276_CR57","unstructured":"Yang, S., Quan, Z., Nie, M., Yang, W.: Transpose: Towards explainable human pose estimation by transformer. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (2020)"},{"key":"4276_CR58","doi-asserted-by":"crossref","unstructured":"Han, J.: Greit-hrnet: Grouped lightweight high-resolution network for human pose estimation. ArXiv abs\/2407.07389 (2024)","DOI":"10.1007\/978-981-96-0885-0_15"},{"key":"4276_CR59","unstructured":"Bazarevsky, V., Grishchenko, I., Raveendran, K., Zhu, T., Zhang, F., Grundmann, M.: Blazepose: On-device real-time body pose tracking. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (2020)"},{"key":"4276_CR60","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A., Kaiser, L., Polosukhin, I.: Attention is all you need. Neural Information Processing Systems (2017)"},{"key":"4276_CR61","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (2020)"},{"key":"4276_CR62","doi-asserted-by":"publisher","unstructured":"Sun, X., Xiao, B., Wei, F., Liang, S., Wei, Y.: Integral Human Pose Regression, pp. 536\u2013553 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01231-1_33","DOI":"10.1007\/978-3-030-01231-1_33"},{"key":"4276_CR63","doi-asserted-by":"publisher","unstructured":"Wang, T., Hsieh, Y.-Y., Wong, F.-W., Chen, Y.-F.: Mask-rcnn based people detection using a top-view fisheye camera. In: 2019 International Conference on Technologies and Applications of Artificial Intelligence (TAAI) (2019). https:\/\/doi.org\/10.1109\/taai48200.2019.8959887","DOI":"10.1109\/taai48200.2019.8959887"},{"key":"4276_CR64","doi-asserted-by":"publisher","unstructured":"Papandreou, G., Zhu, T., Kanazawa, N., Toshev, A., Tompson, J., Bregler, C., Murphy, K.: Towards accurate multi-person pose estimation in the wild. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017). https:\/\/doi.org\/10.1109\/cvpr.2017.395","DOI":"10.1109\/cvpr.2017.395"},{"key":"4276_CR65","doi-asserted-by":"publisher","unstructured":"Fang, H.-S., Xie, S., Tai, Y.-W., Lu, C.: Rmpe: Regional multi-person pose estimation. In: 2017 IEEE International Conference on Computer Vision (ICCV) (2017). https:\/\/doi.org\/10.1109\/iccv.2017.256","DOI":"10.1109\/iccv.2017.256"},{"key":"4276_CR66","doi-asserted-by":"publisher","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: 2021 IEEE\/CVF International Conference on Computer Vision (ICCV) (2021). https:\/\/doi.org\/10.1109\/iccv48922.2021.00986","DOI":"10.1109\/iccv48922.2021.00986"},{"key":"4276_CR67","doi-asserted-by":"publisher","unstructured":"Rajchl, M., Lee, M.C.H., Oktay, O., Kamnitsas, K., Passerat-Palmbach, J., Bai, W., Damodaram, M., Rutherford, M.A., Hajnal, J.V., Kainz, B., Rueckert, D.: Deepcut: Object segmentation from bounding box annotations using convolutional neural networks. IEEE Transactions on Medical Imaging, 674\u2013683 (2017) https:\/\/doi.org\/10.1109\/tmi.2016.2621185","DOI":"10.1109\/tmi.2016.2621185"},{"key":"4276_CR68","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2025.111454","volume":"165","author":"PT Huu","year":"2025","unstructured":"Huu, P.T., An, N.T., Trung, N.N.: Contextual and uncertainty-aware approach for multi-person pose estimation. Pattern Rec. 165, 111454 (2025)","journal-title":"Pattern Rec."},{"key":"4276_CR69","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2024.129154","volume":"619","author":"Z Xu","year":"2025","unstructured":"Xu, Z., Dai, M., Zhang, Q., Jiang, X.: Hrpvt: High-resolution pyramid vision transformer for medium and small-scale human pose estimation. Neurocomputing 619, 129154 (2025)","journal-title":"Neurocomputing"},{"key":"4276_CR70","doi-asserted-by":"publisher","first-page":"152","DOI":"10.1016\/j.neucom.2023.03.063","volume":"537","author":"C Wang","year":"2023","unstructured":"Wang, C., Zhou, Y., Zhang, F., Mok, P.Y.: Unbiased feature position alignment for human pose estimation. Neurocomputing 537, 152\u2013163 (2023). https:\/\/doi.org\/10.1016\/j.neucom.2023.03.063","journal-title":"Neurocomputing"},{"key":"4276_CR71","unstructured":"Liu, Z., Feng, R., Chen, H., Wu, S., Gao, Y., Gao, Y., Wang, X.: Temporal feature alignment and mutual information maximization for video-based human pose estimation"},{"key":"4276_CR72","doi-asserted-by":"crossref","unstructured":"Singh, H., Verma, M., Cheruku, R.: Dsfnet: Video salient object detection using a novel lightweight deformable separable fusion network. IEEE Transactions on Instrumentation and Measurement, 73 (2024)","DOI":"10.1109\/TIM.2024.3470045"},{"issue":"2","key":"4276_CR73","first-page":"1","volume":"14","author":"H Singh","year":"2025","unstructured":"Singh, H., Verma, M., Cheruku, R.: Dmfnet: geometric multi-scale pixel-level contrastive learning for video salient object detection. Int. J. Multimed. Inf. Retr. 14(2), 1\u201321 (2025)","journal-title":"Int. J. Multimed. Inf. Retr."},{"key":"4276_CR74","doi-asserted-by":"publisher","unstructured":"Singh, H., Verma, M., Cheruku, R.: Dsnet: Efficient lightweight model for video salient object detection for iot and wot applications. In: Companion Proceedings of the ACM Web Conference 2023, pp. 1286\u20131295. Association for Computing Machinery, New York, NY, USA (2023). https:\/\/doi.org\/10.1145\/3543873.3587592","DOI":"10.1145\/3543873.3587592"},{"key":"4276_CR75","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TIM.2023.3302911","volume":"72","author":"H Singh","year":"2023","unstructured":"Singh, H., Verma, M., Cheruku, R.: Novel dilated separable convolution networks for efficient video salient object detection in the wild. IEEE Trans. Instrum. Measurement 72, 1\u201313 (2023). https:\/\/doi.org\/10.1109\/TIM.2023.3302911","journal-title":"IEEE Trans. Instrum. Measurement"},{"issue":"6","key":"4276_CR76","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s00034-024-02983-w","volume":"44","author":"H Singh","year":"2025","unstructured":"Singh, H., Verma, M., Cheruku, R.: Hsnet: A novel edge-preserving hierarchical separable network for video shadow detection. Circuits, Syst., Signal Processing 44(6), 1\u201330 (2025)","journal-title":"Circuits, Syst., Signal Processing"},{"key":"4276_CR77","doi-asserted-by":"publisher","unstructured":"Zhang, S.-H., Li, R., Dong, X., Rosin, P., Cai, Z., Han, X., Yang, D., Huang, H., Hu, S.-M.: Pose2seg: Detection free human instance segmentation. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2019). https:\/\/doi.org\/10.1109\/cvpr.2019.00098","DOI":"10.1109\/cvpr.2019.00098"},{"key":"4276_CR78","doi-asserted-by":"publisher","unstructured":"Hu, J., Shen, L., Albanie, S., Sun, G., Wu, E.: Squeeze-and-excitation networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 2011\u20132023 (2020) https:\/\/doi.org\/10.1109\/tpami.2019.2913372","DOI":"10.1109\/tpami.2019.2913372"},{"key":"4276_CR79","doi-asserted-by":"publisher","unstructured":"Woo, S., Park, J., Lee, J.-Y., Kweon, I.S.: CBAM: Convolutional Block Attention Module, pp. 3\u201319 (2018). https:\/\/doi.org\/10.1007\/978-3-030-01234-2_1","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"4276_CR80","unstructured":"Jaderberg, M., Simonyan, K., Zisserman, A., Kavukcuoglu, K.: Spatial transformer networks. Neural Information Processing Systems (2015)"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04276-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-04276-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-04276-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T13:02:40Z","timestamp":1772629360000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-04276-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,9]]},"references-count":80,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["4276"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-04276-y","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"value":"0178-2789","type":"print"},{"value":"1432-2315","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,9]]},"assertion":[{"value":"3 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"33"}}