{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,22]],"date-time":"2025-05-22T13:49:45Z","timestamp":1747921785834,"version":"3.37.3"},"reference-count":59,"publisher":"Springer Science and Business Media LLC","issue":"8-9","license":[{"start":{"date-parts":[[2024,6,13]],"date-time":"2024-06-13T00:00:00Z","timestamp":1718236800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,13]],"date-time":"2024-06-13T00:00:00Z","timestamp":1718236800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100003787","name":"Natural Science Foundation of Hebei Province","doi-asserted-by":"crossref","award":["F2019201451","F2019201451","F2019201451"],"award-info":[{"award-number":["F2019201451","F2019201451","F2019201451"]}],"id":[{"id":"10.13039\/501100003787","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s11760-024-03261-7","type":"journal-article","created":{"date-parts":[[2024,6,13]],"date-time":"2024-06-13T15:03:15Z","timestamp":1718290995000},"page":"5647-5661","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Combining self-attention and depth-wise convolution for human pose estimation"],"prefix":"10.1007","volume":"18","author":[{"given":"Fan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingxuan","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanli","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,6,13]]},"reference":[{"key":"3261_CR1","doi-asserted-by":"crossref","unstructured":"Li, B., Dai, Y., Cheng, X., Chen, H., Lin, Y., He, M.: Skeleton based action recognition using translation-scale invariant image mapping and multi-scale deep CNN, pp. 601\u2013604 (2017)","DOI":"10.1109\/ICMEW.2017.8026282"},{"key":"3261_CR2","doi-asserted-by":"publisher","first-page":"22901","DOI":"10.1007\/s11042-018-5642-0","volume":"77","author":"B Li","year":"2018","unstructured":"Li, B., He, M., Dai, Y., Cheng, X., Chen, Y.: 3D skeleton based action recognition by video-domain translation-scale invariant mapping and multi-scale dilated CNN. Multimedia Tools Appl. 77, 22901\u201322921 (2018)","journal-title":"Multimedia Tools Appl."},{"key":"3261_CR3","doi-asserted-by":"crossref","unstructured":"Luvizon, D.C., Picard, D., Tabia, H.: 2D\/3D pose estimation and action recognition using multitask deep learning, pp. 5137\u20135146 (2018)","DOI":"10.1109\/CVPR.2018.00539"},{"key":"3261_CR4","unstructured":"Li, B., Chen, H., Chen, Y., Dai, Y., He, M.: Skeleton boxes: solving skeleton based action detection with a single deep convolutional neural network, pp. 613\u2013616 (2017)"},{"key":"3261_CR5","doi-asserted-by":"crossref","unstructured":"Zhou, T., Wang, W., Qi, S., Ling, H., Shen, J.: Cascaded human-object interaction recognition, pp. 4263\u20134272 (2020)","DOI":"10.1109\/CVPR42600.2020.00432"},{"key":"3261_CR6","doi-asserted-by":"crossref","unstructured":"Wan, B., Zhou, D., Liu, Y., Li, R., He, X.: Pose-aware multi-level feature network for human object interaction detection, pp. 9469\u20139478 (2019)","DOI":"10.1109\/ICCV.2019.00956"},{"key":"3261_CR7","doi-asserted-by":"crossref","unstructured":"Yang, J., Zhang, J., Yu, F., Jiang, X., Zhang, M., Sun, X., Chen, Y.-C., Zheng, W.-S.: Learning to know where to see: a visibility-aware approach for occluded person re-identification, pp. 11885\u201311894 (2021)","DOI":"10.1109\/ICCV48922.2021.01167"},{"key":"3261_CR8","doi-asserted-by":"crossref","unstructured":"Chen, H., Lagadec, B., Bremond, F.: Ice: Inter-instance contrastive encoding for unsupervised person re-identification, pp. 14960\u201314969 (2021)","DOI":"10.1109\/ICCV48922.2021.01469"},{"key":"3261_CR9","doi-asserted-by":"crossref","unstructured":"Luo, Z., Wang, Z., Huang, Y., Wang, L., Tan, T., Zhou, E.: Rethinking the heatmap regression for bottom-up human pose estimation, pp. 13264\u201313273 (2021)","DOI":"10.1109\/CVPR46437.2021.01306"},{"key":"3261_CR10","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"3261_CR11","doi-asserted-by":"crossref","unstructured":"Woo, S., Park, J., Lee, J.-Y., Kweon, I.S.: CBAM: convolutional block attention module, pp. 3\u201319 (2018)","DOI":"10.1007\/978-3-030-01234-2_1"},{"key":"3261_CR12","doi-asserted-by":"crossref","unstructured":"Fu, J., Liu, J., Tian, H., Li, Y., Bao, Y., Fang, Z., Lu, H.: Dual attention network for scene segmentation, pp. 3146\u20133154 (2019)","DOI":"10.1109\/CVPR.2019.00326"},{"key":"3261_CR13","doi-asserted-by":"crossref","unstructured":"Liu, H., Liu, F., Fan, X., Huang, D.: Polarized self-attention: Towards high-quality pixel-wise regression (2021). arXiv preprint arXiv:2107.00782","DOI":"10.1016\/j.neucom.2022.07.054"},{"key":"3261_CR14","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation, pp. 5693\u20135703 (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"issue":"11","key":"3261_CR15","doi-asserted-by":"publisher","first-page":"1254","DOI":"10.1109\/34.730558","volume":"20","author":"L Itti","year":"1998","unstructured":"Itti, L., Koch, C., Niebur, E.: A model of saliency-based visual attention for rapid scene analysis. IEEE Trans. Pattern Anal. Mach. Intell. 20(11), 1254\u20131259 (1998)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"3","key":"3261_CR16","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1038\/nrn755","volume":"3","author":"M Corbetta","year":"2002","unstructured":"Corbetta, M., Shulman, G.L.: Control of goal-directed and stimulus-driven attention in the brain. Nat. Rev. Neurosci. 3(3), 201\u2013215 (2002)","journal-title":"Nat. Rev. Neurosci."},{"issue":"1\u20133","key":"3261_CR17","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1080\/135062800394667","volume":"7","author":"RA Rensink","year":"2000","unstructured":"Rensink, R.A.: The dynamic representation of scenes. Vis. Cogn. 7(1\u20133), 17\u201342 (2000)","journal-title":"Vis. Cogn."},{"key":"3261_CR18","doi-asserted-by":"crossref","unstructured":"Pan, X., Ge, C., Lu, R., Song, S., Chen, G., Huang, Z., Huang, G.: On the integration of self-attention and convolution, pp. 815\u2013825 (2022)","DOI":"10.1109\/CVPR52688.2022.00089"},{"key":"3261_CR19","doi-asserted-by":"crossref","unstructured":"Zhu, L., Wang, X., Ke, Z., Zhang, W., Lau, R.W.: Biformer: Vision transformer with bi-level routing attention, pp. 10323\u201310333 (2023)","DOI":"10.1109\/CVPR52729.2023.00995"},{"key":"3261_CR20","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding (2018). arXiv preprint arXiv:1810.04805"},{"key":"3261_CR21","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141, Polosukhin, I.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"3261_CR22","doi-asserted-by":"crossref","unstructured":"Li, W., Chen, H., Guo, J., Zhang, Z., Wang, Y.: Brain-inspired multilayer perceptron with spiking neurons, pp. 783\u2013793 (2022)","DOI":"10.1109\/CVPR52688.2022.00086"},{"key":"3261_CR23","unstructured":"Chu, X., Tian, Z., Zhang, B., Wang, X., Wei, X., Xia, H., Shen, C.: Conditional positional encodings for vision transformers (2021). arXiv preprint arXiv:2102.10882"},{"key":"3261_CR24","doi-asserted-by":"crossref","unstructured":"Guo, J., Tang, Y., Han, K., Chen, X., Wu, H., Xu, C., Xu, C., Wang, Y.: Hire-MLP: Vision MLP via hierarchical rearrangement, pp. 826\u2013836 (2022)","DOI":"10.1109\/CVPR52688.2022.00090"},{"key":"3261_CR25","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et al.: An image is worth 16x16 words: transformers for image recognition at scale (2020). arXiv preprint arXiv:2010.11929"},{"key":"3261_CR26","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention, pp. 10347\u201310357 (2021)"},{"key":"3261_CR27","unstructured":"Li, Y., Zhang, K., Cao, J., Timofte, R., Van Gool, L.: Localvit: Bringing locality to vision transformers (2021). arXiv preprint arXiv:2104.05707"},{"key":"3261_CR28","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"3261_CR29","doi-asserted-by":"crossref","unstructured":"Liu, H., Jiang, X., Li, X., Bao, Z., Jiang, D., Ren, B.: Nommer: Nominate synergistic context in vision transformer for visual recognition, pp. 12073\u201312082 (2022)","DOI":"10.1109\/CVPR52688.2022.01176"},{"key":"3261_CR30","doi-asserted-by":"crossref","unstructured":"Xia, Z., Pan, X., Song, S., Li, L.E., Huang, G.: Vision transformer with deformable attention, pp. 4794\u20134803 (2022)","DOI":"10.1109\/CVPR52688.2022.00475"},{"key":"3261_CR31","doi-asserted-by":"crossref","unstructured":"Wu, H., Xiao, B., Codella, N., Liu, M., Dai, X., Yuan, L., Zhang, L.: CVT: introducing convolutions to vision transformers, pp. 22\u201331 (2021)","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"3261_CR32","doi-asserted-by":"crossref","unstructured":"Lin, W., Wu, Z., Chen, J., Huang, J., Jin, L.: Scale-aware modulation meet transformer, pp. 6015\u20136026 (2023)","DOI":"10.1109\/ICCV51070.2023.00553"},{"key":"3261_CR33","doi-asserted-by":"crossref","unstructured":"Tang, W., Yu, P., Wu, Y.: Deeply learned compositional models for human pose estimation, pp. 190\u2013206 (2018)","DOI":"10.1007\/978-3-030-01219-9_12"},{"key":"3261_CR34","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Toshev, A., Jaitly, N.: Chained predictions using convolutional neural networks, pp. 728\u2013743 (2016)","DOI":"10.1007\/978-3-319-46493-0_44"},{"key":"3261_CR35","doi-asserted-by":"crossref","unstructured":"Lifshitz, I., Fetaya, E., Ullman, S.: Human pose estimation using deep consensus voting, pp. 246\u2013260 (2016)","DOI":"10.1007\/978-3-319-46475-6_16"},{"key":"3261_CR36","doi-asserted-by":"crossref","unstructured":"Cai, Y., Wang, Z., Luo, Z., Yin, B., Du, A., Wang, H., Zhang, X., Zhou, X., Zhou, E., Sun, J.: Learning delicate local representations for multi-person pose estimation, pp. 455\u2013472 (2020)","DOI":"10.1007\/978-3-030-58580-8_27"},{"key":"3261_CR37","doi-asserted-by":"crossref","unstructured":"Cheng, B., Xiao, B., Wang, J., Shi, H., Huang, T.S., Zhang, L.: Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation, pp. 5386\u20135395 (2020)","DOI":"10.1109\/CVPR42600.2020.00543"},{"key":"3261_CR38","doi-asserted-by":"crossref","unstructured":"Pfister, T., Charles, J., Zisserman, A.: Flowing convnets for human pose estimation in videos, pp. 1913\u20131921 (2015)","DOI":"10.1109\/ICCV.2015.222"},{"key":"3261_CR39","doi-asserted-by":"crossref","unstructured":"Chu, X., Yang, W., Ouyang, W., Ma, C., Yuille, A.L., Wang, X.: Multi-context attention for human pose estimation, pp. 1831\u20131840 (2017)","DOI":"10.1109\/CVPR.2017.601"},{"key":"3261_CR40","doi-asserted-by":"crossref","unstructured":"Xiao, B., Wu, H., Wei, Y.: Simple baselines for human pose estimation and tracking, pp. 466\u2013481 (2018)","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"3261_CR41","doi-asserted-by":"crossref","unstructured":"Newell, A., Yang, K., Deng, J.: Stacked hourglass networks for human pose estimation, pp. 483\u2013499 (2016)","DOI":"10.1007\/978-3-319-46484-8_29"},{"key":"3261_CR42","doi-asserted-by":"crossref","unstructured":"Wei, S.-E., Ramakrishna, V., Kanade, T., Sheikh, Y.: Convolutional pose machines, pp. 4724\u20134732 (2016)","DOI":"10.1109\/CVPR.2016.511"},{"key":"3261_CR43","doi-asserted-by":"crossref","unstructured":"Yang, W., Li, S., Ouyang, W., Li, H., Wang, X.: Learning feature pyramids for human pose estimation, pp. 1281\u20131290 (2017)","DOI":"10.1109\/ICCV.2017.144"},{"key":"3261_CR44","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, Z., Peng, Y., Zhang, Z., Yu, G., Sun, J.: Cascaded pyramid network for multi-person pose estimation, pp. 7103\u20137112 (2018)","DOI":"10.1109\/CVPR.2018.00742"},{"key":"3261_CR45","doi-asserted-by":"crossref","unstructured":"Ramakrishna, V., Munoz, D., Hebert, M., Andrew Bagnell, J., Sheikh, Y.: Pose machines: Articulated pose estimation via inference machines, pp. 33\u201347 (2014)","DOI":"10.1007\/978-3-319-10605-2_3"},{"key":"3261_CR46","unstructured":"Tompson, J.J., Jain, A., LeCun, Y., Bregler, C.: Joint training of a convolutional network and a graphical model for human pose estimation. In: Advances in neural information processing systems, vol. 27 (2014)"},{"key":"3261_CR47","doi-asserted-by":"crossref","unstructured":"Papandreou, G., Zhu, T., Chen, L.-C., Gidaris, S., Tompson, J., Murphy, K.: Personlab: Person pose estimation and instance segmentation with a bottom-up, part-based, geometric embedding model, pp. 269\u2013286 (2018)","DOI":"10.1007\/978-3-030-01264-9_17"},{"key":"3261_CR48","doi-asserted-by":"crossref","unstructured":"Li, K., Wang, S., Zhang, X., Xu, Y., Xu, W., Tu, Z.: Pose recognition with cascade transformers, pp. 1944\u20131953 (2021)","DOI":"10.1109\/CVPR46437.2021.00198"},{"key":"3261_CR49","unstructured":"Yuan, Y., Fu, R., Huang, L., Lin, W., Zhang, C., Chen, X., Wang, J.: Hrformer: High-resolution transformer for dense prediction (2021). arXiv preprint arXiv:2110.09408"},{"key":"3261_CR50","first-page":"38571","volume":"35","author":"Y Xu","year":"2022","unstructured":"Xu, Y., Zhang, J., Zhang, Q., Tao, D.: Vitpose: Simple vision transformer baselines for human pose estimation. Adv. Neural. Inf. Process. Syst. 35, 38571\u201338584 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"3261_CR51","doi-asserted-by":"crossref","unstructured":"Yang, S., Quan, Z., Nie, M., Yang, W.: Transpose: Keypoint localization via transformer, pp. 11802\u201311812 (2021)","DOI":"10.1109\/ICCV48922.2021.01159"},{"key":"3261_CR52","doi-asserted-by":"crossref","unstructured":"Huang, L., Tan, J., Liu, J., Yuan, J.: Hand-transformer: non-autoregressive structured modeling for 3D hand pose estimation, pp. 17\u201333 (2020)","DOI":"10.1007\/978-3-030-58595-2_2"},{"key":"3261_CR53","doi-asserted-by":"crossref","unstructured":"Lin, K., Wang, L., Liu, Z.: End-to-end human pose and mesh reconstruction with transformers, pp. 1954\u20131963 (2021)","DOI":"10.1109\/CVPR46437.2021.00199"},{"key":"3261_CR54","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhang, S., Wang, Z., Yang, S., Yang, W., Xia, S.-T., Zhou, E.: Tokenpose: Learning keypoint tokens for human pose estimation, pp. 11313\u201311322 (2021)","DOI":"10.1109\/ICCV48922.2021.01112"},{"key":"3261_CR55","doi-asserted-by":"crossref","unstructured":"Li, K., Wang, Y., Zhang, J., Gao, P., Song, G., Liu, Y., Li, H., Qiao, Y.: Uniformer: Unifying convolution and self-attention for visual recognition. IEEE Trans. Pattern Anal. Mach. Intell. (2023)","DOI":"10.1109\/TPAMI.2023.3282631"},{"key":"3261_CR56","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: common objects in context, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"3261_CR57","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Pishchulin, L., Gehler, P., Schiele, B.: 2D human pose estimation: new benchmark and state of the art analysis, pp. 3686\u20133693 (2014)","DOI":"10.1109\/CVPR.2014.471"},{"key":"3261_CR58","doi-asserted-by":"crossref","unstructured":"Zhang, F., Zhu, X., Dai, H., Ye, M., Zhu, C.: Distribution-aware coordinate representation for human pose estimation, pp. 7093\u20137102 (2020)","DOI":"10.1109\/CVPR42600.2020.00712"},{"key":"3261_CR59","unstructured":"Stoffl, L., Vidal, M., Mathis, A.: End-to-end trainable multi-instance pose estimation with transformers (2021). arXiv preprint arXiv:2103.12115"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-024-03261-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-024-03261-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-024-03261-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T20:06:36Z","timestamp":1732219596000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-024-03261-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6,13]]},"references-count":59,"journal-issue":{"issue":"8-9","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["3261"],"URL":"https:\/\/doi.org\/10.1007\/s11760-024-03261-7","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"type":"print","value":"1863-1703"},{"type":"electronic","value":"1863-1711"}],"subject":[],"published":{"date-parts":[[2024,6,13]]},"assertion":[{"value":"17 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 May 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 June 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors have no Conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Ethical approval is not applicable to this article, as it does not involve research with humans or animals.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}]}}