{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T07:52:32Z","timestamp":1782978752345,"version":"3.54.5"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,1,24]],"date-time":"2025-01-24T00:00:00Z","timestamp":1737676800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,24]],"date-time":"2025-01-24T00:00:00Z","timestamp":1737676800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100007129","name":"Natural Science Foundation of Shandong Province","doi-asserted-by":"publisher","award":["ZR2022MF260"],"award-info":[{"award-number":["ZR2022MF260"]}],"id":[{"id":"10.13039\/501100007129","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61673396"],"award-info":[{"award-number":["61673396"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-025-06923-6","type":"journal-article","created":{"date-parts":[[2025,1,24]],"date-time":"2025-01-24T05:09:09Z","timestamp":1737695349000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["MAQT: multi-scale attention and query-optimized transformer for end-to-end pose estimation"],"prefix":"10.1007","volume":"81","author":[{"given":"Hong","family":"Liang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cuiping","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingwen","family":"Shao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qian","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,1,24]]},"reference":[{"key":"6923_CR1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_29","volume-title":"Stacked hourglass networks for human pose estimation","author":"A Newell","year":"2016","unstructured":"Newell A, Yang K, Deng J (2016) Stacked hourglass networks for human pose estimation. Springer International Publishing, Cham"},{"issue":"10","key":"6923_CR2","doi-asserted-by":"publisher","first-page":"3349","DOI":"10.1109\/TPAMI.2020.2983686","volume":"43","author":"J Wang","year":"2020","unstructured":"Wang J, Sun K, Cheng T, Jiang B, Deng C, Zhao Y, Liu D, Mu Y, Tan M, Wang X et al (2020) Deep high-resolution representation learning for visual recognition. IEEE Trans Pattern Anal Mach Intell 43(10):3349\u20133364","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6923_CR3","doi-asserted-by":"crossref","unstructured":"Xiao B, Wu H, Wei Y(2018) Simple baselines for human pose estimation and tracking. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 466\u2013481","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"6923_CR4","doi-asserted-by":"crossref","unstructured":"Chen Y, Wang Z, Peng Y, Zhang Z, Yu G, Sun J (2018) Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7103\u20137112","DOI":"10.1109\/CVPR.2018.00742"},{"key":"6923_CR5","doi-asserted-by":"crossref","unstructured":"Sun K, Xiao B, Liu D, Wang J (2019) Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5693\u20135703","DOI":"10.1109\/CVPR.2019.00584"},{"key":"6923_CR6","doi-asserted-by":"crossref","unstructured":"McNally W, Vats K, Wong A, McPhee J (2022) Rethinking keypoint representations: Modeling keypoints and poses as objects for multi-person human pose estimation. In: European Conference on Computer Vision, pp. 37\u201354. Springer","DOI":"10.1007\/978-3-031-20068-7_3"},{"key":"6923_CR7","doi-asserted-by":"crossref","unstructured":"Papandreou G, Zhu T, Chen L-C, Gidaris S, Tompson J, Murphy K (2018) Personlab: Person pose estimation and instance segmentation with a bottom-up, part-based, geometric embedding model. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 269\u2013286","DOI":"10.1007\/978-3-030-01264-9_17"},{"key":"6923_CR8","doi-asserted-by":"crossref","unstructured":"Kreiss S, Bertoni L, Alahi A (2019) Pifpaf: Composite fields for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11977\u201311986","DOI":"10.1109\/CVPR.2019.01225"},{"key":"6923_CR9","doi-asserted-by":"crossref","unstructured":"Geng Z, Sun K, Xiao B, Zhang Z, Wang J (2021) Bottom-up human pose estimation via disentangled keypoint regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14676\u201314686","DOI":"10.1109\/CVPR46437.2021.01444"},{"key":"6923_CR10","doi-asserted-by":"crossref","unstructured":"Cheng B, Xiao B, Wang J, Shi H, Huang TS, Zhang L (2020) Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5386\u20135395","DOI":"10.1109\/CVPR42600.2020.00543"},{"key":"6923_CR11","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"6923_CR12","doi-asserted-by":"crossref","unstructured":"Zhao Y, Lv W, Xu S, Wei J, Wang G, Dang Q, Liu Y, Chen J (2024) Detrs beat yolos on real-time object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16965\u201316974","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"6923_CR13","doi-asserted-by":"crossref","unstructured":"Shi D, Wei X, Li L, Ren Y, Tan W (2022) End-to-end multi-person pose estimation with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11069\u201311078","DOI":"10.1109\/CVPR52688.2022.01079"},{"key":"6923_CR14","unstructured":"Yang J, Zeng A, Liu S, Li F, Zhang R, Zhang L(2023) Explicit box detection unifies end-to-end multi-person pose estimation. arXiv preprint arXiv:2302.01593"},{"key":"6923_CR15","doi-asserted-by":"crossref","unstructured":"Liu H, Chen Q, Tan Z, Liu J-J, Wang J, Su X, Li X, Yao K, Han J, Ding E, et al.: Group pose: A simple baseline for end-to-end multi-person pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15029\u201315038 (2023)","DOI":"10.1109\/ICCV51070.2023.01380"},{"key":"6923_CR16","unstructured":"Zhu X, Su W, Lu L, Li B, Wang X, Dai J (2020) Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159"},{"key":"6923_CR17","unstructured":"Zhang H, Li F, Liu S, Zhang L, Su H, Zhu J, Ni LM, Shum H-Y (2022) Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605"},{"key":"6923_CR18","doi-asserted-by":"crossref","unstructured":"Li K, Wang S, Zhang X, Xu Y, Xu W, Tu Z Pose recognition with cascade transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1944\u20131953 (2021)","DOI":"10.1109\/CVPR46437.2021.00198"},{"key":"6923_CR19","unstructured":"Yuan Y, Fu R, Huang L, Lin W, Zhang C, Chen X, Wang J Hrformer: High-resolution transformer for dense prediction. arXiv preprint arXiv:2110.09408 (2021)"},{"key":"6923_CR20","doi-asserted-by":"crossref","unstructured":"Li Y, Zhang S, Wang Z, Yang S, Yang W, Xia S-T, Zhou E Tokenpose: Learning keypoint tokens for human pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11313\u201311322 (2021)","DOI":"10.1109\/ICCV48922.2021.01112"},{"key":"6923_CR21","doi-asserted-by":"crossref","unstructured":"Yang S, Quan Z, Nie M, Yang W (2021) Transpose: Keypoint localization via transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11802\u201311812","DOI":"10.1109\/ICCV48922.2021.01159"},{"key":"6923_CR22","doi-asserted-by":"crossref","unstructured":"Ye S, Zhang Y, Hu J, Cao L, Zhang S, Shen L, Wang J, Ding S, Ji R Distilpose: Tokenized pose regression with heatmap distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2163\u20132172 (2023)","DOI":"10.1109\/CVPR52729.2023.00215"},{"issue":"10","key":"6923_CR23","first-page":"101819","volume":"35","author":"H Cheng","year":"2023","unstructured":"Cheng H, Wang J, Zhao A, Zhong Y, Li J, Dong L (2023) Joint graph convolution networks and transformer for human pose estimation in sports technique analysis. J King Saud Univ Comput Inform Sci 35(10):101819","journal-title":"J King Saud Univ Comput Inform Sci"},{"key":"6923_CR24","unstructured":"Chen Q, Chen X, Zeng G, Wang J Group detr: Fast training convergence with decoupled one-to-many label assignment. arXiv preprint arXiv:2207.13085 2(3), 12 (2022)"},{"key":"6923_CR25","unstructured":"Liu S, Li F, Zhang H, Yang X, Qi X, Su H, Zhu J, Zhang L Dab-detr: Dynamic anchor boxes are better queries for detr. arXiv preprint arXiv:2201.12329 (2022)"},{"key":"6923_CR26","doi-asserted-by":"crossref","unstructured":"Li F, Zhang H, Liu S, Guo J, Ni LM, Zhang L Dn-detr: Accelerate detr training by introducing query denoising. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13619\u201313627 (2022)","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"6923_CR27","first-page":"12464","volume":"35","author":"Y Xiao","year":"2022","unstructured":"Xiao Y, Su K, Wang X, Yu D, Jin L, He M, Yuan Z (2022) Querypose: sparse multi-person pose regression via spatial-aware part-level query. Adv Neural Inf Process Syst 35:12464\u201312477","journal-title":"Adv Neural Inf Process Syst"},{"key":"6923_CR28","doi-asserted-by":"crossref","unstructured":"Meng D, Chen X, Fan Z, Zeng G, Li H, Yuan Y, Sun L, Wang J Conditional detr for fast training convergence. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3651\u20133660 (2021)","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"6923_CR29","doi-asserted-by":"crossref","unstructured":"Nie X, Feng J, Zuo Y, Yan S Human pose estimation with parsing induced learner. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2100\u20132108 (2018)","DOI":"10.1109\/CVPR.2018.00224"},{"key":"6923_CR30","doi-asserted-by":"crossref","unstructured":"Peng X, Tang Z, Yang F, Feris RS, Metaxas D (2018) Jointly optimize data augmentation and network training: Adversarial data augmentation in human pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2226\u20132234","DOI":"10.1109\/CVPR.2018.00237"},{"key":"6923_CR31","doi-asserted-by":"crossref","unstructured":"Sun K, Lan C, Xing J, Zeng W, Liu D, Wang J Human pose estimation using global and local normalization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5599\u20135607 (2017)","DOI":"10.1109\/ICCV.2017.597"},{"key":"6923_CR32","doi-asserted-by":"crossref","unstructured":"Shi D, Wei X, Yu X, Tan W, Ren Y, Pu S Inspose: instance-aware networks for single-stage multi-person pose estimation. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 3079\u20133087 (2021)","DOI":"10.1145\/3474085.3475447"},{"issue":"1","key":"6923_CR33","doi-asserted-by":"publisher","first-page":"7608","DOI":"10.1038\/s41598-024-58175-8","volume":"14","author":"R Li","year":"2024","unstructured":"Li R, Li Q, Yang S, Zeng X, Yan A (2024) An efficient and accurate 2d human pose estimation method using vttranspose network. Sci Rep 14(1):7608","journal-title":"Sci Rep"},{"key":"6923_CR34","unstructured":"Vaswani, A.: Attention is all you need. Advances in Neural Information Processing Systems (2017)"},{"key":"6923_CR35","doi-asserted-by":"crossref","unstructured":"Tan M, Pang R, Le QV Efficientdet: Scalable and efficient object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10781\u201310790 (2020)","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"6923_CR36","doi-asserted-by":"crossref","unstructured":"Lin T-Y, Maire M, Belongie S, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pp. 740\u2013755 (2014). Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"6923_CR37","doi-asserted-by":"crossref","unstructured":"Li J, Wang C, Zhu H, Mao Y, Fang H-S, Lu C Crowdpose: Efficient crowded scenes pose estimation and a new benchmark. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10863\u201310872 (2019)","DOI":"10.1109\/CVPR.2019.01112"},{"key":"6923_CR38","unstructured":"Paszke A, Gross S, Chintala S, Chanan G, Yang E, DeVito Z, Lin Z, Desmaison A, Antiga L, Lerer A Automatic differentiation in pytorch (2017)"},{"key":"6923_CR39","doi-asserted-by":"crossref","unstructured":"He K, Gkioxari G, Doll\u00e1r P, Girshick R (2017) Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969","DOI":"10.1109\/ICCV.2017.322"},{"key":"6923_CR40","doi-asserted-by":"crossref","unstructured":"Mao W, Ge Y, Shen C, Tian Z, Wang X, Wang Z, Hengel Av Poseur: Direct human pose regression with transformers. In: European Conference on Computer Vision, pp. 72\u201388 (2022). Springer","DOI":"10.1007\/978-3-031-20068-7_5"},{"key":"6923_CR41","doi-asserted-by":"crossref","unstructured":"Geng Z, Sun K, Xiao B, Zhang Z, Wang J Bottom-up human pose estimation via disentangled keypoint regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14676\u201314686 (2021)","DOI":"10.1109\/CVPR46437.2021.01444"},{"key":"6923_CR42","doi-asserted-by":"crossref","unstructured":"Xue N, Wu T, Xia G-S, Zhang L Learning local-global contextual adaptation for multi-person pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13065\u201313074 (2022)","DOI":"10.1109\/CVPR52688.2022.01272"},{"key":"6923_CR43","unstructured":"Tian Z, Chen H, Shen C Directpose: Direct end-to-end multi-person pose estimation. arXiv preprint arXiv:1911.07451 (2019)"},{"key":"6923_CR44","doi-asserted-by":"crossref","unstructured":"Mao W, Tian Z, Wang X, Shen C (2021) Fcpose: Fully convolutional multi-person pose estimation with dynamic instance-aware convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9034\u20139043","DOI":"10.1109\/CVPR46437.2021.00892"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-06923-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-025-06923-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-025-06923-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,24]],"date-time":"2025-01-24T05:09:42Z","timestamp":1737695382000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-025-06923-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,1,24]]},"references-count":44,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2025,1]]}},"alternative-id":["6923"],"URL":"https:\/\/doi.org\/10.1007\/s11227-025-06923-6","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,1,24]]},"assertion":[{"value":"7 January 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 January 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"429"}}