{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T20:44:16Z","timestamp":1787690656752,"version":"build-2784847793"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612236","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"2353-2361","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":20,"title":["Lightweight Super-Resolution Head for Human Pose Estimation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7159-2432","authenticated-orcid":false,"given":"Haonan","family":"Wang","sequence":"first","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9297-7729","authenticated-orcid":false,"given":"Jie","family":"Liu","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6086-3559","authenticated-orcid":false,"given":"Jie","family":"Tang","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1391-1762","authenticated-orcid":false,"given":"Gangshan","family":"Wu","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.471"},{"key":"e_1_3_2_1_2_1","volume-title":"European Conference on Computer Vision. Springer, 455--472","author":"Yuanhao","unstructured":"Yuanhao Cai et al. 2020. Learning delicate local representations for multi-person pose estimation. In European Conference on Computer Vision. Springer, 455--472."},{"key":"e_1_3_2_1_3_1","volume-title":"Openpose: realtime multi-person 2d pose estimation using part affinity fields","author":"Cao Zhe","unstructured":"Zhe Cao, Gines Hidalgo, Tomas Simon, Shih-En Wei, and Yaser Sheikh. 2019. Openpose: realtime multi-person 2d pose estimation using part affinity fields. IEEE transactions on pattern analysis and machine intelligence, 43, 1, 172--186."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Haoming Chen Runyang Feng Sifan Wu Hao Xu Fengcheng Zhou and Zhenguang Liu. 2022. 2d human pose estimation: a survey. arXiv preprint arXiv:2204.07370.","DOI":"10.1007\/s00530-022-01019-0"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00742"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01267-0_28"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00543"},{"key":"e_1_3_2_1_8_1","unstructured":"MMPose Contributors. 2020. Openmmlab pose estimation toolbox and benchmark. https:\/\/github.com\/open-mmlab\/mmpose. (2020)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Nicola Garau Niccol\u00f2 Bisagno Piotr Br\u00f3dka and Nicola Conci. 2021. Deca: deep viewpoint-equivariant human pose estimation using capsule autoencoders. arXiv preprint arXiv:2108.08557.","DOI":"10.1109\/ICCV48922.2021.01147"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00574"},{"key":"e_1_3_2_1_12_1","volume-title":"3d convolutional neural networks for human action recognition","author":"Ji Shuiwang","unstructured":"Shuiwang Ji, Wei Xu, Ming Yang, and Kai Yu. 2012. 3d convolutional neural networks for human action recognition. IEEE transactions on pattern analysis and machine intelligence, 35, 1, 221--231."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01084"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01112"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00198"},{"key":"e_1_3_2_1_16_1","unstructured":"Wenbo Li et al. 2019. Rethinking on multi-stage networks for human pose estimation. arXiv preprint arXiv:1901.00148."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20068-7_6"},{"key":"e_1_3_2_1_18_1","unstructured":"Yanjie Li Shoukui Zhang Zhicheng Wang Sen Yang Wankou Yang Shu-Tao Xia and Erjin Zhou. 2021. Tokenpose: learning keypoint tokens for human pose estimation. arXiv preprint arXiv:2104.03516."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_21_1","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3241606"},{"key":"e_1_3_2_1_23_1","unstructured":"Weian Mao Yongtao Ge Chunhua Shen Zhi Tian Xinlong Wang and Zhibin Wang. 2021. Tfpose: direct human pose estimation with transformers. arXiv preprint arXiv:2103.15320."},{"key":"e_1_3_2_1_24_1","volume-title":"Tel Aviv, Israel","author":"Mao Weian","year":"2022","unstructured":"Weian Mao, Yongtao Ge, Chunhua Shen, Zhi Tian, Xinlong Wang, Zhibin Wang, and Anton van den Hengel. 2022. Poseur: direct human pose regression with transformers. In Computer Vision-ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23-27, 2022, Proceedings, Part VI. Springer, 72--88."},{"key":"e_1_3_2_1_25_1","unstructured":"Alejandro Newell Zhiao Huang and Jia Deng. 2017. Associative embedding: end-to-end learning for joint detection and grouping. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00705"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_17"},{"key":"e_1_3_2_1_28_1","unstructured":"Joseph Redmon and Ali Farhadi. 2018. Yolov3: an incremental improvement. arXiv preprint arXiv:1804.02767."},{"key":"e_1_3_2_1_29_1","unstructured":"Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: towards real-time object detection with region proposal networks. Advances in neural information processing systems 28."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.207"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00584"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.284"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_33"},{"key":"e_1_3_2_1_34_1","unstructured":"Zhi Tian Hao Chen and Chunhua Shen. 2019. Directpose: direct end-to-end multi-person pose estimation. arXiv preprint arXiv:1911.07451."},{"key":"e_1_3_2_1_35_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01110"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Tom Wehrbein Marco Rudolph Bodo Rosenhahn and Bastian Wandt. 2021. Probabilistic monocular 3d human pose estimation with normalizing flows. arXiv preprint arXiv:2107.13788.","DOI":"10.1109\/ICCV48922.2021.01101"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58607-2_31"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"e_1_3_2_1_40_1","unstructured":"Yufei Xu Jing Zhang Qiming Zhang and Dacheng Tao. 2022. Vitpose: simple vision transformer baselines for human pose estimation. arXiv preprint arXiv:2204.12484."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01159"},{"key":"e_1_3_2_1_42_1","first-page":"7281","article-title":"Hrformer: high-resolution vision transformer for dense predict","volume":"34","author":"Yuan Yuhui","year":"2021","unstructured":"Yuhui Yuan, Rao Fu, Lang Huang, Weihong Lin, Chao Zhang, Xilin Chen, and Jingdong Wang. 2021. Hrformer: high-resolution vision transformer for dense predict. Advances in Neural Information Processing Systems, 34, 7281--7293.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Ailing Zeng Xiao Sun Lei Yang Nanxuan Zhao Minhao Liu and Qiang Xu. 2021. Learning skeletal graph neural networks for hard 3d pose estimation. arXiv preprint arXiv:2108.07181.","DOI":"10.1109\/ICCV48922.2021.01124"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00712"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compeleceng.2021.107192"},{"key":"e_1_3_2_1_46_1","unstructured":"Xingyi Zhou Dequan Wang and Philipp Kr\u00e4henb\u00fchl. 2019. Objects as points. arXiv preprint arXiv:1904.07850."},{"key":"e_1_3_2_1_47_1","unstructured":"Shihao Zou Chuan Guo Xinxin Zuo Sen Wang Pengyu Wang Xiaoqin Hu Shoushun Chen Minglun Gong and Li Cheng. 2021. Eventhpe: event-based 3d human pose and shape estimation. arXiv preprint arXiv:2108.06819."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612236","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612236","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:56:49Z","timestamp":1755820609000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612236"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":47,"alternative-id":["10.1145\/3581783.3612236","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612236","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}