{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T14:57:48Z","timestamp":1782313068335,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the Fundamental Research Funds for the Central Universities"},{"name":"Young Elite Scientist Sponsorship Program of China Association for Science and Technology","award":["YESS20200140"],"award-info":[{"award-number":["YESS20200140"]}]},{"name":"National Nature Fund","award":["No.62006244,No.62102039"],"award-info":[{"award-number":["No.62006244,No.62102039"]}]},{"name":"Natural Science Foundation of China","award":["No.62222606"],"award-info":[{"award-number":["No.62222606"]}]},{"name":"Young Elite Scientist Sponsorship Program of Beijing Association for Science and Technology","award":["BYESS2021178"],"award-info":[{"award-number":["BYESS2021178"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3611989","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"1798-1808","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["DecenterNet: Bottom-Up Human Pose Estimation Via Decentralized Pose Representation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4740-6932","authenticated-orcid":false,"given":"Tao","family":"Wang","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4855-2464","authenticated-orcid":false,"given":"Lei","family":"Jin","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7670-4432","authenticated-orcid":false,"given":"Zhang","family":"Wang","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2195-0043","authenticated-orcid":false,"given":"Xiaojin","family":"Fan","sequence":"additional","affiliation":[{"name":"Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9830-0081","authenticated-orcid":false,"given":"Yu","family":"Cheng","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7170-4764","authenticated-orcid":false,"given":"Yinglei","family":"Teng","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6801-0510","authenticated-orcid":false,"given":"Junliang","family":"Xing","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3508-756X","authenticated-orcid":false,"given":"Jian","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of North Electronic Equipment, Intelligent Game and Decision Laboratory, &amp; Peng Cheng Laboratory, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40318-019-00155-6"},{"key":"e_1_3_2_2_2_1","volume-title":"Brian C Van Esesn, Abdul A S Awwal, and Vijayan K Asari.","author":"Alom Md Zahangir","year":"2018","unstructured":"Md Zahangir Alom, Tarek M Taha, Christopher Yakopcic, Stefan Westberg, Paheding Sidike, Mst Shamima Nasrin, Brian C Van Esesn, Abdul A S Awwal, and Vijayan K Asari. 2018. The history began from alexnet: A comprehensive survey on deep learning approaches. arXiv preprint arXiv:1803.01164 (2018)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Mykhaylo Andriluka Leonid Pishchulin Peter Gehler and Bernt Schiele. 2014. 2d human pose estimation: New benchmark and state of the art analysis. In CVPR. 3686--3693.","DOI":"10.1109\/CVPR.2014.471"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Guillem Bras\u00f3 Nikita Kister and Laura Leal-Taix\u00e9. 2021. The center of attention: Center-keypoint grouping via attention for multi-person pose estimation. In ICCV. 11853--11863.","DOI":"10.1109\/ICCV48922.2021.01164"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Yilun Chen Zhicheng Wang Yuxiang Peng Zhiqiang Zhang Gang Yu and Jian Sun. 2018. Cascaded pyramid network for multi-person pose estimation. In CVPR. 7103--7112.","DOI":"10.1109\/CVPR.2018.00742"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Bowen Cheng Ishan Misra Alexander G Schwing Alexander Kirillov and Rohit Girdhar. 2022. Masked-attention mask transformer for universal image segmentation. In CVPR. 1290--1299.","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"e_1_3_2_2_7_1","volume-title":"Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation. In CVPR. 5386--5395.","author":"Cheng Bowen","year":"2020","unstructured":"Bowen Cheng, Bin Xiao, Jingdong Wang, Honghui Shi, Thomas S Huang, and Lei Zhang. 2020. Higherhrnet: Scale-aware representation learning for bottom-up human pose estimation. In CVPR. 5386--5395."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"crossref","unstructured":"Yu Cheng Bo Wang Bo Yang and Robby T Tan. 2021. Monocular 3D multi-person pose estimation by integrating top-down and bottom-up networks. In CVPR. 7649--7659.","DOI":"10.1109\/CVPR46437.2021.00756"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Xiaochuan Fan Kang Zheng Yuewei Lin and Song Wang. 2015. Combining local appearance and holistic view: Dual-source deep neural networks for human pose estimation. In CVPR. 1347--1355.","DOI":"10.1109\/CVPR.2015.7298740"},{"key":"e_1_3_2_2_10_1","volume-title":"Alphapose: Whole-body regional multi-person pose estimation and tracking in real-time","author":"Fang Hao-Shu","year":"2022","unstructured":"Hao-Shu Fang, Jiefeng Li, Hongyang Tang, Chao Xu, Haoyi Zhu, Yuliang Xiu, Yong-Lu Li, and Cewu Lu. 2022. Alphapose: Whole-body regional multi-person pose estimation and tracking in real-time. IEEE TPAMI (2022)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Zigang Geng Ke Sun Bin Xiao Zhaoxiang Zhang and Jingdong Wang. 2021. Bottom-up human pose estimation via disentangled keypoint regression. In CVPR. 14676--14686.","DOI":"10.1109\/CVPR46437.2021.01444"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Ross Girshick Jeff Donahue Trevor Darrell and Jitendra Malik. 2014. Rich feature hierarchies for accurate object detection and semantic segmentation. In CVPR. 580--587.","DOI":"10.1109\/CVPR.2014.81"},{"key":"e_1_3_2_2_13_1","unstructured":"Kaiming He Georgia Gkioxari Piotr Doll\u00e1r and Ross Girshick. 2017. Mask r-cnn. In ICCV. 2961--2969."},{"key":"e_1_3_2_2_14_1","volume-title":"Anti-UAV: A large multi-modal benchmark for UAV tracking. arXiv preprint arXiv:2101.08466","author":"Jiang Nan","year":"2021","unstructured":"Nan Jiang, Kuiran Wang, Xiaoke Peng, Xuehui Yu, Qiang Wang, Junliang Xing, Guorong Li, Jian Zhao, Guodong Guo, and Zhenjun Han. 2021. Anti-UAV: A large multi-modal benchmark for UAV tracking. arXiv preprint arXiv:2101.08466 (2021)."},{"key":"e_1_3_2_2_15_1","volume-title":"Grouping by center: Predicting centripetal offsets for the bottom-up human pose estimation","author":"Jin Lei","year":"2022","unstructured":"Lei Jin, Xiaojuan Wang, Xuecheng Nie, Luoqi Liu, Yandong Guo, and Jian Zhao. 2022. Grouping by center: Predicting centripetal offsets for the bottom-up human pose estimation. IEEE TMM (2022)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3282139"},{"key":"e_1_3_2_2_17_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_2_18_1","volume-title":"Pifpaf: Composite fields for human pose estimation. In CVPR. 11977--11986.","author":"Kreiss Sven","year":"2019","unstructured":"Sven Kreiss, Lorenzo Bertoni, and Alexandre Alahi. 2019. Pifpaf: Composite fields for human pose estimation. In CVPR. 11977--11986."},{"key":"e_1_3_2_2_19_1","volume-title":"Single-Stage is Enough: Multi-Person Absolute 3D Pose Estimation. CVPR","author":"Lei Jin","year":"2022","unstructured":"Jin Lei, Chenyang Xu, Xiaojuan Wang, Yabo Xiao, Yandong Guo, Xuecheng Nie, and Jian Zhao. 2022. Single-Stage is Enough: Multi-Person Absolute 3D Pose Estimation. CVPR (2022)."},{"key":"e_1_3_2_2_20_1","volume-title":"Crowdpose: Efficient crowded scenes pose estimation and a new benchmark. In CVPR. 10863--10872.","author":"Li Jiefeng","year":"2019","unstructured":"Jiefeng Li, Can Wang, Hao Zhu, Yihuan Mao, Hao-Shu Fang, and Cewu Lu. 2019. Crowdpose: Efficient crowded scenes pose estimation and a new benchmark. In CVPR. 10863--10872."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"crossref","unstructured":"Qun Li Ziyi Zhang Fu Xiao Feng Zhang and Bir Bhanu. 2022. Dite-HRNet: Dynamic Lightweight High-Resolution Network for Human Pose Estimation. In IJCAI. 1095--1101.","DOI":"10.24963\/ijcai.2022\/153"},{"key":"e_1_3_2_2_22_1","unstructured":"Qun Li Ziyi Zhang Feng Zhang and Fu Xiao. [n. d.]. HRNeXt: High-Resolution Context Network for Crowd Pose Estimation. ([n. d.])."},{"key":"e_1_3_2_2_23_1","volume-title":"Resnet with one-neuron hidden layers is a universal approximator. NIPS 31","author":"Lin Hongzhou","year":"2018","unstructured":"Hongzhou Lin and Stefanie Jegelka. 2018. Resnet with one-neuron hidden layers is a universal approximator. NIPS 31 (2018)."},{"key":"e_1_3_2_2_24_1","volume-title":"Microsoft coco: Common objects in context","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In ECCV. Springer, 740--755."},{"key":"e_1_3_2_2_25_1","volume-title":"Fcpose: Fully convolutional multi-person pose estimation with dynamic instance-aware convolutions. In CVPR. 9034--9043.","author":"Mao Weian","year":"2021","unstructured":"Weian Mao, Zhi Tian, Xinlong Wang, and Chunhua Shen. 2021. Fcpose: Fully convolutional multi-person pose estimation with dynamic instance-aware convolutions. In CVPR. 9034--9043."},{"key":"e_1_3_2_2_26_1","unstructured":"Paulius Micikevicius Sharan Narang Jonah Alben Gregory Diamos Erich Elsen David Garcia Boris Ginsburg Michael Houston Oleksii Kuchaiev Ganesh Venkatesh et al. 2017. Mixed precision training. arXiv preprint arXiv:1710.03740 (2017)."},{"key":"e_1_3_2_2_27_1","volume-title":"Associative embedding: End-to-end learning for joint detection and grouping. NeuIPS 30","author":"Newell Alejandro","year":"2017","unstructured":"Alejandro Newell, Zhiao Huang, and Jia Deng. 2017. Associative embedding: End-to-end learning for joint detection and grouping. NeuIPS 30 (2017)."},{"key":"e_1_3_2_2_28_1","unstructured":"Xuecheng Nie Jiashi Feng Junliang Xing and Shuicheng Yan. 2018. Pose partition networks for multi-person pose estimation. In ECCV. 684--699."},{"key":"e_1_3_2_2_29_1","unstructured":"Xuecheng Nie Jiashi Feng Jianfeng Zhang and Shuicheng Yan. 2019. Single-stage multi-person pose machines. In ICCV. 6951--6960."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"George Papandreou Tyler Zhu Nori Kanazawa Alexander Toshev Jonathan Tompson Chris Bregler and Kevin Murphy. 2017. Towards accurate multi-person pose estimation in the wild. In CVPR. 4903--4911.","DOI":"10.1109\/CVPR.2017.395"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Leonid Pishchulin Mykhaylo Andriluka Peter Gehler and Bernt Schiele. 2013. Poselet conditioned pictorial structures. In CVPR. 588--595.","DOI":"10.1109\/CVPR.2013.82"},{"key":"e_1_3_2_2_32_1","volume-title":"Peeking into occluded joints: A novel framework for crowd pose estimation","author":"Qiu Lingteng","unstructured":"Lingteng Qiu, Xuanye Zhang, Yanran Li, Guanbin Li, Xiaojun Wu, Zixiang Xiong, Xiaoguang Han, and Shuguang Cui. 2020. Peeking into occluded joints: A novel framework for crowd pose estimation. In ECCV. Springer, 488--504."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Dahu Shi Xing Wei Liangqi Li Ye Ren and Wenming Tan. 2022. End-to-end multi-person pose estimation with transformers. In CVPR. 11069--11078.","DOI":"10.1109\/CVPR52688.2022.01079"},{"key":"e_1_3_2_2_34_1","volume-title":"Caner Sahin, and Tae-Kyun Kim.","author":"Sock Juil","year":"2018","unstructured":"Juil Sock, Kwang In Kim, Caner Sahin, and Tae-Kyun Kim. 2018. Multi-task deep networks for depth-based 6d object pose and joint registration in crowd scenarios. arXiv preprint arXiv:1806.03891 (2018)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Ke Sun Cuiling Lan Junliang Xing Wenjun Zeng Dong Liu and Jingdong Wang. 2017. Human pose estimation using global and local normalization. In ICCV. 5599--5607.","DOI":"10.1109\/ICCV.2017.597"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","unstructured":"Ke Sun Bin Xiao Dong Liu and Jingdong Wang. 2019. Deep high-resolution representation learning for human pose estimation. In CVPR. 5693--5703.","DOI":"10.1109\/CVPR.2019.00584"},{"key":"e_1_3_2_2_37_1","first-page":"6278","article-title":"Robust Pose Estimation in Crowded Scenes with Direct Pose-Level Inference","volume":"34","author":"Wang Dongkai","year":"2021","unstructured":"Dongkai Wang, Shiliang Zhang, and Gang Hua. 2021. Robust Pose Estimation in Crowded Scenes with Direct Pose-Level Inference. NeuIPS 34 (2021), 6278--6289.","journal-title":"NeuIPS"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547780"},{"key":"e_1_3_2_2_39_1","volume-title":"evolve: A high-performance face recognition library. arXiv preprint arXiv:2107.08621","author":"Wang Qingzhong","year":"2021","unstructured":"Qingzhong Wang, Pengfei Zhang, Haoyi Xiong, and Jian Zhao. 2021. Face. evolve: A high-performance face recognition library. arXiv preprint arXiv:2107.08621 (2021)."},{"key":"e_1_3_2_2_40_1","volume-title":"Cvt: Introducing convolutions to vision transformers. In ICCV. 22--31.","author":"Wu Haiping","year":"2021","unstructured":"Haiping Wu, Bin Xiao, Noel Codella, Mengchen Liu, Xiyang Dai, Lu Yuan, and Lei Zhang. 2021. Cvt: Introducing convolutions to vision transformers. In ICCV. 22--31."},{"key":"e_1_3_2_2_41_1","unstructured":"J. Wu H. Zheng B. Zhao Y. Li B. Yan R. Liang W. Wang S. Zhou G. Lin and Y. Fu. 2017. AI Challenger: A Large-scale Dataset for Going Deeper in Image Understanding. In ICME."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Bin Xiao Haiping Wu and Yichen Wei. 2018. Simple baselines for human pose estimation and tracking. In ECCV. 466--481.","DOI":"10.1007\/978-3-030-01231-1_29"},{"key":"e_1_3_2_2_43_1","unstructured":"Yabo Xiao Kai Su Xiaojuan Wang Dongdong Yu Lei Jin Mingshu He and Zehuan Yuan. 2022. QueryPose: Sparse Multi-Person Pose Regression via Spatial-Aware Part-Level Query. (2022)."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20185"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Lumin Xu Ruihan Xu and Sheng Jin. 2020. Hieve acm mm grand challenge 2020: Pose tracking in crowded scenes. In ACMMM. 4689--4693.","DOI":"10.1145\/3394171.3416295"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Nan Xue Tianfu Wu Gui-Song Xia and Liangpei Zhang. 2022. Learning Local-Global Contextual Adaptation for Multi-Person Pose Estimation. In CVPR. 13065--13074.","DOI":"10.1109\/CVPR52688.2022.01272"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Yi Yang and Deva Ramanan. 2011. Articulated pose estimation with flexible mixtures-of-parts. In CVPR. 1385--1392.","DOI":"10.1109\/CVPR.2011.5995741"},{"key":"e_1_3_2_2_48_1","volume-title":"Lite-hrnet: A lightweight high-resolution network. In CVPR. 10440--10450.","author":"Yu Changqian","year":"2021","unstructured":"Changqian Yu, Bin Xiao, Changxin Gao, Lu Yuan, Lei Zhang, Nong Sang, and Jingdong Wang. 2021. Lite-hrnet: A lightweight high-resolution network. In CVPR. 10440--10450."},{"key":"e_1_3_2_2_49_1","unstructured":"Fisher Yu Dequan Wang Evan Shelhamer and Trevor Darrell. 2018. Deep layer aggregation. In CVPR. 2403--2412."},{"key":"e_1_3_2_2_50_1","unstructured":"Christoph Zauner. 2010. Implementation and benchmarking of perceptual image hash functions. (2010)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019251"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01252-7"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"crossref","unstructured":"Jian Zhao12 Jianshu Li Fang Zhao Shuicheng Yan13 and Jiashi Feng. 2017. Marginalized CNN: Learning deep invariant representations. (2017).","DOI":"10.5244\/C.31.127"},{"key":"e_1_3_2_2_54_1","unstructured":"C. Zhe T. Simon S. E. Wei and Y. Sheikh. 2017. Realtime multi-person 2d pose estimation using part affinity fields. In CVPR."},{"key":"e_1_3_2_2_55_1","volume-title":"Objects as points. arXiv preprint arXiv:1904.07850","author":"Zhou Xingyi","year":"2019","unstructured":"Xingyi Zhou, Dequan Wang, and Philipp Kr\u00e4henb\u00fchl. 2019. Objects as points. arXiv preprint arXiv:1904.07850 (2019)."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611989","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3611989","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:12:54Z","timestamp":1755821574000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611989"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":55,"alternative-id":["10.1145\/3581783.3611989","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3611989","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}