{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T00:01:41Z","timestamp":1784937701953,"version":"3.55.0"},"reference-count":57,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100014188","name":"Korea Ministry of Science and ICT","doi-asserted-by":"publisher","award":["RS-2025-02634821"],"award-info":[{"award-number":["RS-2025-02634821"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002701","name":"Korea Ministry of Education","doi-asserted-by":"publisher","award":["RS-2023-00240109"],"award-info":[{"award-number":["RS-2023-00240109"]}],"id":[{"id":"10.13039\/501100002701","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.neucom.2026.133857","type":"journal-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T15:06:50Z","timestamp":1777993610000},"page":"133857","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Mitigating token role drift under occlusion with no inference overhead for compositional-token pose estimation"],"prefix":"10.1016","volume":"693","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-0464-3185","authenticated-orcid":false,"given":"Huanheng","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Soon","family":"Kwon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5401-2459","authenticated-orcid":false,"given":"Sungho","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.133857_bib0005","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"11782","article-title":"TransPose: keypoint localization via transformer","author":"Yang","year":"2021"},{"key":"10.1016\/j.neucom.2026.133857_bib0010","series-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems, NeurIPS 2022","article-title":"ViTPose: simple vision transformer baselines for human pose estimation","author":"Xu","year":"2022"},{"key":"10.1016\/j.neucom.2026.133857_bib0015","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"11293","article-title":"TokenPose: learning keypoint tokens for human pose estimation","author":"Li","year":"2021"},{"key":"10.1016\/j.neucom.2026.133857_bib0020","author":"Wang"},{"key":"10.1016\/j.neucom.2026.133857_bib0025","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"660","article-title":"Human pose as compositional tokens","author":"Geng","year":"2023"},{"key":"10.1016\/j.neucom.2026.133857_bib0030","first-page":"1","article-title":"Multimodal feature interaction and High-Quality pseudolabel generation with Self-Training for cognitive state detection","volume":"PP","author":"Tong","year":"2025","journal-title":"IEEE Trans. Ind. Inform."},{"key":"10.1016\/j.neucom.2026.133857_bib0035","first-page":"1","article-title":"Neural rendering and Flow-Assisted unsupervised Multi-View stereo for Real-Time monocular tracking and scene perception","volume":"PP","author":"Tong","year":"2025","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"key":"10.1016\/j.neucom.2026.133857_bib0040","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2024.102399","article-title":"Large-Scale aerial scene perception based on Self-Supervised Multi-View stereo via cycled generative adversarial network","volume":"109","author":"Tong","year":"2024","journal-title":"Inf. Fusion"},{"issue":"7","key":"10.1016\/j.neucom.2026.133857_bib0045","doi-asserted-by":"crossref","first-page":"4646","DOI":"10.1007\/s11263-025-02409-3","article-title":"Multi-Text guidance is important: multi-modality image fusion via large generative Vision-Language model","volume":"133","author":"Wang","year":"2025","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.133857_bib0050","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"12637","article-title":"Highlight what you want: weakly-supervised instance-level controllable infrared-visible image fusion","author":"Wang","year":"2025"},{"issue":"10","key":"10.1016\/j.neucom.2026.133857_bib0055","doi-asserted-by":"crossref","first-page":"2529","DOI":"10.1007\/s11263-023-01806-w","article-title":"When multi-focus image fusion networks meet traditional Edge-Preservation technology","volume":"131","author":"Wang","year":"2023","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.133857_bib0060","doi-asserted-by":"crossref","DOI":"10.1016\/j.inffus.2025.104058","article-title":"SD-fuse: an image structure-driven model for multi-focus image fusion","volume":"129","author":"Wang","year":"2026","journal-title":"Inf. Fusion"},{"key":"10.1016\/j.neucom.2026.133857_bib0065","doi-asserted-by":"crossref","first-page":"1321","DOI":"10.1109\/TIP.2026.3654370","article-title":"Rethinking multi-focus image fusion: an input space optimization view","volume":"35","author":"Wang","year":"2026","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.neucom.2026.133857_bib0070","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.113022","article-title":"Infrared and visible image fusion via iterative feature decomposition and deep balanced fusion","volume":"174","author":"Li","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.neucom.2026.133857_bib0075","series-title":"Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","first-page":"5903","article-title":"Rethinking visibility in human pose estimation: occluded pose reasoning via transformers","author":"Sun","year":"2024"},{"key":"10.1016\/j.neucom.2026.133857_bib0080","series-title":"Proceedings of the British Machine Vision Conference (BMVC)","article-title":"Structured spatial reasoning for human pose estimation","author":"Huang","year":"2022"},{"issue":"6","key":"10.1016\/j.neucom.2026.133857_bib0085","doi-asserted-by":"crossref","first-page":"663","DOI":"10.26599\/TST.2018.9010100","article-title":"Deep learning based 2D human pose estimation: a survey","volume":"24","author":"Dang","year":"2019","journal-title":"Tsinghua Sci. Technol."},{"key":"10.1016\/j.neucom.2026.133857_bib0090","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2019.102897","article-title":"Monocular human pose estimation: a survey of deep learning-based methods","volume":"192","author":"Chen","year":"2020","journal-title":"Comput. Vis. Image Underst."},{"issue":"1","key":"10.1016\/j.neucom.2026.133857_bib0095","doi-asserted-by":"crossref","DOI":"10.1145\/3603618","article-title":"Deep learning-based human pose estimation: a survey","volume":"56","author":"Zheng","year":"2023","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.neucom.2026.133857_bib0100","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"4724","article-title":"Convolutional pose machines","author":"Wei","year":"2016"},{"key":"10.1016\/j.neucom.2026.133857_bib0105","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","article-title":"Stacked hourglass networks for human pose estimation","author":"Newell","year":"2016"},{"key":"10.1016\/j.neucom.2026.133857_bib0110","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"7103","article-title":"Cascaded pyramid network for multi-person pose estimation","author":"Chen","year":"2018"},{"key":"10.1016\/j.neucom.2026.133857_bib0115","author":"Xiao"},{"key":"10.1016\/j.neucom.2026.133857_bib0120","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5686","article-title":"Deep high-resolution representation learning for human pose estimation","author":"Sun","year":"2019"},{"key":"10.1016\/j.neucom.2026.133857_bib0125","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"1653","article-title":"DeepPose: human pose estimation via deep neural networks","author":"Toshev","year":"2014"},{"key":"10.1016\/j.neucom.2026.133857_bib0130","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","article-title":"Integral human pose regression","author":"Sun","year":"2018"},{"key":"10.1016\/j.neucom.2026.133857_bib0135","author":"Nibali"},{"key":"10.1016\/j.neucom.2026.133857_bib0140","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"7091","article-title":"Distribution-Aware coordinate representation for human pose estimation","author":"Zhang","year":"2020"},{"issue":"1","key":"10.1016\/j.neucom.2026.133857_bib0145","doi-asserted-by":"crossref","first-page":"172","DOI":"10.1109\/TPAMI.2019.2929257","article-title":"OpenPose: realtime multi-person 2D pose estimation using part affinity fields","volume":"43","author":"Cao","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.133857_bib0150","series-title":"Advances in Neural Information Processing Systems","article-title":"Associative embedding: end-to-end learning for joint detection and grouping","author":"Newell","year":"2017"},{"key":"10.1016\/j.neucom.2026.133857_bib0155","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","article-title":"PersonLab: person pose estimation and instance segmentation with a Bottom-Up, Part-Based, geometric embedding model","author":"Papandreou","year":"2018"},{"key":"10.1016\/j.neucom.2026.133857_bib0160","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5385","article-title":"HigherHRNet: scale-aware representation learning for bottom-up human pose estimation","author":"Cheng","year":"2020"},{"key":"10.1016\/j.neucom.2026.133857_bib0165","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"11969","article-title":"PifPaf: composite fields for human pose estimation","author":"Kreiss","year":"2019"},{"key":"10.1016\/j.neucom.2026.133857_bib0170","series-title":"Proceedings of the IEEE International Conference on Computer Vision (ICCV)","first-page":"2353","article-title":"RMPE: regional multi-person pose estimation","author":"Fang","year":"2017"},{"key":"10.1016\/j.neucom.2026.133857_bib0175","series-title":"Computer Vision \u2013 ECCV 2014","first-page":"740","article-title":"Microsoft COCO: common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.neucom.2026.133857_bib0180","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"3686","article-title":"2D human pose estimation: new benchmark and state of the art analysis","author":"Andriluka","year":"2014"},{"key":"10.1016\/j.neucom.2026.133857_bib0185","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","article-title":"CrowdPose: efficient crowded scenes pose estimation and a new benchmark","author":"Li","year":"2019"},{"key":"10.1016\/j.neucom.2026.133857_bib0190","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"11850","article-title":"When human pose estimation meets robustness: adversarial algorithms and benchmarks","author":"Wang","year":"2021"},{"key":"10.1016\/j.neucom.2026.133857_bib0195","author":"Ma"},{"key":"10.1016\/j.neucom.2026.133857_bib0200","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5167","article-title":"PoseTrack: a benchmark for human pose estimation and tracking","author":"Andriluka","year":"2018"},{"key":"10.1016\/j.neucom.2026.133857_bib0205","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","article-title":"Spatial temporal graph convolutional networks for skeleton-based action recognition","author":"Yan","year":"2018"},{"key":"10.1016\/j.neucom.2026.133857_bib0210","doi-asserted-by":"crossref","first-page":"269","DOI":"10.1016\/j.neucom.2016.09.033","article-title":"Video pose estimation with global motion cues","volume":"219","author":"Shi","year":"2017","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.133857_bib0215","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2020.107258","article-title":"Exploring temporal consistency for human pose estimation in videos","volume":"103","author":"Li","year":"2020","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.neucom.2026.133857_bib0220","series-title":"advances in neural information processing systems","article-title":"Attention is all you need","volume":"vol. 30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.neucom.2026.133857_bib0225","author":"Dosovitskiy"},{"key":"10.1016\/j.neucom.2026.133857_bib0230","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"9992","article-title":"Swin transformer: hierarchical vision transformer using shifted windows","author":"Liu","year":"2021"},{"key":"10.1016\/j.neucom.2026.133857_bib0235","series-title":"Advances in Neural Information Processing Systems","article-title":"Neural discrete representation learning","author":"van den Oord","year":"2017"},{"key":"10.1016\/j.neucom.2026.133857_bib0240","series-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems, NeurIPS 2017","first-page":"3394","article-title":"Deep sets","author":"Zaheer","year":"2017"},{"key":"10.1016\/j.neucom.2026.133857_bib0245","series-title":"Proceedings of the 36th International Conference on Machine Learning (ICML)","first-page":"3744","article-title":"Set transformer: a framework for attention-based Permutation-Invariant neural networks","author":"Lee","year":"2019"},{"key":"10.1016\/j.neucom.2026.133857_bib0250","series-title":"Computer Vision \u2013 ECCV 2020","first-page":"213","article-title":"End-to-End object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.neucom.2026.133857_bib0255","series-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","article-title":"Object-centric learning with slot attention","author":"Locatello","year":"2020"},{"key":"10.1016\/j.neucom.2026.133857_bib0260","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Conditional object-centric learning from video","author":"Kipf","year":"2022"},{"key":"10.1016\/j.neucom.2026.133857_bib0265","series-title":"Proceedings of the 40th International Conference on Machine Learning (ICML)","article-title":"Invariant slot attention: object discovery with slot-centric reference frames","author":"Biza","year":"2023"},{"key":"10.1016\/j.neucom.2026.133857_bib0270","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5401","article-title":"Temporally consistent object-centric learning by contrasting slots","author":"Manasyan","year":"2025"},{"key":"10.1016\/j.neucom.2026.133857_bib0275","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","article-title":"Cascade r-CNN: delving into high quality object detection","author":"Cai","year":"2018"},{"key":"10.1016\/j.neucom.2026.133857_bib0280","series-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","article-title":"HRFormer: high-resolution transformer for dense prediction","author":"Yuan","year":"2021"},{"key":"10.1016\/j.neucom.2026.133857_bib0285","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","article-title":"SHaRPose: sparse high-resolution representation for human pose estimation","author":"An","year":"2024"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226012543?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226012543?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T23:05:54Z","timestamp":1784934354000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226012543"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":57,"alternative-id":["S0925231226012543"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.133857","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Mitigating token role drift under occlusion with no inference overhead for compositional-token pose estimation","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.133857","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133857"}}