{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T13:05:18Z","timestamp":1784898318977,"version":"3.55.0"},"reference-count":56,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62473362"],"award-info":[{"award-number":["62473362"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1016\/j.eswa.2026.131825","type":"journal-article","created":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T17:14:15Z","timestamp":1772730855000},"page":"131825","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Dual-enhanced human object interaction detection: Content-aware positional embedding and cognition-inspired reasoning"],"prefix":"10.1016","volume":"317","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-7881-2274","authenticated-orcid":false,"given":"Haojun","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7851-9230","authenticated-orcid":false,"given":"Yuequan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1801-3363","authenticated-orcid":false,"given":"Zhiqiang","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3627-0371","authenticated-orcid":false,"given":"Junzhi","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9047-4415","authenticated-orcid":false,"given":"Xu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.131825_bib0001","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"23492","article-title":"Re-mine, learn and reason: Exploring the cross-modal semantic correlations for language-guided HOI detection","author":"Cao","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0002","series-title":"European conference on computer vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.eswa.2026.131825_bib0003","series-title":"2018 IEEE winter conference on applications of computer vision (WACV)","first-page":"381","article-title":"Learning to detect human-object interactions","author":"Chao","year":"2018"},{"key":"10.1016\/j.eswa.2026.131825_bib0004","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"1017","article-title":"HICO: A benchmark for recognizing human-object interactions in images","author":"Chao","year":"2015"},{"key":"10.1016\/j.eswa.2026.131825_bib0005","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9004","article-title":"Reformulating HOI detection as adaptive set prediction","author":"Chen","year":"2021"},{"key":"10.1016\/j.eswa.2026.131825_bib0006","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S. et al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. 10.48550\/arXiv.2010.11929."},{"key":"10.1016\/j.eswa.2026.131825_bib0007","series-title":"Proceedings of the European conference on computer vision (ECCV)","first-page":"51","article-title":"Pairwise body-part attention for recognizing human-object interactions","author":"Fang","year":"2018"},{"key":"10.1016\/j.eswa.2026.131825_bib0008","doi-asserted-by":"crossref","first-page":"3125","DOI":"10.1109\/TMM.2023.3307896","article-title":"HODN: Disentangling human-object feature for HOI detection","volume":"26","author":"Fang","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.131825_bib0009","series-title":"British machine vision conference","article-title":"iCAN: Instance-centric attention network for human-object interaction detection","author":"Gao","year":"2018"},{"key":"10.1016\/j.eswa.2026.131825_bib0010","doi-asserted-by":"crossref","DOI":"10.1016\/j.cviu.2025.104551","article-title":"SCESS-Net: Semantic consistency enhancement and segment selection network for audio-visual event localization","volume":"262","author":"Gao","year":"2025","journal-title":"Computer Vision and Image Understanding"},{"key":"10.1016\/j.eswa.2026.131825_bib0011","series-title":"2018 IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8359","article-title":"Detecting and recognizing human-object interactions","author":"Gkioxari","year":"2018"},{"key":"10.1016\/j.eswa.2026.131825_bib0012","unstructured":"Gupta, S., & Malik, J. (2015). Visual semantic role labeling. 10.48550\/arXiv.1505.04474."},{"key":"10.1016\/j.eswa.2026.131825_bib0013","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.eswa.2026.131825_bib0014","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125297","article-title":"Deep scene understanding with extended text description for human object interaction detection","volume":"259","author":"Hong","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.131825_bib0015","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"495","article-title":"Affordance transfer learning for human-object interaction detection","author":"Hou","year":"2021"},{"key":"10.1016\/j.eswa.2026.131825_bib0016","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5353","article-title":"What to look at and where: Semantic and spatial refined transformer for detecting human-object interactions","author":"Iftekhar","year":"2022"},{"key":"10.1016\/j.eswa.2026.131825_bib0017","series-title":"European conference on computer vision","first-page":"498","article-title":"UnionDet: Union-level detector towards real-time human-object interaction detection","author":"Kim","year":"2020"},{"key":"10.1016\/j.eswa.2026.131825_bib0018","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"74","article-title":"HOTR: End-to-end human-object interaction detection with transformers","author":"Kim","year":"2021"},{"key":"10.1016\/j.eswa.2026.131825_bib0019","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"2925","article-title":"Relational context learning for human-object interaction detection","author":"Kim","year":"2023"},{"issue":"1-2","key":"10.1016\/j.eswa.2026.131825_bib0020","doi-asserted-by":"crossref","first-page":"83","DOI":"10.1002\/nav.3800020109","article-title":"The hungarian method for the assignment problem","volume":"2","author":"Kuhn","year":"1955","journal-title":"Naval Research Logistics Quarterly"},{"key":"10.1016\/j.eswa.2026.131825_bib0021","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"28191","article-title":"Disentangled pre-training for human-object interaction detection","author":"Li","year":"2024"},{"key":"10.1016\/j.eswa.2026.131825_bib0022","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"482","article-title":"PPDM: Parallel point detection and matching for real-time human-object interaction detection","author":"Liao","year":"2020"},{"key":"10.1016\/j.eswa.2026.131825_bib0023","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"20123","article-title":"Gen-VLKT: Simplify association and enhance interaction understanding for HOI detection","author":"Liao","year":"2022"},{"key":"10.1016\/j.eswa.2026.131825_bib0024","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"2980","article-title":"Focal loss for dense object detection","author":"Lin","year":"2017"},{"key":"10.1016\/j.eswa.2026.131825_bib0025","series-title":"Computer vision\u2013ECCV 2014: 13th European conference, Zurich, Switzerland, september 6-12, 2014, proceedings, part v 13","first-page":"740","article-title":"Microsoft CoCo: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.eswa.2026.131825_bib0026","series-title":"European conference on computer vision","first-page":"248","article-title":"Amplifying key cues for human-object-interaction detection","author":"Liu","year":"2020"},{"key":"10.1016\/j.eswa.2026.131825_bib0027","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3262","article-title":"PETRv2: A unified framework for 3d perception from multi-camera images","author":"Liu","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0028","unstructured":"Loshchilov, I., & Hutter, F. (2017). Decoupled weight decay regularization. arXiv: 1711.05101."},{"issue":"4","key":"10.1016\/j.eswa.2026.131825_bib0029","doi-asserted-by":"crossref","first-page":"2415","DOI":"10.1109\/TPAMI.2023.3331738","article-title":"FGAHOI: Fine-grained anchors for human-object interaction detection","volume":"46","author":"Ma","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.131825_bib0030","doi-asserted-by":"crossref","first-page":"45895","DOI":"10.52202\/075280-1989","article-title":"CLIP4HOI: Towards adapting clip for practical zero-shot HOI detection","volume":"36","author":"Mao","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.131825_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"23507","article-title":"HOICLIP: Efficient knowledge transfer for HOI detection with vision-language models","author":"Ning","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0032","doi-asserted-by":"crossref","unstructured":"Qiao, T., Li, R., Li, F. W. B., Kubotani, Y., Morishima, S., & Shum, H. P. H. (2025). Geometric visual fusion graph neural networks for multi-person human-object interaction recognition in videos. arXiv: 2506.03440.","DOI":"10.1016\/j.eswa.2025.128344"},{"key":"10.1016\/j.eswa.2026.131825_bib0033","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19558","article-title":"Distillation using oracle queries for transformer-based human-object interaction detection","author":"Qu","year":"2022"},{"key":"10.1016\/j.eswa.2026.131825_bib0034","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.131825_bib0035","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"658","article-title":"Generalized intersection over union: A metric and a loss for bounding box regression","author":"Rezatofighi","year":"2019"},{"key":"10.1016\/j.eswa.2026.131825_bib0036","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10410","article-title":"QPIC: Query-based pairwise human-object interaction detection with image-wide contextual information","author":"Tamura","year":"2021"},{"key":"10.1016\/j.eswa.2026.131825_bib0037","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"21614","article-title":"Agglomerative transformer for human-object interaction detection","author":"Tu","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0038","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13617","article-title":"VSGNet: Spatial attention network for detecting human object interactions using graph convolutions","author":"Ulutan","year":"2020"},{"key":"10.1016\/j.eswa.2026.131825_bib0039","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. Advances in Neural Information Processing Systems,(pp. 5998\u20136008)."},{"key":"10.1016\/j.eswa.2026.131825_bib0040","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9469","article-title":"Pose-aware multi-level feature network for human object interaction detection","author":"Wan","year":"2019"},{"key":"10.1016\/j.eswa.2026.131825_bib0041","series-title":"European conference on computer vision","first-page":"248","article-title":"Contextual heterogeneous graph network for human-object interaction detection","author":"Wang","year":"2020"},{"key":"10.1016\/j.eswa.2026.131825_bib0042","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4116","article-title":"Learning human-object interaction detection using interaction points","author":"Wang","year":"2020"},{"issue":"7","key":"10.1016\/j.eswa.2026.131825_bib0043","doi-asserted-by":"crossref","first-page":"5603","DOI":"10.1109\/TCSVT.2024.3358952","article-title":"TED-Net: Dispersal attention for perceiving interaction region in indirectly-contact HOI detection","volume":"34","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.131825_bib0044","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"2839","article-title":"End-to-end zero-shot HOI detection via vision and language knowledge distillation","volume":"vol. 37","author":"Wu","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0045","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15275","article-title":"Category query learning for human-object interaction classification","author":"Xie","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0046","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16954","article-title":"Open-world human-object interaction detection via multi-modal prompts","author":"Yang","year":"2024"},{"key":"10.1016\/j.eswa.2026.131825_bib0047","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"3206","article-title":"Detecting human-object interactions with object-guided cross-modal calibrated semantics","volume":"vol. 36","author":"Yuan","year":"2022"},{"key":"10.1016\/j.eswa.2026.131825_bib0048","first-page":"17209","article-title":"Mining the benefits of two-stage and one-stage HOI detection","volume":"34","author":"Zhang","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.131825_bib0049","series-title":"2023 IEEE international conference on multimedia and expo (ICME)","first-page":"2219","article-title":"HOD: Human-object decoupling network for HOI detection","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0050","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19548","article-title":"Exploring structure-aware transformer over interaction proposals for human-object interaction detection","author":"Zhang","year":"2022"},{"key":"10.1016\/j.eswa.2026.131825_bib0051","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19392","article-title":"Open-category human-object interaction pre-training via language modeling framework","author":"Zheng","year":"2023"},{"key":"10.1016\/j.eswa.2026.131825_bib0052","series-title":"European conference on computer vision","first-page":"69","article-title":"Polysemy deciphering network for human-object interaction detection","author":"Zhong","year":"2020"},{"key":"10.1016\/j.eswa.2026.131825_bib0053","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13234","article-title":"Glance and gaze: Inferring action-aware points for one-stage human-object interaction detection","author":"Zhong","year":"2021"},{"key":"10.1016\/j.eswa.2026.131825_bib0054","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19568","article-title":"Human-object interaction detection via disentangled transformer","author":"Zhou","year":"2022"},{"key":"10.1016\/j.eswa.2026.131825_bib0055","first-page":"1","article-title":"Geometric features enhanced human-object interaction detection","volume":"73","author":"Zhu","year":"2024","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"10.1016\/j.eswa.2026.131825_bib0056","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11825","article-title":"End-to-end human object interaction detection with HOI transformer","author":"Zou","year":"2021"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426007384?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426007384?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T12:41:44Z","timestamp":1784896904000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426007384"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":56,"alternative-id":["S0957417426007384"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131825","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Dual-enhanced human object interaction detection: Content-aware positional embedding and cognition-inspired reasoning","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.131825","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"131825"}}