{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T13:17:40Z","timestamp":1783171060457,"version":"3.54.6"},"reference-count":47,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"NSFC","doi-asserted-by":"publisher","award":["62472025"],"award-info":[{"award-number":["62472025"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Vision and Image Understanding"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1016\/j.cviu.2026.104752","type":"journal-article","created":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T16:14:11Z","timestamp":1776701651000},"page":"104752","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Exploring representation learning from vision-language models for open-vocabulary HOI detection"],"prefix":"10.1016","volume":"268","author":[{"given":"Chunyan","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhu","family":"Teng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baopeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.cviu.2026.104752_b1","doi-asserted-by":"crossref","unstructured":"Abdelfattah,\u00a0R., Guo,\u00a0Q., Li,\u00a0X., Wang,\u00a0X., Wang,\u00a0S., 2023. Cdul: Clip-driven unsupervised learning for multi-label image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 1348\u20131357.","DOI":"10.1109\/ICCV51070.2023.00130"},{"key":"10.1016\/j.cviu.2026.104752_b2","article-title":"Detecting any human-object interaction relationship: Universal HOI detector with spatial prompt learning on foundation models","volume":"36","author":"Cao","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104752_b3","doi-asserted-by":"crossref","unstructured":"Cao,\u00a0Y., Tang,\u00a0Q., Yang,\u00a0F., Su,\u00a0X., You,\u00a0S., Lu,\u00a0X., Xu,\u00a0C., 2023. Re-mine, learn and reason: Exploring the cross-modal semantic correlations for language-guided hoi detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 23492\u201323503.","DOI":"10.1109\/ICCV51070.2023.02147"},{"key":"10.1016\/j.cviu.2026.104752_b4","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.cviu.2026.104752_b5","series-title":"2018 IEEE Winter Conference on Applications of Computer Vision","first-page":"381","article-title":"Learning to detect human-object interactions","author":"Chao","year":"2018"},{"issue":"1","key":"10.1016\/j.cviu.2026.104752_b6","doi-asserted-by":"crossref","first-page":"38","DOI":"10.1007\/s11633-022-1369-5","article-title":"Vlp: A survey on vision-language pre-training","volume":"20","author":"Chen","year":"2023","journal-title":"Mach. Intell. Res."},{"key":"10.1016\/j.cviu.2026.104752_b7","doi-asserted-by":"crossref","unstructured":"Du,\u00a0Y., Wei,\u00a0F., Zhang,\u00a0Z., Shi,\u00a0M., Gao,\u00a0Y., Li,\u00a0G., 2022. Learning to prompt for open-vocabulary object detection with vision-language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14084\u201314093.","DOI":"10.1109\/CVPR52688.2022.01369"},{"issue":"2","key":"10.1016\/j.cviu.2026.104752_b8","doi-asserted-by":"crossref","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","article-title":"Clip-adapter: Better vision-language models with feature adapters","volume":"132","author":"Gao","year":"2024","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.cviu.2026.104752_b9","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XII","first-page":"696","article-title":"Drg: Dual relation graph for human-object interaction detection","volume":"Vol. 16","author":"Gao","year":"2020"},{"key":"10.1016\/j.cviu.2026.104752_b10","series-title":"Ican: Instance-centric attention network for human-object interaction detection","author":"Gao","year":"2018"},{"key":"10.1016\/j.cviu.2026.104752_b11","doi-asserted-by":"crossref","unstructured":"Gkioxari,\u00a0G., Girshick,\u00a0R., Doll\u00e1r,\u00a0P., He,\u00a0K., 2018. Detecting and recognizing human-object interactions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp. 8359\u20138367.","DOI":"10.1109\/CVPR.2018.00872"},{"key":"10.1016\/j.cviu.2026.104752_b12","doi-asserted-by":"crossref","unstructured":"Gondal,\u00a0M.W., Gast,\u00a0J., Ruiz,\u00a0I.A., Droste,\u00a0R., Macri,\u00a0T., Kumar,\u00a0S., Staudigl,\u00a0L., 2024. Domain aligned CLIP for few-shot classification. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 5721\u20135730.","DOI":"10.1109\/WACV57701.2024.00562"},{"key":"10.1016\/j.cviu.2026.104752_b13","doi-asserted-by":"crossref","unstructured":"Gupta,\u00a0T., Schwing,\u00a0A., Hoiem,\u00a0D., 2019. No-frills human-object interaction detection: Factorization, layout encodings, and training techniques. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9677\u20139685.","DOI":"10.1109\/ICCV.2019.00977"},{"key":"10.1016\/j.cviu.2026.104752_b14","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XV","first-page":"584","article-title":"Visual compositional learning for human-object interaction detection","volume":"Vol. 16","author":"Hou","year":"2020"},{"key":"10.1016\/j.cviu.2026.104752_b15","doi-asserted-by":"crossref","unstructured":"Hou,\u00a0Z., Yu,\u00a0B., Qiao,\u00a0Y., Peng,\u00a0X., Tao,\u00a0D., 2021a. Affordance transfer learning for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 495\u2013504.","DOI":"10.1109\/CVPR46437.2021.00056"},{"key":"10.1016\/j.cviu.2026.104752_b16","doi-asserted-by":"crossref","unstructured":"Hou,\u00a0Z., Yu,\u00a0B., Qiao,\u00a0Y., Peng,\u00a0X., Tao,\u00a0D., 2021b. Detecting human-object interaction via fabricated compositional learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14646\u201314655.","DOI":"10.1109\/CVPR46437.2021.01441"},{"key":"10.1016\/j.cviu.2026.104752_b17","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XV","first-page":"498","article-title":"Uniondet: Union-level detector towards real-time human-object interaction detection","volume":"Vol. 16","author":"Kim","year":"2020"},{"key":"10.1016\/j.cviu.2026.104752_b18","doi-asserted-by":"crossref","unstructured":"Kim,\u00a0S., Jung,\u00a0D., Cho,\u00a0M., 2023. Relational context learning for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2925\u20132934.","DOI":"10.1109\/CVPR52729.2023.00286"},{"key":"10.1016\/j.cviu.2026.104752_b19","doi-asserted-by":"crossref","unstructured":"Kim,\u00a0B., Lee,\u00a0J., Kang,\u00a0J., Kim,\u00a0E.-S., Kim,\u00a0H.J., 2021. Hotr: End-to-end human-object interaction detection with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 74\u201383.","DOI":"10.1109\/CVPR46437.2021.00014"},{"key":"10.1016\/j.cviu.2026.104752_b20","doi-asserted-by":"crossref","unstructured":"Lei,\u00a0T., Caba,\u00a0F., Chen,\u00a0Q., Jin,\u00a0H., Peng,\u00a0Y., Liu,\u00a0Y., 2023. Efficient Adaptive Human-Object Interaction Detection with Concept-guided Memory. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6480\u20136490.","DOI":"10.1109\/ICCV51070.2023.00596"},{"key":"10.1016\/j.cviu.2026.104752_b21","doi-asserted-by":"crossref","unstructured":"Lei,\u00a0T., Yin,\u00a0S., Liu,\u00a0Y., 2024. Exploring the potential of large foundation models for open-vocabulary hoi detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 16657\u201316667.","DOI":"10.1109\/CVPR52733.2024.01576"},{"key":"10.1016\/j.cviu.2026.104752_b22","doi-asserted-by":"crossref","unstructured":"Li,\u00a0Y.-L., Zhou,\u00a0S., Huang,\u00a0X., Xu,\u00a0L., Ma,\u00a0Z., Fang,\u00a0H.-S., Wang,\u00a0Y., Lu,\u00a0C., 2019. Transferable interactiveness knowledge for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3585\u20133594.","DOI":"10.1109\/CVPR.2019.00370"},{"key":"10.1016\/j.cviu.2026.104752_b23","doi-asserted-by":"crossref","unstructured":"Liao,\u00a0Y., Liu,\u00a0S., Wang,\u00a0F., Chen,\u00a0Y., Qian,\u00a0C., Feng,\u00a0J., 2020. Ppdm: Parallel point detection and matching for real-time human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 482\u2013490.","DOI":"10.1109\/CVPR42600.2020.00056"},{"key":"10.1016\/j.cviu.2026.104752_b24","doi-asserted-by":"crossref","unstructured":"Liao,\u00a0Y., Zhang,\u00a0A., Lu,\u00a0M., Wang,\u00a0Y., Li,\u00a0X., Liu,\u00a0S., 2022. Gen-vlkt: Simplify association and enhance interaction understanding for hoi detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 20123\u201320132.","DOI":"10.1109\/CVPR52688.2022.01949"},{"key":"10.1016\/j.cviu.2026.104752_b25","doi-asserted-by":"crossref","unstructured":"Liu,\u00a0X., Li,\u00a0Y.-L., Wu,\u00a0X., Tai,\u00a0Y.-W., Lu,\u00a0C., Tang,\u00a0C.-K., 2022. Interactiveness field in human-object interactions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 20113\u201320122.","DOI":"10.1109\/CVPR52688.2022.01948"},{"key":"10.1016\/j.cviu.2026.104752_b26","series-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2017"},{"key":"10.1016\/j.cviu.2026.104752_b27","article-title":"Fgahoi: Fine-grained anchors for human-object interaction detection","author":"Ma","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104752_b28","doi-asserted-by":"crossref","unstructured":"Ning,\u00a0S., Qiu,\u00a0L., Liu,\u00a0Y., He,\u00a0X., 2023. Hoiclip: Efficient knowledge transfer for hoi detection with vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 23507\u201323517.","DOI":"10.1109\/CVPR52729.2023.02251"},{"key":"10.1016\/j.cviu.2026.104752_b29","doi-asserted-by":"crossref","unstructured":"Qu,\u00a0X., Ding,\u00a0C., Li,\u00a0X., Zhong,\u00a0X., Tao,\u00a0D., 2022. Distillation using oracle queries for transformer-based human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 19558\u201319567.","DOI":"10.1109\/CVPR52688.2022.01895"},{"key":"10.1016\/j.cviu.2026.104752_b30","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"issue":"6","key":"10.1016\/j.cviu.2026.104752_b31","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume":"39","author":"Ren","year":"2016","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104752_b32","doi-asserted-by":"crossref","unstructured":"Tamura,\u00a0M., Ohashi,\u00a0H., Yoshinaga,\u00a0T., 2021. Qpic: Query-based pairwise human-object interaction detection with image-wide contextual information. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10410\u201310419.","DOI":"10.1109\/CVPR46437.2021.01027"},{"key":"10.1016\/j.cviu.2026.104752_b33","doi-asserted-by":"crossref","unstructured":"Wan,\u00a0B., Zhou,\u00a0D., Liu,\u00a0Y., Li,\u00a0R., He,\u00a0X., 2019. Pose-aware multi-level feature network for human object interaction detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9469\u20139478.","DOI":"10.1109\/ICCV.2019.00956"},{"key":"10.1016\/j.cviu.2026.104752_b34","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0S., Duan,\u00a0Y., Ding,\u00a0H., Tan,\u00a0Y.-P., Yap,\u00a0K.-H., Yuan,\u00a0J., 2022. Learning transferable human-object interaction detector with natural language supervision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 939\u2013948.","DOI":"10.1109\/CVPR52688.2022.00101"},{"issue":"7","key":"10.1016\/j.cviu.2026.104752_b35","doi-asserted-by":"crossref","first-page":"5603","DOI":"10.1109\/TCSVT.2024.3358952","article-title":"Ted-net: Dispersal attention for perceiving interaction region in indirectly-contact hoi detection","volume":"34","author":"Wang","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.cviu.2026.104752_b36","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0T., Yang,\u00a0T., Danelljan,\u00a0M., Khan,\u00a0F.S., Zhang,\u00a0X., Sun,\u00a0J., 2020. Learning human-object interaction detection using interaction points. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4116\u20134125.","DOI":"10.1109\/CVPR42600.2020.00417"},{"key":"10.1016\/j.cviu.2026.104752_b37","doi-asserted-by":"crossref","unstructured":"Wang,\u00a0S., Yap,\u00a0K.-H., Ding,\u00a0H., Wu,\u00a0J., Yuan,\u00a0J., Tan,\u00a0Y.-P., 2021. Discovering human interactions with large-vocabulary objects via query and multi-scale detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 13475\u201313484.","DOI":"10.1109\/ICCV48922.2021.01322"},{"key":"10.1016\/j.cviu.2026.104752_b38","doi-asserted-by":"crossref","unstructured":"Wu,\u00a0M., Gu,\u00a0J., Shen,\u00a0Y., Lin,\u00a0M., Chen,\u00a0C., Sun,\u00a0X., 2023. End-to-end zero-shot hoi detection via vision and language knowledge distillation. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 37, pp. 2839\u20132846.","DOI":"10.1609\/aaai.v37i3.25385"},{"key":"10.1016\/j.cviu.2026.104752_b39","article-title":"Towards open vocabulary learning: A survey","author":"Wu","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.cviu.2026.104752_b40","doi-asserted-by":"crossref","unstructured":"Xie,\u00a0C., Zeng,\u00a0F., Hu,\u00a0Y., Liang,\u00a0S., Wei,\u00a0Y., 2023. Category query learning for human-object interaction classification. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 15275\u201315284.","DOI":"10.1109\/CVPR52729.2023.01466"},{"key":"10.1016\/j.cviu.2026.104752_b41","doi-asserted-by":"crossref","unstructured":"Xu,\u00a0B., Wong,\u00a0Y., Li,\u00a0J., Zhao,\u00a0Q., Kankanhalli,\u00a0M.S., 2019. Learning to detect human-object interactions with knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2019.00212"},{"key":"10.1016\/j.cviu.2026.104752_b42","doi-asserted-by":"crossref","first-page":"37416","DOI":"10.52202\/068431-2712","article-title":"Rlip: Relational language-image pre-training for human-object interaction detection","volume":"35","author":"Yuan","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104752_b43","doi-asserted-by":"crossref","unstructured":"Zareian,\u00a0A., Rosa,\u00a0K.D., Hu,\u00a0D.H., Chang,\u00a0S.-F., 2021. Open-vocabulary object detection using captions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 14393\u201314402.","DOI":"10.1109\/CVPR46437.2021.01416"},{"key":"10.1016\/j.cviu.2026.104752_b44","doi-asserted-by":"crossref","unstructured":"Zhang,\u00a0F.Z., Campbell,\u00a0D., Gould,\u00a0S., 2022. Efficient two-stage detection of human-object interactions with a novel unary-pairwise transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 20104\u201320112.","DOI":"10.1109\/CVPR52688.2022.01947"},{"key":"10.1016\/j.cviu.2026.104752_b45","first-page":"17209","article-title":"Mining the benefits of two-stage and one-stage hoi detection","volume":"34","author":"Zhang","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.cviu.2026.104752_b46","doi-asserted-by":"crossref","unstructured":"Zhong,\u00a0X., Qu,\u00a0X., Ding,\u00a0C., Tao,\u00a0D., 2021. Glance and gaze: Inferring action-aware points for one-stage human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 13234\u201313243.","DOI":"10.1109\/CVPR46437.2021.01303"},{"key":"10.1016\/j.cviu.2026.104752_b47","doi-asserted-by":"crossref","unstructured":"Zhou,\u00a0Y., Zhang,\u00a0R., Chen,\u00a0C., Li,\u00a0C., Tensmeyer,\u00a0C., Yu,\u00a0T., Gu,\u00a0J., Xu,\u00a0J., Sun,\u00a0T., 2022. Towards language-free training for text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 17907\u201317917.","DOI":"10.1109\/CVPR52688.2022.01738"}],"container-title":["Computer Vision and Image Understanding"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001190?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1077314226001190?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T12:33:40Z","timestamp":1783168420000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1077314226001190"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":47,"alternative-id":["S1077314226001190"],"URL":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104752","relation":{},"ISSN":["1077-3142"],"issn-type":[{"value":"1077-3142","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Exploring representation learning from vision-language models for open-vocabulary HOI detection","name":"articletitle","label":"Article Title"},{"value":"Computer Vision and Image Understanding","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.cviu.2026.104752","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104752"}}