{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T22:19:52Z","timestamp":1784413192777,"version":"3.55.0"},"reference-count":108,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,1]],"date-time":"2025-05-01T00:00:00Z","timestamp":1746057600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62076183"],"award-info":[{"award-number":["62076183"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shanghai Science and Technology Innovation Action","award":["20511100700"],"award-info":[{"award-number":["20511100700"]}]},{"DOI":"10.13039\/501100003399","name":"Science and Technology Commission of Shanghai Municipality","doi-asserted-by":"publisher","award":["23511103100"],"award-info":[{"award-number":["23511103100"]}],"id":[{"id":"10.13039\/501100003399","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shanghai Municipal Science and Technology Major Project","award":["2021SHZDZX0100"],"award-info":[{"award-number":["2021SHZDZX0100"]}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1109\/tpami.2025.3531452","type":"journal-article","created":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T18:36:14Z","timestamp":1737138974000},"page":"3515-3529","source":"Crossref","is-referenced-by-count":9,"title":["<i>RelationLMM<\/i>: Large Multimodal Model as Open and Versatile Visual Relationship Generalist"],"prefix":"10.1109","volume":"47","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5808-1742","authenticated-orcid":false,"given":"Chi","family":"Xie","sequence":"first","affiliation":[{"name":"School of Software Engineering, Tongji University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0457-6093","authenticated-orcid":false,"given":"Shuang","family":"Liang","sequence":"additional","affiliation":[{"name":"School of Software Engineering, Tongji University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Li","sequence":"additional","affiliation":[{"name":"Sensetime Research, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1521-8163","authenticated-orcid":false,"given":"Zhao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Sensetime Research, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5759-9176","authenticated-orcid":false,"given":"Feng","family":"Zhu","sequence":"additional","affiliation":[{"name":"Sensetime Research, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5874-131X","authenticated-orcid":false,"given":"Rui","family":"Zhao","sequence":"additional","affiliation":[{"name":"Sensetime Research, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5473-4112","authenticated-orcid":false,"given":"Yichen","family":"Wei","sequence":"additional","affiliation":[{"name":"Shukun Technology, Vienna, Austria"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_51"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.330"},{"key":"ref4","article-title":"Visual semantic role labeling","author":"Gupta","year":"2015"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.122"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2018.00048"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2009.83"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10605-2_27"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00779"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01901"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/3446370"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01324"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00056"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3268066"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3330304"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00641"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00718"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00563"},{"key":"ref19","first-page":"79095","article-title":"Described object detection: Liberating object detection with flexible expressions","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xie"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25385"},{"key":"ref21","first-page":"739","article-title":"Detecting any human-object interaction relationship: Universal HOI detector with spatial prompt learning on foundation models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Cao"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58592-1_36"},{"key":"ref23","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref24","first-page":"34892","article-title":"Visual instruction tuning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liu"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.02484"},{"key":"ref26","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref27","article-title":"Shikra: Unleashing multimodal LLM\u2019s referential dialogue magic","author":"Chen","year":"2023"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00370"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01027"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01949"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01858"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00611"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01096"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-020-01316-z"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01563"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3243306"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3329339"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01888"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02000"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00285"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_4"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01847"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00872"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01307"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01897"},{"key":"ref47","first-page":"17209","article-title":"Mining the benefits of two-stage and one-stage HOI detection","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01894"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01896"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00286"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01466"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01979"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02251"},{"key":"ref54","first-page":"45895","article-title":"Clip4hoi: Towards adapting CLIP for practical zero-shot HOI detection","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Mao"},{"key":"ref55","first-page":"37416","article-title":"RLIP: Relational language-image pre-training for human-object interaction detection","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yuan"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19812-0_27"},{"key":"ref58","first-page":"21158","article-title":"Neural-logic human-object interaction detection","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00285"},{"key":"ref60","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3054048"},{"key":"ref62","first-page":"43041","article-title":"RIO: A benchmark for reasoning intention-oriented objects in open environments","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Qu"},{"issue":"1","key":"ref63","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3605781","article-title":"Transformer-based relational inference network for complex visual relational reasoning","volume":"20","author":"Tan","year":"2023","journal-title":"ACM Trans. Multimedia Comput., Commun. Appl."},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01245"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00808"},{"key":"ref66","article-title":"Grounding multimodal large language models to the world","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Peng"},{"key":"ref67","article-title":"QWEN-VL: A frontier large vision-language model with versatile abilities","author":"Bai","year":"2023"},{"key":"ref68","article-title":"Sphinx: The joint mixing of weights, tasks, and visual embeddings for multi-modal large language models","author":"Lin","year":"2023"},{"key":"ref69","first-page":"61501","article-title":"VisionLLM: Large language model is also an open-ended decoder for vision-centric tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wang"},{"key":"ref70","article-title":"Deformable DETR: Deformable transformers for end-to-end object detection","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhu"},{"key":"ref71","first-page":"20067","article-title":"MotionGPT: Human motion as a foreign language","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Jiang"},{"key":"ref72","article-title":"PoseGPT: Chatting about 3D human pose","author":"Feng","year":"2023"},{"key":"ref73","article-title":"LISA: Reasoning segmentation via large language model","author":"Lai","year":"2023"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.876"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref76","article-title":"Pix2seq: A language modeling framework for object detection","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chen"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3285009"},{"key":"ref78","article-title":"Teaching models to express their uncertainty in words","author":"Lin","year":"2022","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"ref80","first-page":"17597","article-title":"TOIST: Task oriented instance segmentation transformer with noun-pronoun distillation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Li"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6616"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01240-3_25"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_8"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01948"},{"key":"ref85","first-page":"24824","article-title":"Chain of thought prompting elicits reasoning in large language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wei"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_41"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00852"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02182"},{"key":"ref91","first-page":"46595","article-title":"Judging LLM-as-a-judge with MT-bench and chatbot arena","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zheng"},{"key":"ref92","article-title":"MMBench: Is your multi-modal model an all-around player?","author":"Liu","year":"2023"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01976"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00955"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00244"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01982"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01223"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3282889"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01790"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01219-9_20"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01180"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00380"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00482"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2021.3079278"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3226624"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01471"},{"key":"ref107","first-page":"32942","article-title":"Coarse-to-fine vision-language pre-training with fusion in the backbone","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dou"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.215"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/10958761\/10845195.pdf?arnumber=10845195","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,10]],"date-time":"2025-04-10T17:13:27Z","timestamp":1744305207000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10845195\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5]]},"references-count":108,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2025.3531452","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5]]}}}