{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T15:43:24Z","timestamp":1781797404763,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":65,"publisher":"ACM","funder":[{"DOI":"10.13039\/100018693","name":"HORIZON EUROPE Framework Programme","doi-asserted-by":"publisher","award":["101120237,101168272,101214398"],"award-info":[{"award-number":["101120237,101168272,101214398"]}],"id":[{"id":"10.13039\/100018693","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100021856","name":"Ministero dell'Universit\u00e0 e della Ricerca","doi-asserted-by":"publisher","award":["PE00000013,2022EX F3HX"],"award-info":[{"award-number":["PE00000013,2022EX F3HX"]}],"id":[{"id":"10.13039\/501100021856","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754770","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:27:39Z","timestamp":1761377259000},"page":"2801-2810","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Dynamic Scoring with Enhanced Semantics for Training-Free Human-Object Interaction Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1938-3449","authenticated-orcid":false,"given":"Francesco","family":"Tonini","sequence":"first","affiliation":[{"name":"University of Trento, Trento, Italy and Fondazione Bruno Kessler, Trento, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1874-3078","authenticated-orcid":false,"given":"Lorenzo","family":"Vaquero","sequence":"additional","affiliation":[{"name":"Fondazione Bruno Kessler, Trento, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3044-1320","authenticated-orcid":false,"given":"Alessandro","family":"Conti","sequence":"additional","affiliation":[{"name":"University of Trento, Trento, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9583-0087","authenticated-orcid":false,"given":"Cigdem","family":"Beyan","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Verona, Verona, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0228-1147","authenticated-orcid":false,"given":"Elisa","family":"Ricci","sequence":"additional","affiliation":[{"name":"University of Trento, Trento, Italy and Fondazione Bruno Kessler, Trento, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"What Do You See? Enhancing Zero-Shot Image Classification with Multimodal Large Language Models. CoRR","author":"Abdelhamed Abdelrahman","year":"2024","unstructured":"Abdelrahman Abdelhamed, Mahmoud Afifi, and Alec Go. 2024. What Do You See? Enhancing Zero-Shot Image Classification with Multimodal Large Language Models. CoRR, Vol. abs\/2405.15668 (2024), 1-13."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2678"},{"key":"e_1_3_2_1_3_1","first-page":"739","volume-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"36","author":"Cao Yichao","year":"2023","unstructured":"Yichao Cao, Qingfei Tang, Xiu Su, Song Chen, Shan You, Xiaobo Lu, and Chang Xu. 2023. Detecting Any Human-Object Interaction Relationship: Universal HOI Detector with Spatial Prompt Learning on Foundation Models. In Adv. Neural Inf. Process. Syst. (NeurIPS), Vol. 36. Curran Associates, Inc., New Orleans, LA, USA, 739-751."},{"key":"e_1_3_2_1_4_1","first-page":"213","volume-title":"End-to-End Object Detection with Transformers. In European Conf. Comput. Vis. (ECCV)","volume":"12346","author":"Carion Nicolas","year":"2020","unstructured":"Nicolas Carion, Francisco Massa, Gabriel Synnaeve, Nicolas Usunier, Alexander Kirillov, and Sergey Zagoruyko. 2020. End-to-End Object Detection with Transformers. In European Conf. Comput. Vis. (ECCV), Vol. 12346. Springer, Glasgow, UK, 213-229."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2018.00048"},{"key":"e_1_3_2_1_6_1","first-page":"22","volume-title":"European Conf. Comput. Vis. (ECCV)","volume":"15094","author":"Chen Yuan","year":"2024","unstructured":"Yuan Chen, Zi-han Ding, Ziqin Wang, Yan Wang, Lijun Zhang, and Si Liu. 2024. Asynchronous large language model enhanced planner for autonomous driving. In European Conf. Comput. Vis. (ECCV), Vol. 15094. Springer, Milan, Italy, 22-38."},{"key":"e_1_3_2_1_7_1","first-page":"54805","volume-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"36","author":"Fel Thomas","year":"2023","unstructured":"Thomas Fel, Victor Boutin, Louis B\u00e9thune, R\u00e9mi Cad\u00e8ne, Mazda Moayeri, L\u00e9o And\u00e9ol, Mathieu Chalvidal, and Thomas Serre. 2023. A Holistic Approach to Unifying Automatic Concept Extraction and Concept Importance Estimation. In Adv. Neural Inf. Process. Syst. (NeurIPS), Vol. 36. Curran Associates, Inc., New Orleans, LA, USA, 54805-54818."},{"key":"e_1_3_2_1_8_1","first-page":"696","volume-title":"DRG: Dual Relation Graph for Human-Object Interaction Detection. In European Conf. Comput. Vis. (ECCV)","volume":"12357","author":"Gao Chen","year":"2020","unstructured":"Chen Gao, Jiarui Xu, Yuliang Zou, and Jia-Bin Huang. 2020. DRG: Dual Relation Graph for Human-Object Interaction Detection. In European Conf. Comput. Vis. (ECCV), Vol. 12357. Springer, Glasgow, UK, 696-712."},{"key":"e_1_3_2_1_9_1","volume-title":"Visual Semantic Role Labeling. CoRR","author":"Gupta Saurabh","year":"2015","unstructured":"Saurabh Gupta and Jitendra Malik. 2015. Visual Semantic Role Labeling. CoRR, Vol. abs\/1505.04474 (2015), 1-11."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01441"},{"key":"e_1_3_2_1_11_1","first-page":"45463","volume-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"36","author":"Hsu Kyle","year":"2023","unstructured":"Kyle Hsu, William Dorrell, James C. R. Whittington, Jiajun Wu, and Chelsea Finn. 2023. Disentanglement via Latent Quantization. In Adv. Neural Inf. Process. Syst. (NeurIPS), Vol. 36. Curran Associates, Inc., New Orleans, LA, USA, 45463-45488."},{"key":"e_1_3_2_1_12_1","first-page":"498","volume-title":"UnionDet: Union-Level Detector Towards Real-Time Human-Object Interaction Detection. In European Conf. Comput. Vis. (ECCV). Springer","author":"Kim Bumsoo","unstructured":"Bumsoo Kim, Taeho Choi, Jaewoo Kang, and Hyunwoo J. Kim. 2020a. UnionDet: Union-Level Detector Towards Real-Time Human-Object Interaction Detection. In European Conf. Comput. Vis. (ECCV). Springer, Glasgow, UK, 498-514."},{"key":"e_1_3_2_1_13_1","volume-title":"HOTR: End-to-End Human-Object Interaction Detection With Transformers. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE, Virtual, 74-83","author":"Kim Bumsoo","unstructured":"Bumsoo Kim, Junhyun Lee, Jaewoo Kang, Eun-Sol Kim, and Hyunwoo J. Kim. 2021. HOTR: End-to-End Human-Object Interaction Detection With Transformers. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE, Virtual, 74-83."},{"key":"e_1_3_2_1_14_1","first-page":"19556","volume-title":"MSTR: Multi-Scale Transformer for End-to-End Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Kim Bumsoo","year":"2022","unstructured":"Bumsoo Kim, Jonghwan Mun, Kyoung-Woon On, Minchul Shin, Junhyun Lee, and Eun-Sol Kim. 2022. MSTR: Multi-Scale Transformer for End-to-End Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, New Orleans, LA, USA, 19556-19565."},{"key":"e_1_3_2_1_15_1","first-page":"718","volume-title":"Detecting Human-Object Interactions with Action Co-occurrence Priors. In European Conf. Comput. Vis. (ECCV). Springer","author":"Kim Dong-Jin","year":"2020","unstructured":"Dong-Jin Kim, Xiao Sun, Jinsoo Choi, Stephen Lin, and In So Kweon. 2020b. Detecting Human-Object Interactions with Action Co-occurrence Priors. In European Conf. Comput. Vis. (ECCV). Springer, Glasgow, UK, 718-736."},{"key":"e_1_3_2_1_16_1","first-page":"2925","volume-title":"Relational Context Learning for Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Kim Sanghyun","year":"2023","unstructured":"Sanghyun Kim, Deunsol Jung, and Minsu Cho. 2023. Relational Context Learning for Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, Vancouver, BC, Canada, 2925-2934."},{"key":"e_1_3_2_1_17_1","volume-title":"Concept Bottleneck Models. In Int. Conf. Mach. Learn. (ICML). PMLR, Virtual, 5338-5348","author":"Koh Pang Wei","year":"2020","unstructured":"Pang Wei Koh, Thao Nguyen, Yew Siang Tang, Stephen Mussmann, Emma Pierson, Been Kim, and Percy Liang. 2020. Concept Bottleneck Models. In Int. Conf. Mach. Learn. (ICML). PMLR, Virtual, 5338-5348."},{"key":"e_1_3_2_1_18_1","first-page":"6457","volume-title":"Efficient Adaptive Human-Object Interaction Detection with Concept-guided Memory. In IEEE Int. Conf. Comput. Vis. (ICCV). IEEE","author":"Lei Ting","year":"2023","unstructured":"Ting Lei, Fabian Caba, Qingchao Chen, Hailin Jin, Yuxin Peng, and Yang Liu. 2023. Efficient Adaptive Human-Object Interaction Detection with Concept-guided Memory. In IEEE Int. Conf. Comput. Vis. (ICCV). IEEE, Paris, France, 6457-6467."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCE.2024.3371451"},{"key":"e_1_3_2_1_20_1","first-page":"12888","volume-title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Int. Conf. Mach. Learn. (ICML)","volume":"162","author":"Li Junnan","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven C. H. Hoi. 2022a. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Int. Conf. Mach. Learn. (ICML), Vol. 162. PMLR, Baltimore, Maryland, USA, 12888-12900."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01018"},{"key":"e_1_3_2_1_22_1","first-page":"3870","article-title":"Transferable Interactiveness Knowledge for Human-Object Interaction Detection","volume":"44","author":"Li Yong-Lu","year":"2022","unstructured":"Yong-Lu Li, Xinpeng Liu, Xiaoqian Wu, Xijie Huang, Liang Xu, and Cewu Lu. 2022b. Transferable Interactiveness Knowledge for Human-Object Interaction Detection. IEEE Trans. Pattern Anal. Mach. Intell., Vol. 44, 7 (2022), 3870-3882.","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"e_1_3_2_1_23_1","first-page":"479","volume-title":"PPDM: Parallel Point Detection and Matching for Real-Time Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE","author":"Liao Yue","year":"2020","unstructured":"Yue Liao, Si Liu, Fei Wang, Yanjie Chen, Chen Qian, and Jiashi Feng. 2020. PPDM: Parallel Point Detection and Matching for Real-Time Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE, Seattle, WA, USA, 479-487."},{"key":"e_1_3_2_1_24_1","first-page":"20091","volume-title":"GEN-VLKT: Simplify Association and Enhance Interaction Understanding for HOI Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Liao Yue","year":"2022","unstructured":"Yue Liao, Aixi Zhang, Miao Lu, Yongliang Wang, Xiaobo Li, and Si Liu. 2022. GEN-VLKT: Simplify Association and Enhance Interaction Understanding for HOI Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, New Orleans, LA, USA, 20091-20100."},{"key":"e_1_3_2_1_25_1","first-page":"4235","article-title":"ConsNet: Learning Consistency Graph for Zero-Shot Human-Object Interaction Detection. In ACM Multimedia (ACMMM). ACM, Seattle","author":"Liu Ye","year":"2020","unstructured":"Ye Liu, Junsong Yuan, and Chang Wen Chen. 2020. ConsNet: Learning Consistency Graph for Zero-Shot Human-Object Interaction Detection. In ACM Multimedia (ACMMM). ACM, Seattle, WA, USA, 4235-4243.","journal-title":"WA, USA"},{"key":"e_1_3_2_1_26_1","first-page":"3204","volume-title":"On Matching Pursuit and Coordinate Descent. In Int. Conf. Mach. Learn. (ICML). PMLR","author":"Locatello Francesco","year":"2018","unstructured":"Francesco Locatello, Anant Raj, Sai Praneeth Karimireddy, Gunnar R\u00e4tsch, Bernhard Sch\u00f6lkopf, Sebastian U. Stich, and Martin Jaggi. 2018. On Matching Pursuit and Coordinate Descent. In Int. Conf. Mach. Learn. (ICML). PMLR, Stockholm, Sweden, 3204-3213."},{"key":"e_1_3_2_1_27_1","first-page":"28212","volume-title":"Discovering Syntactic Interaction Clues for Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Luo Jinguo","year":"2024","unstructured":"Jinguo Luo, Weihong Ren, Weibo Jiang, Xi'ai Chen, Qiang Wang, Zhi Han, and Honghai Liu. 2024. Discovering Syntactic Interaction Clues for Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, Seattle, WA, USA, 28212-28222."},{"key":"e_1_3_2_1_28_1","first-page":"55394","volume-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"36","author":"Maiorca Valentino","year":"2023","unstructured":"Valentino Maiorca, Luca Moschella, Antonio Norelli, Marco Fumero, Francesco Locatello, and Emanuele Rodol\u00e0. 2023. Latent Space Translation via Semantic Alignment. In Adv. Neural Inf. Process. Syst. (NeurIPS), Vol. 36. Curran Associates, Inc., New Orleans, LA, USA, 55394-55414."},{"key":"e_1_3_2_1_29_1","first-page":"45895","volume-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"36","author":"Mao Yunyao","year":"2023","unstructured":"Yunyao Mao, Jiajun Deng, Wengang Zhou, Li Li, Yao Fang, and Houqiang Li. 2023. CLIP4HOI: Towards Adapting CLIP for Practical Zero-Shot HOI Detection. In Adv. Neural Inf. Process. Syst. (NeurIPS), Vol. 36. Curran Associates, Inc., New Orleans, LA, USA, 45895-45906."},{"key":"e_1_3_2_1_30_1","volume-title":"Confer. North Americ. Chap. Assoc. Comp. Ling.: Human Lang. Tech. (NAACL-HLT)","author":"Mikolov Tom\u00e1s","unstructured":"Tom\u00e1s Mikolov, Wen-tau Yih, and Geoffrey Zweig. 2013. Linguistic Regularities in Continuous Space Word Representations. In Confer. North Americ. Chap. Assoc. Comp. Ling.: Human Lang. Tech. (NAACL-HLT). The Association for Computational Linguistics, Atlanta, Georgia, USA, 746-751."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01428"},{"key":"e_1_3_2_1_32_1","volume-title":"Int. Conf. Learn. Represent. (ICLR). OpenReview.net, Kigali, Rwanda, 1-26","author":"Moschella Luca","year":"2023","unstructured":"Luca Moschella, Valentino Maiorca, Marco Fumero, Antonio Norelli, Francesco Locatello, and Emanuele Rodol\u00e0. 2023. Relative representations enable zero-shot latent space communication. In Int. Conf. Learn. Represent. (ICLR). OpenReview.net, Kigali, Rwanda, 1-26."},{"key":"e_1_3_2_1_33_1","first-page":"23507","volume-title":"HOICLIP: Efficient Knowledge Transfer for HOI Detection with Vision-Language Models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Ning Shan","year":"2023","unstructured":"Shan Ning, Longtian Qiu, Yongfei Liu, and Xuming He. 2023. HOICLIP: Efficient Knowledge Transfer for HOI Detection with Vision-Language Models. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, Vancouver, BC, Canada, 23507-23517."},{"key":"e_1_3_2_1_34_1","first-page":"15303","volume-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"36","author":"Norelli Antonio","year":"2023","unstructured":"Antonio Norelli, Marco Fumero, Valentino Maiorca, Luca Moschella, Emanuele Rodol\u00e0, and Francesco Locatello. 2023. ASIF: Coupled Data Turns Unimodal Models to Multimodal without Training. In Adv. Neural Inf. Process. Syst. (NeurIPS), Vol. 36. Curran Associates, Inc., New Orleans, LA, USA, 15303-15319."},{"key":"e_1_3_2_1_35_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. CoRR Vol. abs\/2303.08774 (2023) 1-100."},{"key":"e_1_3_2_1_36_1","first-page":"17152","volume-title":"ViPLO: Vision Transformer Based Pose-Conditioned Self-Loop Graph for Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Park Jeeseung","year":"2023","unstructured":"Jeeseung Park, Jin-Woo Park, and Jong-Seok Lee. 2023. ViPLO: Vision Transformer Based Pose-Conditioned Self-Loop Graph for Human-Object Interaction Detection. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, Vancouver, BC, Canada, 17152-17162."},{"key":"e_1_3_2_1_37_1","first-page":"1","volume-title":"The Linear Representation Hypothesis and the Geometry of Large Language Models. In Int. Conf. Mach. Learn. (ICML). OpenReview.net","author":"Park Kiho","year":"2024","unstructured":"Kiho Park, Yo Joong Choe, and Victor Veitch. 2024. The Linear Representation Hypothesis and the Geometry of Large Language Models. In Int. Conf. Mach. Learn. (ICML). OpenReview.net, Vienna, Austria, 1-24."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1515\/jci-2021-0006"},{"key":"e_1_3_2_1_39_1","first-page":"490","article-title":"Lexical Semantics with Large Language Models: A Case Study of English ''break''. In","author":"Petersen Erika","year":"2023","unstructured":"Erika Petersen and Christopher Potts. 2023. Lexical Semantics with Large Language Models: A Case Study of English ''break''. In Find. Assoc. Comp. Linguist. (EACL). Association for Computational Linguistics, Dubrovnik, Croatia, 490-511.","journal-title":"Find. Assoc. Comp. Linguist. (EACL). Association for Computational Linguistics, Dubrovnik, Croatia"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01438"},{"key":"e_1_3_2_1_41_1","first-page":"8748","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In Int. Conf. Mach. Learn. (ICML)","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Int. Conf. Mach. Learn. (ICML), Vol. 139. PMLR, Virtual, 8748-8763."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"e_1_3_2_1_43_1","first-page":"17959","volume-title":"End-to-End Generative Pretraining for Multimodal Video Captioning. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Seo Paul Hongsuck","year":"2022","unstructured":"Paul Hongsuck Seo, Arsha Nagrani, Anurag Arnab, and Cordelia Schmid. 2022. End-to-End Generative Pretraining for Multimodal Video Captioning. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, New Orleans, LA, USA, 17959-17968."},{"key":"e_1_3_2_1_44_1","first-page":"10410","article-title":"QPIC: Query-Based Pairwise Human-Object Interaction Detection With Image-Wide Contextual Information. In IEEE","author":"Tamura Masato","year":"2021","unstructured":"Masato Tamura, Hiroki Ohashi, and Tomoaki Yoshinaga. 2021. QPIC: Query-Based Pairwise Human-Object Interaction Detection With Image-Wide Contextual Information. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE, Virtual, 10410-10419.","journal-title":"Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE, Virtual"},{"key":"e_1_3_2_1_45_1","first-page":"34431","volume-title":"Discrete Key-Value Bottleneck. In Int. Conf. Mach. Learn. (ICML). PMLR","author":"Tr\u00e4uble Frederik","year":"2023","unstructured":"Frederik Tr\u00e4uble, Anirudh Goyal, Nasim Rahaman, Michael Curtis Mozer, Kenji Kawaguchi, Yoshua Bengio, and Bernhard Sch\u00f6lkopf. 2023. Discrete Key-Value Bottleneck. In Int. Conf. Mach. Learn. (ICML). PMLR, Honolulu, Hawaii, USA, 34431-34455."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_6"},{"key":"e_1_3_2_1_47_1","first-page":"13614","volume-title":"VSGNet: Spatial Attention Network for Detecting Human Object Interactions Using Graph Convolutions. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE","author":"Ulutan Oytun","unstructured":"Oytun Ulutan, A. S. M. Iftekhar, and B. S. Manjunath. 2020. VSGNet: Spatial Attention Network for Detecting Human Object Interactions Using Graph Convolutions. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE, Seattle, WA, USA, 13614-13623."},{"key":"e_1_3_2_1_48_1","first-page":"5998","volume-title":"Adv. Neural Inf. Process. Syst. (NIPS)","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Adv. Neural Inf. Process. Syst. (NIPS), Vol. 30. Curran Associates, Inc., Long Beach, CA, USA, 5998-6008."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02642"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2015.2498132"},{"key":"e_1_3_2_1_51_1","first-page":"929","volume-title":"Learning Transferable Human-Object Interaction Detector with Natural Language Supervision. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE","author":"Wang Suchen","year":"2022","unstructured":"Suchen Wang, Yueqi Duan, Henghui Ding, Yap-Peng Tan, Kim-Hui Yap, and Junsong Yuan. 2022. Learning Transferable Human-Object Interaction Detector with Natural Language Supervision. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). IEEE, New Orleans, LA, USA, 929-938."},{"key":"e_1_3_2_1_52_1","first-page":"4115","volume-title":"Learning Human-Object Interaction Detection Using Interaction Points. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE","author":"Wang Tiancai","year":"2020","unstructured":"Tiancai Wang, Tong Yang, Martin Danelljan, Fahad Shahbaz Khan, Xiangyu Zhang, and Jian Sun. 2020. Learning Human-Object Interaction Detection Using Interaction Points. In IEEE Conf. Comput. Vis. Pattern Recog. (CVPR). Computer Vision Foundation \/ IEEE, Seattle, WA, USA, 4115-4124."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01687"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25385"},{"key":"e_1_3_2_1_55_1","first-page":"6066","volume-title":"Toward Open-Set Human Object Interaction Detection. In AAAI Conf. Artif. Intell. (AAAI). AAAI Press","author":"Wu Mingrui","year":"2024","unstructured":"Mingrui Wu, Yuqi Liu, Jiayi Ji, Xiaoshuai Sun, and Rongrong Ji. 2024b. Toward Open-Set Human Object Interaction Detection. In AAAI Conf. Artif. Intell. (AAAI). AAAI Press, Vancouver, Canada, 6066-6073."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2025.3531452"},{"key":"e_1_3_2_1_57_1","first-page":"12996","article-title":"Deconfounded Image Captioning","volume":"45","author":"Yang Xu","year":"2023","unstructured":"Xu Yang, Hanwang Zhang, and Jianfei Cai. 2023. Deconfounded Image Captioning: A Causal Retrospect. IEEE Trans. Pattern Anal. Mach. Intell., Vol. 45, 11 (2023), 12996-13010.","journal-title":"A Causal Retrospect. IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"e_1_3_2_1_58_1","volume-title":"Int. Conf. Learn. Represent. (ICLR). OpenReview.net, Kigali, Rwanda, 1-20","author":"Y\u00fcksekg\u00f6n\u00fcl Mert","year":"2023","unstructured":"Mert Y\u00fcksekg\u00f6n\u00fcl, Federico Bianchi, Pratyusha Kalluri, Dan Jurafsky, and James Zou. 2023. When and Why Vision-Language Models Behave like Bags-Of-Words, and What to Do About It?. In Int. Conf. Learn. Represent. (ICLR). OpenReview.net, Kigali, Rwanda, 1-20."},{"key":"e_1_3_2_1_59_1","first-page":"17209","volume-title":"Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"34","author":"Zhang Aixi","year":"2021","unstructured":"Aixi Zhang, Yue Liao, Si Liu, Miao Lu, Yongliang Wang, Chen Gao, and Xiaobo Li. 2021b. Mining the Benefits of Two-stage and One-stage HOI Detection. In Adv. Neural Inf. Process. Syst. (NeurIPS), Vol. 34. Curran Associates, Inc., Virtual, 17209-17220."},{"key":"e_1_3_2_1_60_1","first-page":"310","volume-title":"Long-CLIP: Unlocking the Long-Text Capability of CLIP. In European Conf. Comput. Vis. (ECCV)","volume":"15109","author":"Zhang Beichen","year":"2024","unstructured":"Beichen Zhang, Pan Zhang, Xiaoyi Dong, Yuhang Zang, and Jiaqi Wang. 2024. Long-CLIP: Unlocking the Long-Text Capability of CLIP. In European Conf. Comput. Vis. (ECCV), Vol. 15109. Springer, Milan, Italy, 310-325."},{"key":"e_1_3_2_1_61_1","volume-title":"Spatially Conditioned Graphs for Detecting Human-Object Interactions. In IEEE Int. Conf. Comput. Vis. (ICCV). IEEE, Montreal, QC, Canada, 13299-13307","author":"Zhang Frederic Z.","year":"2021","unstructured":"Frederic Z. Zhang, Dylan Campbell, and Stephen Gould. 2021a. Spatially Conditioned Graphs for Detecting Human-Object Interactions. In IEEE Int. Conf. Comput. Vis. (ICCV). IEEE, Montreal, QC, Canada, 13299-13307."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01947"},{"key":"e_1_3_2_1_63_1","first-page":"6939","volume-title":"Unified Visual Relationship Detection with Vision and Language Models. In IEEE Int. Conf. Comput. Vis. (ICCV). IEEE","author":"Zhao Long","year":"2023","unstructured":"Long Zhao, Liangzhe Yuan, Boqing Gong, Yin Cui, Florian Schroff, Ming-Hsuan Yang, Hartwig Adam, and Ting Liu. 2023. Unified Visual Relationship Detection with Vision and Language Models. In IEEE Int. Conf. Comput. Vis. (ICCV). IEEE, Paris, France, 6939-6950."},{"key":"e_1_3_2_1_64_1","first-page":"444","volume-title":"Towards Hard-Positive Query Mining for DETR-Based Human-Object Interaction Detection. In European Conf. Comput. Vis. (ECCV). Springer","author":"Zhong Xubin","year":"2022","unstructured":"Xubin Zhong, Changxing Ding, Zijian Li, and Shaoli Huang. 2022. Towards Hard-Positive Query Mining for DETR-Based Human-Object Interaction Detection. In European Conf. Comput. Vis. (ECCV). Springer, Tel Aviv, Israel, 444-460."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01896"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754770","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:57:32Z","timestamp":1765342652000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754770"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":65,"alternative-id":["10.1145\/3746027.3754770","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754770","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}