{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,20]],"date-time":"2026-02-20T16:17:53Z","timestamp":1771604273136,"version":"3.50.1"},"reference-count":71,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T00:00:00Z","timestamp":1766534400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T00:00:00Z","timestamp":1766534400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62225207"],"award-info":[{"award-number":["62225207"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62436008"],"award-info":[{"award-number":["62436008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476260"],"award-info":[{"award-number":["62476260"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1007\/s11263-025-02616-y","type":"journal-article","created":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T03:42:53Z","timestamp":1766547773000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mamba-Driven Comprehensive Context Learning for Zero-Shot HOI Detection"],"prefix":"10.1007","volume":"134","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9940-6366","authenticated-orcid":false,"given":"Jiawei","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongchao","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sen","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuexuan","family":"Qi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng-Jun","family":"Zha","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,24]]},"reference":[{"key":"2616_CR1","unstructured":"Biderman, S., Schoelkopf, H., and Anthony, Q.G., Bradley, H., O\u2019Brien, K., Hallahan, E., Khan, M.A., Purohit, S., Prashanth, U.S., Raff, E., & others (2023). Pythia: A suite for analyzing large language models across training and scaling. In: International Conference on Machine Learning, PMLR, pp 2397\u20132430"},{"key":"2616_CR2","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al. (2020). Language models are few-shot learners. Advances in Neural Information Processing Systems, 33, 1877\u20131901.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR3","first-page":"739","volume":"36","author":"Y Cao","year":"2023","unstructured":"Cao, Y., Tang, Q., Su, X., Chen, S., You, S., Lu, X., & Xu, C. (2023). Detecting any human-object interaction relationship: Universal hoi detector with spatial prompt learning on foundation models. Advances in Neural Information Processing Systems, 36, 739\u2013751.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR4","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020). End-to-end object detection with transformers. In: European Conference on Computer Vision, Springer, pp 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2616_CR5","doi-asserted-by":"crossref","unstructured":"Chao, Y-W., Liu, Y., Liu, X., Zeng, H., & Deng, J. (2018). Learning to detect human-object interactions. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, IEEE, pp 381\u2013389","DOI":"10.1109\/WACV.2018.00048"},{"key":"2616_CR6","unstructured":"Dao, T., & Gu, A. (2024). Transformers are ssms: generalized models and efficient algorithms through structured state space duality. In: International Conference on Machine Learning, pp 10041\u201310071"},{"key":"2616_CR7","unstructured":"Dong, L., Yang, N., Wang, W., Wei, F., Liu, X., Wang, Y., Gao, J., Zhou, M., & Hon, H.W. (2019). Unified language model pre-training for natural language understanding and generation. Advances in Neural Information Processing Systems 32"},{"key":"2616_CR8","unstructured":"Gao, C., Zou, Y., & Huang, J.B. (2018). ican: Instance-centric attention network for human-object interaction detection. arXiv preprint arXiv:1808.10437"},{"key":"2616_CR9","doi-asserted-by":"crossref","unstructured":"Geng, P., Yang, J., & Zhang, S. (2025). Horp: Human-object relation priors guided hoi detection. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp 25325\u201325335.","DOI":"10.1109\/CVPR52734.2025.02358"},{"key":"2616_CR10","unstructured":"Gu, A., & Dao, T. (2023). Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752"},{"key":"2616_CR11","unstructured":"Gu, A., Goel, K., & R\u00e9, C. (2021a). Efficiently modeling long sequences with structured state spaces. arXiv preprint arXiv:2111.00396"},{"key":"2616_CR12","first-page":"572","volume":"34","author":"A Gu","year":"2021","unstructured":"Gu, A., Johnson, I., Goel, K., Saab, K., Dao, T., Rudra, A., & R\u00e9, C. (2021). Combining recurrent, convolutional, and continuous-time models with linear state space layers. Advances in Neural Information Processing Systems, 34, 572\u2013585.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR13","doi-asserted-by":"crossref","unstructured":"Guo, H., Li, J., Dai, T., Ouyang, Z., Ren, X., & Xia, S.-T. (2024a). Mambair: A simple baseline for image restoration with state-space model. In: European Conference on Computer Vision, Springer, pp 222\u2013241.","DOI":"10.1007\/978-3-031-72649-1_13"},{"key":"2616_CR14","doi-asserted-by":"crossref","unstructured":"Guo, Y., Liu, Y., Li, J., Wang, W., & Jia, Q. (2024b). Unseen no more: Unlocking the potential of clip for generative zero-shot hoi detection. In: Proceedings of the 32nd ACM International Conference on Multimedia, pp 1711\u20131720.","DOI":"10.1145\/3664647.3680927"},{"key":"2616_CR15","unstructured":"Gupta, S., & Malik, J. (2015). Visual semantic role labeling. arXiv preprint arXiv:1505.04474"},{"key":"2616_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2616_CR17","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R. (2017). Mask r-cnn. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 2961\u20132969.","DOI":"10.1109\/ICCV.2017.322"},{"key":"2616_CR18","doi-asserted-by":"crossref","unstructured":"Hou, Z., Peng, X., Qiao, Y., & Tao, D. (2020). Visual compositional learning for human-object interaction detection. In: European Conference on Computer Vision, Springer, pp 584\u2013600","DOI":"10.1007\/978-3-030-58555-6_35"},{"key":"2616_CR19","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., Qiao, Y., Peng, X., & Tao, D. (2021a). Affordance transfer learning for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 495\u2013504","DOI":"10.1109\/CVPR46437.2021.00056"},{"key":"2616_CR20","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., Qiao, Y., Peng, X., & Tao, D. (2021b). Detecting human-object interaction via fabricated compositional learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 14646\u201314655","DOI":"10.1109\/CVPR46437.2021.01441"},{"key":"2616_CR21","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., & Tao, D. (2022). Discovering human-object interaction concepts via self-compositional learning. In: European Conference on Computer Vision, Springer, pp 461\u2013478","DOI":"10.1007\/978-3-031-19812-0_27"},{"key":"2616_CR22","doi-asserted-by":"crossref","unstructured":"Huang, T., Pei, X., You, S., Wang, F., Qian, C., Xu, C. (2024a). Localmamba: Visual state space model with windowed selective scan. In: European Conference on Computer Vision, Springer, pp 12\u201322","DOI":"10.1007\/978-3-031-91979-4_2"},{"issue":"9","key":"2616_CR23","doi-asserted-by":"publisher","first-page":"3977","DOI":"10.1007\/s11263-024-02017-7","volume":"132","author":"Y Huang","year":"2024","unstructured":"Huang, Y., Yang, L., Chen, G., Zhang, H., Lu, F., & Sato, Y. (2024). Matching compound prototypes for few-shot action recognition. International Journal of Computer Vision, 132(9), 3977\u20134002.","journal-title":"International Journal of Computer Vision"},{"key":"2616_CR24","doi-asserted-by":"crossref","unstructured":"Kim, B., Lee, J., Kang, J., Kim, E.-S., & Kim, H.J. (2021). Hotr: End-to-end human-object interaction detection with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 74\u201383.","DOI":"10.1109\/CVPR46437.2021.00014"},{"key":"2616_CR25","doi-asserted-by":"crossref","unstructured":"Kim, B., Mun, J., On, K.-W., Shin, M., Lee, J., & Kim, E.-S. (2022). Mstr: Multi-scale transformer for end-to-end human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 19578\u201319587.","DOI":"10.1109\/CVPR52688.2022.01897"},{"key":"2616_CR26","doi-asserted-by":"crossref","unstructured":"Kim, S., Jung, D., & Cho, M. (2023). Relational context learning for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 2925\u20132934.","DOI":"10.1109\/CVPR52729.2023.00286"},{"key":"2616_CR27","first-page":"55831","volume":"37","author":"Q Lei","year":"2025","unstructured":"Lei, Q., Wang, B., & Tan, R. (2025). Ez-hoi: Vlm adaptation via guided prompt learning for zero-shot hoi detection. Advances in Neural Information Processing Systems, 37, 55831\u201355857.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR28","doi-asserted-by":"crossref","unstructured":"Lei, T., Caba, F., Chen, Q., Jin, H., Peng, Y., & Liu, Y. (2023). Efficient adaptive human-object interaction detection with concept-guided memory. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 6480\u20136490","DOI":"10.1109\/ICCV51070.2023.00596"},{"key":"2616_CR29","doi-asserted-by":"crossref","unstructured":"Lei, T., Yin, S., Peng, Y., & Liu, Y. (2025b). Exploring conditional multi-modal prompts for zero-shot hoi detection. In: European Conference on Computer Vision, Springer, pp 1\u201319.","DOI":"10.1007\/978-3-031-73007-8_1"},{"key":"2616_CR30","first-page":"7290","volume":"35","author":"J Li","year":"2022","unstructured":"Li, J., He, X., Wei, L., Qian, L., Zhu, L., Xie, L., Zhuang, Y., Tian, Q., & Tang, S. (2022). Fine-grained semantically aligned vision-language pre-training. Advances in Neural Information Processing Systems, 35, 7290\u20137303.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR31","unstructured":"Li, J., Li, D., Xiong, C., & Hoi, S. (2022b). Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, PMLR, pp 12888\u201312900."},{"key":"2616_CR32","unstructured":"Li, J., Li, D., Savarese, S., & Hoi, S. (2023). Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, PMLR, pp 19730\u201319742."},{"key":"2616_CR33","doi-asserted-by":"crossref","unstructured":"Liao, Y., Liu, S., Wang, F., Chen, Y., Qian, C., & Feng, J. (2020). Ppdm: Parallel point detection and matching for real-time human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 482\u2013490.","DOI":"10.1109\/CVPR42600.2020.00056"},{"key":"2616_CR34","doi-asserted-by":"crossref","unstructured":"Liao, Y., Zhang, A., Lu, M., Wang, Y., Li, X., & Liu, S. (2022). Gen-vlkt: Simplify association and enhance interaction understanding for hoi detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 20123\u201320132.","DOI":"10.1109\/CVPR52688.2022.01949"},{"key":"2616_CR35","doi-asserted-by":"crossref","unstructured":"Lin, B., Jiang, W., Chen, P., Zhang, Y., Liu, S., & Chen, Y.-C. (2024). Mtmamba: Enhancing multi-task dense scene understanding by mamba-based decoders. In: European Conference on Computer Vision, Springer, pp 314\u2013330","DOI":"10.1007\/978-3-031-72897-6_18"},{"key":"2616_CR36","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C.L. (2014). Microsoft coco: Common objects in context. In: European Conference on Computer Vision, Springer, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2616_CR37","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., & Doll\u00e1r, P. (2017). Focal loss for dense object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 2980\u20132988","DOI":"10.1109\/ICCV.2017.324"},{"key":"2616_CR38","doi-asserted-by":"crossref","unstructured":"Liu, J., Zha, Z.-J., Chen, D., Hong, R., & Wang, M. (2019). Adaptive transfer network for cross-domain person re-identification. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 7202\u20137211.","DOI":"10.1109\/CVPR.2019.00737"},{"key":"2616_CR39","first-page":"103031","volume":"37","author":"Y Liu","year":"2024","unstructured":"Liu, Y., Tian, Y., Zhao, Y., Yu, H., Xie, L., Wang, Y., Ye, Q., Jiao, J., & Liu, Y. (2024). Vmamba: Visual state space model. Advances in Neural Information Processing Systems, 37, 103031\u2013103063.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR40","unstructured":"Loshchilov, I., & Hutter, F. (2017). Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101"},{"key":"2616_CR41","doi-asserted-by":"crossref","unstructured":"Luo, J., Ren, W., Jiang, W., Chen, X., Wang, Q., Han, Z., & Liu, H. (2024). Discovering syntactic interaction clues for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 28212\u201328222","DOI":"10.1109\/CVPR52733.2024.02665"},{"key":"2616_CR42","first-page":"45895","volume":"36","author":"Y Mao","year":"2023","unstructured":"Mao, Y., Deng, J., Zhou, W., Li, L., Fang, Y., & Li, H. (2023). Clip4hoi: towards adapting clip for practical zero-shot hoi detection. Advances in Neural Information Processing Systems, 36, 45895\u201345906.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR43","doi-asserted-by":"crossref","unstructured":"Ning, S., Qiu, L., Liu, Y., & He, X. (2023). Hoiclip: Efficient knowledge transfer for hoi detection with vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 23507\u201323517.","DOI":"10.1109\/CVPR52729.2023.02251"},{"key":"2616_CR44","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., & others. (2021). Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, PMLR, pp 8748\u20138763."},{"issue":"6","key":"2616_CR45","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., & Sun, J. (2016). Faster r-cnn: Towards real-time object detection with region proposal networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 39(6), 1137\u20131149.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2616_CR46","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., & Rabinovich, A. (2015). Going deeper with convolutions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1\u20139.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"2616_CR47","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., & Wojna, Z. (2016). Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 2818\u20132826.","DOI":"10.1109\/CVPR.2016.308"},{"key":"2616_CR48","doi-asserted-by":"crossref","unstructured":"Tamura, M., Ohashi, H., & Yoshinaga, T. (2021). Qpic: Query-based pairwise human-object interaction detection with image-wide contextual information. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 10410\u201310419.","DOI":"10.1109\/CVPR46437.2021.01027"},{"key":"2616_CR49","unstructured":"Tan, M., & Le, Q.V. (2019). Mixconv: Mixed depthwise convolutional kernels. arXiv preprint arXiv:1907.09595"},{"key":"2616_CR50","doi-asserted-by":"crossref","unstructured":"Ulutan, O., Iftekhar, A., & Manjunath, B.S. (2020). Vsgnet: Spatial attention network for detecting human object interactions using graph convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 13617\u201313626.","DOI":"10.1109\/CVPR42600.2020.01363"},{"key":"2616_CR51","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. Advances in Neural Information Processing Systems 30."},{"key":"2616_CR52","doi-asserted-by":"crossref","unstructured":"Ventura, L., Schmid, C., & Varol, G. (2024). Learning text-to-video retrieval from image captioning. International Journal of Computer Vision pp 1\u201321.","DOI":"10.1007\/s11263-024-02202-8"},{"key":"2616_CR53","doi-asserted-by":"crossref","unstructured":"Wang, G., Guo, Y., Xu, Z., & Kankanhalli, M. (2024). Bilateral adaptation for human-object interaction detection with occlusion-robustness. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 27970\u201327980.","DOI":"10.1109\/CVPR52733.2024.02642"},{"key":"2616_CR54","doi-asserted-by":"crossref","unstructured":"Wang, T., Yang, T., Danelljan, M., Khan, F.S., Zhang, X., & Sun, J. (2020). Learning human-object interaction detection using interaction points. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 4116\u20134125.","DOI":"10.1109\/CVPR42600.2020.00417"},{"key":"2616_CR55","doi-asserted-by":"crossref","unstructured":"Wang, Y., Liu, X., Yan, T., Liu, Y., Zheng, A., Zhang, P., & Lu, H. (2025). Mambapro: Multi-modal object re-identification with mamba aggregation and synergistic prompt. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 8150\u20138158.","DOI":"10.1609\/aaai.v39i8.32879"},{"key":"2616_CR56","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q. V., Zhou, D., et al. (2022). Chain-of-thought prompting elicits reasoning in large language models. Advances in Neural Information Processing Systems, 35, 24824\u201324837.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR57","unstructured":"Williams, C.K. (2024). Structured generative models for scene understanding. International Journal of Computer Vision pp 1\u201323."},{"key":"2616_CR58","doi-asserted-by":"crossref","unstructured":"Wu, E.Z.Y., Li, Y., Wang, Y., & Wang, S. (2024). Exploring pose-aware human-object interaction via hybrid learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 17815\u201317825.","DOI":"10.1109\/CVPR52733.2024.01687"},{"key":"2616_CR59","doi-asserted-by":"crossref","unstructured":"Wu, M., Gu, J., Shen, Y., Lin, M., Chen, C., & Sun, X. (2023). End-to-end zero-shot hoi detection via vision and language knowledge distillation. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 2839\u20132846>","DOI":"10.1609\/aaai.v37i3.25385"},{"key":"2616_CR60","unstructured":"Chaodong X., Minghan L., Zhengqiang Z., Deyu M., & Lei Z. (2025). Spatial-mamba: Effective visual state space models via structure-aware state fusion. In: International Conference on Learning Representations"},{"key":"2616_CR61","doi-asserted-by":"crossref","unstructured":"Xu, Y., Liu, J., Tao, S., Zhang, Q., & Zha, Z.-J. (2025). Hoimamba: Efficient mamba-based disentangled progressive learning for hoi detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 8987\u20138995","DOI":"10.1609\/aaai.v39i9.32972"},{"key":"2616_CR62","unstructured":"Yao, L., Huang, R., Hou, L., Lu, G., Niu, M., Xu, H., Liang, X., Li, Z., Jiang, X., & Xu, C. (2022). Filip: Fine-grained interactive language-image pre-training. In: International Conference on Learning Representations"},{"key":"2616_CR63","unstructured":"Yue, Y., & Li, Z. (2024). Medmamba: Vision mamba for medical image classification. arXiv preprint arXiv:2403.03849"},{"key":"2616_CR64","doi-asserted-by":"crossref","unstructured":"Zha, Z.-J., Liu, J., Yang, T., & Zhang, Y. (2019). Spatiotemporal-textual co-attention network for video question answering. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM) 15(2s):1\u201318.","DOI":"10.1145\/3320061"},{"key":"2616_CR65","first-page":"17209","volume":"34","author":"A Zhang","year":"2021","unstructured":"Zhang, A., Liao, Y., Liu, S., Lu, M., Wang, Y., Gao, C., & Li, X. (2021). Mining the benefits of two-stage and one-stage hoi detection. Advances in Neural Information Processing Systems, 34, 17209\u201317220.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2616_CR66","doi-asserted-by":"crossref","unstructured":"Zhang, F.Z., Campbell, D., & Gould, S. (2022). Efficient two-stage detection of human-object interactions with a novel unary-pairwise transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 20104\u201320112.","DOI":"10.1109\/CVPR52688.2022.01947"},{"key":"2616_CR67","doi-asserted-by":"crossref","unstructured":"Zhang, F.Z., Yuan, Y., Campbell, D., Zhong, Z., & Gould, S. (2023). Exploring predicate visual context in detecting of human-object interactions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 10411\u201310421.","DOI":"10.1109\/ICCV51070.2023.00955"},{"key":"2616_CR68","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Lv, W., Xu, S., Wei, J., Wang, G., Dang, Q., Liu, Y., & Chen, J. (2024). Detrs beat yolos on real-time object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 16965\u201316974.","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"2616_CR69","doi-asserted-by":"crossref","unstructured":"Zhong, X., Ding, C., Qu, X., & Tao, D. (2021). Polysemy deciphering network for robust human-object interaction detection. International Journal of Computer Vision, 129(6), 1910\u20131929.","DOI":"10.1007\/s11263-021-01458-8"},{"key":"2616_CR70","doi-asserted-by":"crossref","unstructured":"Zhu, F., Xie, Y., Xie, W., & Jiang, H. (2025). Diagnosing human-object interaction detectors. International Journal of Computer Vision pp 1\u201318","DOI":"10.1007\/s11263-025-02369-8"},{"key":"2616_CR71","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., & Dai, J. (2021). Deformable detr: Deformable transformers for end-to-end object detection. In: International Conference on Learning Representations."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02616-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02616-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02616-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,20]],"date-time":"2026-02-20T15:42:53Z","timestamp":1771602173000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02616-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,24]]},"references-count":71,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["2616"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02616-y","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,24]]},"assertion":[{"value":"17 July 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"10"}}