{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T15:15:31Z","timestamp":1778858131469,"version":"3.51.4"},"reference-count":62,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T00:00:00Z","timestamp":1776211200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T00:00:00Z","timestamp":1776211200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Machine Vision and Applications"],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1007\/s00138-026-01818-9","type":"journal-article","created":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T04:32:19Z","timestamp":1776227539000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Funnel-HOI: top-down perception for zero-shot HOI detection"],"prefix":"10.1007","volume":"37","author":[{"given":"Sandipan","family":"Sarma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Agney","family":"Talwarr","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arijit","family":"Sur","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,4,15]]},"reference":[{"key":"1818_CR1","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Girshick, R., Doll\u00e1r, P., He, K.: Detecting and recognizing human-object interactions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8359\u20138367 (2018)","DOI":"10.1109\/CVPR.2018.00872"},{"key":"1818_CR2","doi-asserted-by":"publisher","first-page":"9150","DOI":"10.1109\/TIP.2021.3113563","volume":"30","author":"D-J Kim","year":"2021","unstructured":"Kim, D.-J., Sun, X., Choi, J., Lin, S., Kweon, I.S.: Acp++: action co-occurrence priors for human-object interaction detection. IEEE Trans. Image Process. 30, 9150\u20139163 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"1818_CR3","doi-asserted-by":"crossref","unstructured":"Shen, L., Yeung, S., Hoffman, J., Mori, G., Fei-Fei, L.: Scaling human-object interaction recognition through zero-shot learning. In: 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 1568\u20131576 (2018). IEEE","DOI":"10.1109\/WACV.2018.00181"},{"key":"1818_CR4","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229 (2020). Springer","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1818_CR5","doi-asserted-by":"crossref","unstructured":"Liao, Y., Zhang, A., Lu, M., Wang, Y., Li, X., Liu, S.: Gen-vlkt: Simplify association and enhance interaction understanding for hoi detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20123\u201320132 (2022)","DOI":"10.1109\/CVPR52688.2022.01949"},{"key":"1818_CR6","doi-asserted-by":"crossref","unstructured":"Ning, S., Qiu, L., Liu, Y., He, X.: Hoiclip: efficient knowledge transfer for hoi detection with vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23507\u201323517 (2023)","DOI":"10.1109\/CVPR52729.2023.02251"},{"issue":"4","key":"1818_CR7","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1007\/s00138-024-01558-8","volume":"35","author":"L Xia","year":"2024","unstructured":"Xia, L., Xiao, Q.: Human-object interaction detection based on disentangled axial attention transformer. Mach. Vis. Appl. 35(4), 72 (2024)","journal-title":"Mach. Vis. Appl."},{"key":"1818_CR8","first-page":"2839","volume":"37","author":"M Wu","year":"2023","unstructured":"Wu, M., Gu, J., Shen, Y., Lin, M., Chen, C., Sun, X.: End-to-end zero-shot hoi detection via vision and language knowledge distillation. Proc. AAAI Conf. Artif. Intell. 37, 2839\u20132846 (2023)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"issue":"23","key":"1818_CR9","doi-asserted-by":"publisher","first-page":"12492","DOI":"10.1007\/s10489-024-05774-7","volume":"54","author":"K Xue","year":"2024","unstructured":"Xue, K., Gao, Y., Fang, Z., Jiang, X., Yu, W., Chen, M., Wu, C.: Adaptive multimodal prompt for human-object interaction with local feature enhanced transformer. Appl. Intell. 54(23), 12492\u201312504 (2024)","journal-title":"Appl. Intell."},{"key":"1818_CR10","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021). PMLR"},{"issue":"19","key":"1818_CR11","doi-asserted-by":"publisher","first-page":"9008","DOI":"10.1007\/s10489-024-05653-1","volume":"54","author":"X Wang","year":"2024","unstructured":"Wang, X., Gao, Y., Yu, W., Wu, C., Chen, M., Ma, H., Chen, Z.: QLDT: adaptive query learning for hoi detection via vision-language knowledge transfer. Appl. Intell. 54(19), 9008\u20139027 (2024)","journal-title":"Appl. Intell."},{"issue":"3","key":"1818_CR12","doi-asserted-by":"publisher","first-page":"2831","DOI":"10.1007\/s10489-024-05324-1","volume":"54","author":"L Xia","year":"2024","unstructured":"Xia, L., Ding, X.: Human-object interaction detection based on cascade multi-scale transformer: L. Xia and X. Ding. Appl. Intell. 54(3), 2831\u20132850 (2024)","journal-title":"Appl. Intell."},{"key":"1818_CR13","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"1818_CR14","doi-asserted-by":"publisher","first-page":"636","DOI":"10.3389\/fpsyg.2014.00636","volume":"5","author":"RI Schubotz","year":"2014","unstructured":"Schubotz, R.I., Wurm, M.F., Wittmann, M.K., Cramon, D.Y.: Objects tell us what action we can expect: dissociating brain areas for retrieval and exploitation of action knowledge during action observation in fmri. Front. Psychol. 5, 636 (2014)","journal-title":"Front. Psychol."},{"issue":"1","key":"1818_CR15","doi-asserted-by":"publisher","first-page":"25","DOI":"10.5334\/joc.28","volume":"1","author":"N Gaspelin","year":"2018","unstructured":"Gaspelin, N., Luck, S.J.: Top-down\u2019\u2019 does not mean \u201cvoluntary. J. Cogn. 1(1), 25 (2018)","journal-title":"J. Cogn."},{"key":"1818_CR16","doi-asserted-by":"crossref","unstructured":"Chao, Y.-W., Liu, Y., Liu, X., Zeng, H., Deng, J.: Learning to detect human-object interactions. In: 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 381\u2013389 (2018). IEEE","DOI":"10.1109\/WACV.2018.00048"},{"key":"1818_CR17","unstructured":"Gupta, S., Malik, J.: Visual semantic role labeling. arXiv preprint arXiv:1505.04474 (2015)"},{"key":"1818_CR18","unstructured":"Gao, C., Zou, Y., Huang, J.-B.: ICAN: instance-centric attention network for human-object interaction detection. In: British Machine Vision Conference (2018)"},{"issue":"6","key":"1818_CR19","doi-asserted-by":"publisher","first-page":"1423","DOI":"10.1109\/TMM.2019.2943753","volume":"22","author":"B Xu","year":"2019","unstructured":"Xu, B., Li, J., Wong, Y., Zhao, Q., Kankanhalli, M.S.: Interact as you intend: intention-driven human-object interaction detection. IEEE Trans. Multimed. 22(6), 1423\u20131432 (2019)","journal-title":"IEEE Trans. Multimed."},{"key":"1818_CR20","doi-asserted-by":"crossref","unstructured":"Hou, Z., Peng, X., Qiao, Y., Tao, D.: Visual compositional learning for human\u2013object interaction detection. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XV 16, pp. 584\u2013600 (2020). Springer","DOI":"10.1007\/978-3-030-58555-6_35"},{"key":"1818_CR21","doi-asserted-by":"crossref","unstructured":"Gao, C., Xu, J., Zou, Y., Huang, J.-B.: DRG: dual relation graph for human-object interaction detection. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XII 16, pp. 696\u2013712 (2020). Springer","DOI":"10.1007\/978-3-030-58610-2_41"},{"key":"1818_CR22","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., Qiao, Y., Peng, X., Tao, D.: Affordance transfer learning for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 495\u2013504 (2021)","DOI":"10.1109\/CVPR46437.2021.00056"},{"key":"1818_CR23","doi-asserted-by":"publisher","first-page":"8306","DOI":"10.1109\/TIP.2021.3093784","volume":"30","author":"Y Gao","year":"2021","unstructured":"Gao, Y., Kuang, Z., Li, G., Zhang, W., Lin, L.: Hierarchical reasoning network for human-object interaction detection. IEEE Trans. Image Process. 30, 8306\u20138317 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"1818_CR24","doi-asserted-by":"publisher","first-page":"6583","DOI":"10.1109\/TIP.2021.3096333","volume":"30","author":"H Wang","year":"2021","unstructured":"Wang, H., Jiao, L., Liu, F., Li, L., Liu, X., Ji, D., Gan, W.: IPGN: Interactiveness proposal graph network for human-object interaction detection. IEEE Trans. Image Process. 30, 6583\u20136593 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"1818_CR25","doi-asserted-by":"publisher","first-page":"6826","DOI":"10.1109\/TPAMI.2024.3386891","volume":"46","author":"Y Liao","year":"2024","unstructured":"Liao, Y., Liu, S., Gao, Y., Zhang, A., Li, Z., Wang, F., Li, B.: PPDM++: parallel point detection and matching for fast and accurate hoi detection. IEEE Trans. Pattern Anal. Mach. Intell. 46, 6826 (2024)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1818_CR26","first-page":"6066","volume":"38","author":"M Wu","year":"2024","unstructured":"Wu, M., Liu, Y., Ji, J., Sun, X., Ji, R.: Toward open-set human object interaction detection. Proc. AAAI Conf. Artif. Intell. 38, 6066\u20136073 (2024)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"1818_CR27","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. Advances in neural information processing systems 28 (2015)"},{"key":"1818_CR28","doi-asserted-by":"publisher","first-page":"104162","DOI":"10.1016\/j.cviu.2024.104162","volume":"249","author":"Q Ye","year":"2024","unstructured":"Ye, Q., Xu, X., Li, R., Zhang, Y.: Human-object interaction detection algorithm based on graph structure and improved cascade pyramid network. Comput. Vis. Image Underst. 249, 104162 (2024)","journal-title":"Comput. Vis. Image Underst."},{"key":"1818_CR29","doi-asserted-by":"crossref","unstructured":"Hui, X., Qu, H., Rahmani, H., Liu, J.: An image-like diffusion method for human-object interaction detection. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 14002\u201314012 (2025)","DOI":"10.1109\/CVPR52734.2025.01307"},{"key":"1818_CR30","doi-asserted-by":"crossref","unstructured":"Geng, P., Yang, J., Zhang, S.: Horp: human-object relation priors guided hoi detection. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 25325\u201325335 (2025)","DOI":"10.1109\/CVPR52734.2025.02358"},{"key":"1818_CR31","doi-asserted-by":"crossref","unstructured":"Tamura, M., Ohashi, H., Yoshinaga, T.: Qpic: query-based pairwise human-object interaction detection with image-wide contextual information. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10410\u201310419 (2021)","DOI":"10.1109\/CVPR46437.2021.01027"},{"key":"1818_CR32","first-page":"17209","volume":"34","author":"A Zhang","year":"2021","unstructured":"Zhang, A., Liao, Y., Liu, S., Lu, M., Wang, Y., Gao, C., Li, X.: Mining the benefits of two-stage and one-stage hoi detection. Adv. Neural. Inf. Process. Syst. 34, 17209\u201317220 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1818_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, F.Z., Campbell, D., Gould, S.: Efficient two-stage detection of human-object interactions with a novel unary-pairwise transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20104\u201320112 (2022)","DOI":"10.1109\/CVPR52688.2022.01947"},{"key":"1818_CR34","doi-asserted-by":"publisher","first-page":"2415","DOI":"10.1109\/TPAMI.2023.3331738","volume":"46","author":"S Ma","year":"2023","unstructured":"Ma, S., Wang, Y., Wang, S., Wei, Y.: Fgahoi: Fine-grained anchors for human-object interaction detection. IEEE Trans. Pattern Anal. Mach. Intell. 46, 2415 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1818_CR35","doi-asserted-by":"publisher","first-page":"17805","DOI":"10.1109\/TNNLS.2023.3309104","volume":"35","author":"D Zong","year":"2023","unstructured":"Zong, D., Sun, S.: Zero-shot human-object interaction detection via similarity propagation. IEEE Trans. Neural Netw. Learn. Syst. 35, 17805 (2023)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"1818_CR36","doi-asserted-by":"publisher","first-page":"4237","DOI":"10.1109\/TETCI.2024.3392687","volume":"8","author":"S Chan","year":"2024","unstructured":"Chan, S., Wang, W., Shao, Z., Wang, Z., Bai, C.: Region mining and refined query improved hoi detection in transformer. IEEE Trans. Emerg. Top. Comput. Intell. 8, 4237 (2024)","journal-title":"IEEE Trans. Emerg. Top. Comput. Intell."},{"key":"1818_CR37","doi-asserted-by":"publisher","first-page":"110021","DOI":"10.1016\/j.patcog.2023.110021","volume":"146","author":"Y Cheng","year":"2024","unstructured":"Cheng, Y., Duan, H., Wang, C., Chen, Z.: Parallel disentangling network for human\u2013object interaction detection. Pattern Recogn. 146, 110021 (2024)","journal-title":"Pattern Recogn."},{"key":"1818_CR38","doi-asserted-by":"publisher","first-page":"111075","DOI":"10.1016\/j.patcog.2024.111075","volume":"159","author":"M Sun","year":"2025","unstructured":"Sun, M., Suo, W., Wang, J., Wang, P., Zhang, Y.: CHA: conditional hyper-adapter method for detecting human-object interaction. Pattern Recogn. 159, 111075 (2025)","journal-title":"Pattern Recogn."},{"key":"1818_CR39","first-page":"10460","volume":"34","author":"A Bansal","year":"2020","unstructured":"Bansal, A., Rambhatla, S.S., Shrivastava, A., Chellappa, R.: Detecting human-object interactions via functional generalization. Proc. AAAI Conf. Artif. Intell. 34, 10460\u201310469 (2020)","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"1818_CR40","doi-asserted-by":"crossref","unstructured":"Liu, Y., Yuan, J., Chen, C.W.: Consnet: learning consistency graph for zero-shot human-object interaction detection. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4235\u20134243 (2020)","DOI":"10.1145\/3394171.3413600"},{"key":"1818_CR41","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., Qiao, Y., Peng, X., Tao, D.: Detecting human-object interaction via fabricated compositional learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14646\u201314655 (2021)","DOI":"10.1109\/CVPR46437.2021.01441"},{"key":"1818_CR42","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., Tao, D.: Discovering human-object interaction concepts via self-compositional learning. In: European Conference on Computer Vision, pp. 461\u2013478 (2022). Springer","DOI":"10.1007\/978-3-031-19812-0_27"},{"key":"1818_CR43","unstructured":"Sarullo, A., Mu, T.: Zero-shot human-object interaction recognition via affordance graphs. arXiv preprint arXiv:2009.01039 (2020)"},{"key":"1818_CR44","doi-asserted-by":"crossref","unstructured":"Li, X., Song, K., Feng, S., Wang, D., Zhang, Y.: A co-attention neural network model for emotion cause analysis with emotional context awareness. In: Riloff, E., Chiang, D., Hockenmaier, J., Tsujii, J. (eds.) Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 4752\u20134757. Association for Computational Linguistics, Brussels, Belgium (2018). https:\/\/doi.org\/10.18653\/v1\/D18-1506 . https:\/\/aclanthology.org\/D18-1506","DOI":"10.18653\/v1\/D18-1506"},{"key":"1818_CR45","doi-asserted-by":"crossref","unstructured":"Lu, X., Wang, W., Ma, C., Shen, J., Shao, L., Porikli, F.: See more, know more: Unsupervised video object segmentation with co-attention siamese networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3623\u20133632 (2019)","DOI":"10.1109\/CVPR.2019.00374"},{"key":"1818_CR46","doi-asserted-by":"crossref","unstructured":"Zhong, X., Ding, C., Qu, X., Tao, D.: Polysemy deciphering network for human-object interaction detection. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XX 16, pp. 69\u201385 (2020). Springer","DOI":"10.1007\/978-3-030-58565-5_5"},{"issue":"1\u20132","key":"1818_CR47","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1002\/nav.3800020109","volume":"2","author":"HW Kuhn","year":"1955","unstructured":"Kuhn, H.W.: The Hungarian method for the assignment problem. Naval Res. Log. Q. 2(1\u20132), 83\u201397 (1955)","journal-title":"Naval Res. Log. Q."},{"key":"1818_CR48","doi-asserted-by":"crossref","unstructured":"Rezatofighi, H., Tsoi, N., Gwak, J., Sadeghian, A., Reid, I., Savarese, S.: Generalized intersection over union: a metric and a loss for bounding box regression. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 658\u2013666 (2019)","DOI":"10.1109\/CVPR.2019.00075"},{"issue":"10","key":"1818_CR49","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3674980","volume":"20","author":"S Chan","year":"2024","unstructured":"Chan, S., Zeng, X., Wang, X., Hu, J., Bai, C.: Auxiliary feature fusion and noise suppression for hoi detection. ACM Trans. Multimed. Comput. Commun. Appl. 20(10), 1\u201318 (2024)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"1818_CR50","doi-asserted-by":"crossref","unstructured":"Wang, T., Anwer, R.M., Khan, M.H., Khan, F.S., Pang, Y., Shao, L., Laaksonen, J.: Deep contextual attention for human-object interaction detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5694\u20135702 (2019)","DOI":"10.1109\/ICCV.2019.00579"},{"key":"1818_CR51","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110897","volume":"157","author":"X Lin","year":"2025","unstructured":"Lin, X., Zou, Q., Xu, X.: Human-object interaction detection via recycling of ground-truth annotations. Pattern Recogn. 157, 110897 (2025)","journal-title":"Pattern Recogn."},{"issue":"7","key":"1818_CR52","doi-asserted-by":"publisher","first-page":"3870","DOI":"10.1109\/TPAMI.2021.3054048","volume":"44","author":"Y-L Li","year":"2022","unstructured":"Li, Y.-L., Liu, X., Wu, X., Huang, X., Xu, L., Lu, C.: Transferable interactiveness knowledge for human-object interaction detection. IEEE Trans. Pattern Anal. Mach. Intell. 44(7), 3870\u20133882 (2022). https:\/\/doi.org\/10.1109\/TPAMI.2021.3054048","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1818_CR53","first-page":"1","volume":"71","author":"D Gu","year":"2022","unstructured":"Gu, D., Ma, S., Cai, S.: DSSF: dynamic semantic sampling and fusion for one-stage human-object interaction detection. IEEE Trans. Instrum. Meas. 71, 1\u201313 (2022)","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"1818_CR54","doi-asserted-by":"crossref","unstructured":"Chen, M., Liao, Y., Liu, S., Chen, Z., Wang, F., Qian, C.: Reformulating hoi detection as adaptive set prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9004\u20139013 (2021)","DOI":"10.1109\/CVPR46437.2021.00889"},{"key":"1818_CR55","doi-asserted-by":"publisher","first-page":"964","DOI":"10.1109\/TIP.2022.3231528","volume":"32","author":"J Lim","year":"2023","unstructured":"Lim, J., Baskaran, V.M., Lim, J.M.-Y., Wong, K., See, J., Tistarelli, M.: Ernet: an efficient and reliable human-object interaction detection network. IEEE Trans. Image Process. 32, 964\u2013979 (2023)","journal-title":"IEEE Trans. Image Process."},{"key":"1818_CR56","doi-asserted-by":"publisher","first-page":"6274","DOI":"10.1109\/TIP.2023.3330304","volume":"32","author":"T He","year":"2023","unstructured":"He, T., Gao, L., Song, J., Li, Y.-F.: Toward a unified transformer-based framework for scene graph generation and human-object interaction detection. IEEE Trans. Image Process. 32, 6274\u20136288 (2023)","journal-title":"IEEE Trans. Image Process."},{"key":"1818_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, F.Z., Campbell, D., Gould, S.: Spatially conditioned graphs for detecting human-object interactions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13319\u201313327 (2021)","DOI":"10.1109\/ICCV48922.2021.01307"},{"key":"1818_CR58","doi-asserted-by":"crossref","unstructured":"Zhuang, Z., Qian, R., Xie, C., Liang, S.: Compositional learning in transformer-based human-object interaction detection. In: 2023 IEEE International Conference on Multimedia and Expo (ICME), pp. 1038\u20131043 (2023). IEEE","DOI":"10.1109\/ICME55011.2023.00182"},{"key":"1818_CR59","doi-asserted-by":"crossref","unstructured":"Yang, K., Deng, J., An, X., Li, J., Feng, Z., Guo, J., Yang, J., Liu, T.: Alip: adaptive language-image pre-training with synthetic caption. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2922\u20132931 (2023)","DOI":"10.1109\/ICCV51070.2023.00273"},{"key":"1818_CR60","doi-asserted-by":"crossref","unstructured":"Mu, N., Kirillov, A., Wagner, D., Xie, S.: Slip: self-supervision meets language-image pre-training. In: European Conference on Computer Vision, pp. 529\u2013544 (2022). Springer","DOI":"10.1007\/978-3-031-19809-0_30"},{"key":"1818_CR61","unstructured":"Chuang, Y.-S., Li, Y., Wang, D., Yeh, C.-F., Lyu, K., Raghavendra, R., Glass, J., Huang, L., Weston, J., Zettlemoyer, L., Chen, X., Liu, Z., Xie, S., Yih, W.-T., Li, S.-W., Xu, H.: Metaclip 2: a worldwide scaling recipe. In: Advances in Neural Information Processing Systems (2025). to appear. https:\/\/arxiv.org\/abs\/2507.22062"},{"key":"1818_CR62","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"}],"container-title":["Machine Vision and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-026-01818-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00138-026-01818-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00138-026-01818-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T14:30:39Z","timestamp":1778855439000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00138-026-01818-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,15]]},"references-count":62,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,5]]}},"alternative-id":["1818"],"URL":"https:\/\/doi.org\/10.1007\/s00138-026-01818-9","relation":{},"ISSN":["0932-8092","1432-1769"],"issn-type":[{"value":"0932-8092","type":"print"},{"value":"1432-1769","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4,15]]},"assertion":[{"value":"12 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 January 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 April 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Informed consent was obtained from the Indian Institute of Technology Guwahati and from all authors for the publication of this article.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval"}}],"article-number":"59"}}