{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T10:19:40Z","timestamp":1779358780924,"version":"3.51.4"},"publisher-location":"Cham","reference-count":37,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031734106","type":"print"},{"value":"9783031734113","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T00:00:00Z","timestamp":1732320000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T00:00:00Z","timestamp":1732320000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73411-3_24","type":"book-chapter","created":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T20:09:04Z","timestamp":1732306144000},"page":"419-435","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Interaction-Centric Spatio-Temporal Context Reasoning for\u00a0Multi-person Video HOI Recognition"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7299-1247","authenticated-orcid":false,"given":"Yisong","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7334-7772","authenticated-orcid":false,"given":"Nan","family":"Xi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7104-608X","authenticated-orcid":false,"given":"Jingjing","family":"Meng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7901-8793","authenticated-orcid":false,"given":"Junsong","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,23]]},"reference":[{"key":"24_CR1","doi-asserted-by":"publisher","unstructured":"Chopra, S., Hadsell, R., LeCun, Y.: Learning a similarity metric discriminatively, with application to face verification. In: 2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2005), vol.\u00a01, pp. 539\u2013546 (2005). https:\/\/doi.org\/10.1109\/CVPR.2005.202","DOI":"10.1109\/CVPR.2005.202"},{"key":"24_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2021.107584","volume":"235","author":"S Deng","year":"2022","unstructured":"Deng, S., et al.: Low-resource extraction with knowledge-aware pairwise prototype learning. Knowl.-Based Syst. 235, 107584 (2022)","journal-title":"Knowl.-Based Syst."},{"key":"24_CR3","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"24_CR4","doi-asserted-by":"publisher","unstructured":"Dreher, C.R.G., W\u00e4chter, M., Asfour, T.: Learning object-action relations from bimanual human demonstration using graph networks. IEEE Robot. Autom. Lett. (RA-L) 5(1), 187\u2013194 (2020). https:\/\/doi.org\/10.1109\/LRA.2019.2949221","DOI":"10.1109\/LRA.2019.2949221"},{"key":"24_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"696","DOI":"10.1007\/978-3-030-58610-2_41","volume-title":"Computer Vision \u2013 ECCV 2020","author":"C Gao","year":"2020","unstructured":"Gao, C., Xu, J., Zou, Y., Huang, J.-B.: DRG: dual relation graph for human-object interaction detection. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12357, pp. 696\u2013712. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58610-2_41"},{"key":"24_CR6","unstructured":"Gao, K., Chen, L., Zhang, H., Xiao, J., Sun, Q.: Compositional prompt tuning with motion cues for open-vocabulary video relation detection. In: The Eleventh International Conference on Learning Representations (2022)"},{"key":"24_CR7","doi-asserted-by":"crossref","unstructured":"Gupta, T., Schwing, A., Hoiem, D.: No-frills human-object interaction detection: Factorization, layout encodings, and training techniques. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9677\u20139685 (2019)","DOI":"10.1109\/ICCV.2019.00977"},{"key":"24_CR8","doi-asserted-by":"crossref","unstructured":"Iftekhar, A., Chen, H., Kundu, K., Li, X., Tighe, J., Modolo, D.: What to look at and where: semantic and spatial refined transformer for detecting human-object interactions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5353\u20135363 (2022)","DOI":"10.1109\/CVPR52688.2022.00528"},{"key":"24_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"498","DOI":"10.1007\/978-3-030-58555-6_30","volume-title":"Computer Vision \u2013 ECCV 2020","author":"B Kim","year":"2020","unstructured":"Kim, B., Choi, T., Kang, J., Kim, H.J.: UnionDet: union-level detector towards real-time human-object interaction detection. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12360, pp. 498\u2013514. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58555-6_30"},{"key":"24_CR10","doi-asserted-by":"crossref","unstructured":"Kim, B., Lee, J., Kang, J., Kim, E.S., Kim, H.J.: HOTR: end-to-end human-object interaction detection with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 74\u201383, June 2021","DOI":"10.1109\/CVPR46437.2021.00014"},{"issue":"1","key":"24_CR11","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1109\/TPAMI.2015.2430335","volume":"38","author":"HS Koppula","year":"2015","unstructured":"Koppula, H.S., Saxena, A.: Anticipating human activities using object affordances for reactive robotic response. IEEE Trans. Pattern Anal. Mach. Intell. 38(1), 14\u201329 (2015)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"8","key":"24_CR12","doi-asserted-by":"publisher","first-page":"951","DOI":"10.1177\/0278364913478446","volume":"32","author":"HS Koppula","year":"2013","unstructured":"Koppula, H.S., Gupta, R., Saxena, A.: Learning human activities and object affordances from RGB-D videos. Int. J. Robot. Res. 32(8), 951\u2013970 (2013)","journal-title":"Int. J. Robot. Res."},{"key":"24_CR13","doi-asserted-by":"publisher","unstructured":"Koppula, H.S., Gupta, R., Saxena, A.: Learning human activities and object affordances from RGB-D videos. Int. J. Rob. Res. 32(8), 951\u2013970 (2013). https:\/\/doi.org\/10.1177\/0278364913478446","DOI":"10.1177\/0278364913478446"},{"key":"24_CR14","doi-asserted-by":"crossref","unstructured":"Lea, C., Flynn, M.D., Vidal, R., Reiter, A., Hager, G.D.: Temporal convolutional networks for action segmentation and detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 156\u2013165 (2017)","DOI":"10.1109\/CVPR.2017.113"},{"key":"24_CR15","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"24_CR16","doi-asserted-by":"crossref","unstructured":"Li, Y.L., et al.: Transferable interactiveness knowledge for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3585\u20133594 (2019)","DOI":"10.1109\/CVPR.2019.00370"},{"key":"24_CR17","doi-asserted-by":"crossref","unstructured":"Liao, Y., Liu, S., Wang, F., Chen, Y., Qian, C., Feng, J.: PPDM: parallel point detection and matching for real-time human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 482\u2013490 (2020)","DOI":"10.1109\/CVPR42600.2020.00056"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988 (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"24_CR19","doi-asserted-by":"crossref","unstructured":"Liu, X., Li, Y.L., Wu, X., Tai, Y.W., Lu, C., Tang, C.K.: Interactiveness field in human-object interactions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20113\u201320122 (2022)","DOI":"10.1109\/CVPR52688.2022.01948"},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Luan, T., Wang, Y., Zhang, J., Wang, Z., Zhou, Z., Qiao, Y.: PC-HMR: pose calibration for 3D human mesh recovery from 2D images\/videos. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 2269\u20132276 (2021)","DOI":"10.1609\/aaai.v35i3.16326"},{"key":"24_CR21","doi-asserted-by":"crossref","unstructured":"Luan, T., et al.: High fidelity 3D hand shape reconstruction via scalable graph frequency decomposition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16795\u201316804 (2023)","DOI":"10.1109\/CVPR52729.2023.01611"},{"key":"24_CR22","unstructured":"Mettes, P., Van\u00a0der Pol, E., Snoek, C.: Hyperspherical prototype networks. Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"24_CR23","doi-asserted-by":"crossref","unstructured":"Morais, R., Le, V., Venkatesh, S., Tran, T.: Learning asynchronous and sparse human-object interaction in videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16041\u201316050 (2021)","DOI":"10.1109\/CVPR46437.2021.01578"},{"key":"24_CR24","doi-asserted-by":"crossref","unstructured":"Nagarajan, T., Feichtenhofer, C., Grauman, K.: Grounded human-object interaction hotspots from video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8688\u20138697 (2019)","DOI":"10.1109\/ICCV.2019.00878"},{"key":"24_CR25","doi-asserted-by":"crossref","unstructured":"Park, J., Park, J.W., Lee, J.S.: ViPLO: vision transformer based pose-conditioned self-loop graph for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 17152\u201317162, June 2023","DOI":"10.1109\/CVPR52729.2023.01645"},{"key":"24_CR26","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"407","DOI":"10.1007\/978-3-030-01240-3_25","volume-title":"Computer Vision \u2013 ECCV 2018","author":"S Qi","year":"2018","unstructured":"Qi, S., Wang, W., Jia, B., Shen, J., Zhu, S.-C.: Learning human-object interactions by graph parsing neural networks. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11213, pp. 407\u2013423. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01240-3_25"},{"key":"24_CR27","series-title":"LNCS","first-page":"474","volume-title":"ECCV 2022","author":"T Qiao","year":"2022","unstructured":"Qiao, T., Men, Q., Li, F.W., Kubotani, Y., Morishima, S., Shum, H.P.: Geometric features informed multi-person human-object interaction recognition in videos. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13664, pp. 474\u2013491. Springer, Cham (2022)"},{"key":"24_CR28","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"24_CR29","doi-asserted-by":"crossref","unstructured":"Sener, O., Saxena, A.: rCRF: recursive belief estimation over CRFs in RGB-D activity videos. In: Robotics: Science and systems (2015)","DOI":"10.15607\/RSS.2015.XI.024"},{"key":"24_CR30","doi-asserted-by":"crossref","unstructured":"Sunkesula, S.P.R., Dabral, R., Ramakrishnan, G.: Lighten: learning interactions with graph and hierarchical temporal networks for hoi in videos. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 691\u2013699 (2020)","DOI":"10.1145\/3394171.3413778"},{"key":"24_CR31","doi-asserted-by":"crossref","unstructured":"Tamura, M., Ohashi, H., Yoshinaga, T.: QPIC: query-based pairwise human-object interaction detection with image-wide contextual information. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10410\u201310419, June 2021","DOI":"10.1109\/CVPR46437.2021.01027"},{"key":"24_CR32","first-page":"23345","volume":"35","author":"D Tu","year":"2022","unstructured":"Tu, D., Sun, W., Min, X., Zhai, G., Shen, W.: Video-based human-object interaction detection from tubelet tokens. Adv. Neural. Inf. Process. Syst. 35, 23345\u201323357 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"24_CR33","doi-asserted-by":"crossref","unstructured":"Wang, T., et al.: Deep contextual attention for human-object interaction detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5694\u20135702 (2019)","DOI":"10.1109\/ICCV.2019.00579"},{"key":"24_CR34","doi-asserted-by":"crossref","unstructured":"Xi, N., Meng, J., Yuan, J.: Chain-of-look prompting for verb-centric surgical triplet recognition in endoscopic videos. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5007\u20135016 (2023)","DOI":"10.1145\/3581783.3611898"},{"key":"24_CR35","doi-asserted-by":"crossref","unstructured":"Xi, N., Meng, J., Yuan, J.: Open set video HOI detection from action-centric chain-of-look prompting. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3079\u20133089 (2023)","DOI":"10.1109\/ICCV51070.2023.00286"},{"key":"24_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Pan, Y., Yao, T., Huang, R., Mei, T., Chen, C.W.: Exploring structure-aware transformer over interaction proposals for human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19548\u201319557 (2022)","DOI":"10.1109\/CVPR52688.2022.01894"},{"key":"24_CR37","doi-asserted-by":"crossref","unstructured":"Zhong, X., Qu, X., Ding, C., Tao, D.: Glance and gaze: Inferring action-aware points for one-stage human-object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13234\u201313243 (2021)","DOI":"10.1109\/CVPR46437.2021.01303"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73411-3_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T21:27:55Z","timestamp":1732310875000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73411-3_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,23]]},"ISBN":["9783031734106","9783031734113"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73411-3_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,23]]},"assertion":[{"value":"23 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}