{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T18:21:58Z","timestamp":1776277318322,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372387, 61802053"],"award-info":[{"award-number":["62372387, 61802053"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2020M683353"],"award-info":[{"award-number":["2020M683353"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["2682022JX007, 2682022KJ044, 2682023ZTPY004"],"award-info":[{"award-number":["2682022JX007, 2682022KJ044, 2682023ZTPY004"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Grant of Institute of Applied Physics and Computational Mathematics, Beijing","award":["HXO2020118"],"award-info":[{"award-number":["HXO2020118"]}]},{"name":"Key R&D Program of Guangxi Zhuang Autonomous Region, China","award":["AB22080038, AB22080039"],"award-info":[{"award-number":["AB22080038, AB22080039"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3611813","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:12Z","timestamp":1698391632000},"page":"2233-2242","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["Human-Object-Object Interaction: Towards Human-Centric Complex Interaction Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-7302-5335","authenticated-orcid":false,"given":"Mingxuan","family":"Zhang","sequence":"first","affiliation":[{"name":"Southwest Jiaotong University &amp; Engineering Research Center of Sustainable Urban Intelligent Transportation, Ministry of Education, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8322-8558","authenticated-orcid":false,"given":"Xiao","family":"Wu","sequence":"additional","affiliation":[{"name":"Southwest Jiaotong University &amp; Engineering Research Center of Sustainable Urban Intelligent Transportation, Ministry of Education, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4083-5155","authenticated-orcid":false,"given":"Zhaoquan","family":"Yuan","sequence":"additional","affiliation":[{"name":"Southwest Jiaotong University &amp; Engineering Research Center of Sustainable Urban Intelligent Transportation, Ministry of Education, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6109-9417","authenticated-orcid":false,"given":"Qi","family":"He","sequence":"additional","affiliation":[{"name":"Southwest Jiaotong University &amp; Engineering Research Center of Sustainable Urban Intelligent Transportation, Ministry of Education, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5569-4255","authenticated-orcid":false,"given":"Xiang","family":"Huang","sequence":"additional","affiliation":[{"name":"Southwest Jiaotong University &amp; Engineering Research Center of Sustainable Urban Intelligent Transportation, Ministry of Education, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11554-018-0840-6"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.38"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00644"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/65"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01930"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01081"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547943"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964326"},{"key":"e_1_3_2_1_10_1","volume-title":"Video ecommerce: Toward large scale online video advertising","author":"Cheng Zhi-Qi","year":"2017","unstructured":"Zhi-Qi Cheng, Xiao Wu, Yang Liu, and Xian-Sheng Hua. 2017a. Video ecommerce: Toward large scale online video advertising. IEEE transactions on multimedia, Vol. 19, 6 (2017), 1170--1183."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.444"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3463944.3469097"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMM48946.2020.9141973"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_1_16_1","volume-title":"iCAN: Instance-Centric Attention Network for Human-Object Interaction Detection. CoRR","author":"Gao Chen","year":"2018","unstructured":"Chen Gao, Yuliang Zou, and Jia-Bin Huang. 2018. iCAN: Instance-Centric Attention Network for Human-Object Interaction Detection. CoRR, Vol. abs\/1808.10437 (2018)."},{"key":"e_1_3_2_1_17_1","volume-title":"Yolox: Exceeding yolo series in","author":"Ge Zheng","year":"2021","unstructured":"Zheng Ge, Songtao Liu, Feng Wang, Zeming Li, and Jian Sun. 2021. Yolox: Exceeding yolo series in 2021. arXiv preprint arXiv:2107.08430 (2021)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00633"},{"key":"e_1_3_2_1_19_1","volume-title":"Visual semantic role labeling. arXiv preprint arXiv:1505.04474","author":"Gupta Saurabh","year":"2015","unstructured":"Saurabh Gupta and Jitendra Malik. 2015. Visual semantic role labeling. arXiv preprint arXiv:1505.04474 (2015)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2020.05.118"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00800"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.472"},{"key":"e_1_3_2_1_24_1","volume-title":"Fahad Shahbaz Khan, and Mubarak Shah.","author":"Khan Salman","year":"2022","unstructured":"Salman Khan, Muzammal Naseer, Munawar Hayat, Syed Waqas Zamir, Fahad Shahbaz Khan, and Mubarak Shah. 2022. Transformers in vision: A survey. ACM computing surveys, Vol. 54, 10s (2022), 1--41."},{"key":"e_1_3_2_1_25_1","volume-title":"HOTR: End-to-End Human-Object Interaction Detection With Transformers. In IEEE Conference on Computer Vision and Pattern Recognition. 74--83","author":"Kim Bumsoo","year":"2021","unstructured":"Bumsoo Kim, Junhyun Lee, Jaewoo Kang, Eun-Sol Kim, and Hyunwoo J Kim. 2021. HOTR: End-to-End Human-Object Interaction Detection With Transformers. In IEEE Conference on Computer Vision and Pattern Recognition. 74--83."},{"key":"e_1_3_2_1_26_1","volume-title":"You only watch once: A unified cnn architecture for real-time spatiotemporal action localization. arXiv preprint arXiv:1911.06644","author":"K\u00f6p\u00fckl\u00fc Okan","year":"2019","unstructured":"Okan K\u00f6p\u00fckl\u00fc, Xiangyu Wei, and Gerhard Rigoll. 2019. You only watch once: A unified cnn architecture for real-time spatiotemporal action localization. arXiv preprint arXiv:1911.06644 (2019)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_19"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01018"},{"key":"e_1_3_2_1_29_1","volume-title":"PaStaNet: Toward Human Activity Knowledge Engine. In IEEE Conference on Computer Vision and Pattern Recognition. 379--388","author":"Li Yong-Lu","year":"2020","unstructured":"Yong-Lu Li, Liang Xu, Xinpeng Liu, Xijie Huang, Yue Xu, Shiyi Wang, Hao-Shu Fang, Ze Ma, Mingyang Chen, and Cewu Lu. 2020b. PaStaNet: Toward Human Activity Knowledge Engine. In IEEE Conference on Computer Vision and Pattern Recognition. 379--388."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00370"},{"key":"e_1_3_2_1_31_1","volume-title":"PPDM: Parallel Point Detection and Matching for Real-Time Human-Object Interaction Detection. In IEEE Conference on Computer Vision and Pattern Recognition. 479--487","author":"Liao Yue","year":"2020","unstructured":"Yue Liao, Si Liu, Fei Wang, Yanjie Chen, Chen Qian, and Jiashi Feng. 2020. PPDM: Parallel Point Detection and Matching for Real-Time Human-Object Interaction Detection. In IEEE Conference on Computer Vision and Pattern Recognition. 479--487."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413600"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00817"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01578"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00053"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547786"},{"key":"e_1_3_2_1_39_1","volume-title":"SWNet: a deep learning based approach for splashed water detection on road","author":"Qiao Jian-Jun","year":"2020","unstructured":"Jian-Jun Qiao, Xiao Wu, Jun-Yan He, Wei Li, and Qiang Peng. 2020. SWNet: a deep learning based approach for splashed water detection on road. IEEE transactions on intelligent transportation systems, Vol. 23, 4 (2020), 3012--3025."},{"key":"e_1_3_2_1_40_1","volume-title":"From show to tell: a survey on deep learning-based image captioning","author":"Stefanini Matteo","year":"2022","unstructured":"Matteo Stefanini, Marcella Cornia, Lorenzo Baraldi, Silvia Cascianelli, Giuseppe Fiameni, and Rita Cucchiara. 2022. From show to tell: a survey on deep learning-based image captioning. IEEE transactions on pattern analysis and machine intelligence, Vol. 45, 1 (2022), 539--559."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58555-6_5"},{"key":"e_1_3_2_1_42_1","first-page":"23345","article-title":"Video-based human-object interaction detection from tubelet tokens","volume":"35","author":"Tu Danyang","year":"2022","unstructured":"Danyang Tu, Wei Sun, Xiongkuo Min, Guangtao Zhai, and Wei Shen. 2022. Video-based human-object interaction detection from tubelet tokens. Advances in Neural Information Processing Systems, Vol. 35 (2022), 23345--23357.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01825"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01363"},{"key":"e_1_3_2_1_45_1","volume-title":"Conference on Neural Information Processing Systems","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Conference on Neural Information Processing Systems, Vol. 30."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475636"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","unstructured":"Qilong Wang Banggu Wu Pengfei Zhu Peihua Li Wangmeng Zuo and Qinghua Hu. 2020. ECA-Net: Efficient channel attention for deep convolutional neural networks. (2020) 11534--11542.","DOI":"10.1109\/CVPR42600.2020.01155"},{"key":"e_1_3_2_1_48_1","volume-title":"Non-local Neural Networks. In IEEE Conference on Computer Vision and Pattern Recognition. 7794--7803","author":"Wang Xiaolong","year":"2018","unstructured":"Xiaolong Wang, Ross Girshick, Abhinav Gupta, and Kaiming He. 2018. Non-local Neural Networks. In IEEE Conference on Computer Vision and Pattern Recognition. 7794--7803."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01228-1_25"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00037"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00035"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvcir.2018.12.019"},{"key":"e_1_3_2_1_53_1","volume-title":"Mining the Benefits of Two-stage and One-stage HOI Detection. Conference on Neural Information Processing Systems","volume":"34","author":"Zhang Aixi","year":"2021","unstructured":"Aixi Zhang, Yue Liao, Si Liu, Miao Lu, Yongliang Wang, Chen Gao, and Xiaobo Li. 2021. Mining the Benefits of Two-stage and One-stage HOI Detection. Conference on Neural Information Processing Systems , Vol. 34 (2021)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01894"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01323"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00320"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01896"},{"key":"e_1_3_2_1_58_1","volume-title":"Overcoming Topology Agnosticism: Enhancing Skeleton-Based Action Recognition through Redefined Skeletal Topology Awareness. arXiv preprint arXiv:2305.11468","author":"Zhou Yuxuan","year":"2023","unstructured":"Yuxuan Zhou, Zhi-Qi Cheng, Jun-Yan He, Bin Luo, Yifeng Geng, Xuansong Xie, and Margret Keuper. 2023. Overcoming Topology Agnosticism: Enhancing Skeleton-Based Action Recognition through Redefined Skeletal Topology Awareness. arXiv preprint arXiv:2305.11468 (2023)."},{"key":"e_1_3_2_1_59_1","volume-title":"Hypergraph transformer for skeleton-based action recognition. arXiv preprint arXiv:2211.09590","author":"Zhou Yuxuan","year":"2022","unstructured":"Yuxuan Zhou, Chao Li, Zhi-Qi Cheng, Yifeng Geng, Xuansong Xie, and Margret Keuper. 2022a. Hypergraph transformer for skeleton-based action recognition. arXiv preprint arXiv:2211.09590 (2022)."},{"key":"e_1_3_2_1_60_1","volume-title":"Unifying Nonlocal Blocks for Neural Networks. In IEEE International Conference on Computer Vision. 12292--12301","author":"Zhu Lei","year":"2021","unstructured":"Lei Zhu, Qi She, Duo Li, Yanye Lu, Xuejing Kang, Jie Hu, and Changhu Wang. 2021. Unifying Nonlocal Blocks for Neural Networks. In IEEE International Conference on Computer Vision. 12292--12301."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01165"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611813","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3611813","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:55:56Z","timestamp":1755820556000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611813"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":61,"alternative-id":["10.1145\/3581783.3611813","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3611813","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}