{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:41:24Z","timestamp":1755823284738,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Science Fund for Distinguished Young Scholars","award":["62025603"],"award-info":[{"award-number":["62025603"]}]},{"name":"the Natural Science Foundation of Fujian Province of China","award":["2021J01002,2022J06001"],"award-info":[{"award-number":["2021J01002,2022J06001"]}]},{"name":"National Key R&D Program of China","award":["2022ZD0118202"],"award-info":[{"award-number":["2022ZD0118202"]}]},{"name":"the National Natural Science Foundation of China","award":["U21B2037,U22B2051,62176222,62176223,62176226,62072386,62072387,62072389,62002305,62272401"],"award-info":[{"award-number":["U21B2037,U22B2051,62176222,62176223,62176226,62072386,62072387,62072389,62002305,62272401"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3611735","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"5455-5463","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["Improving Human-Object Interaction Detection via Virtual Image Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4414-7744","authenticated-orcid":false,"given":"Shuman","family":"Fang","sequence":"first","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3998-7556","authenticated-orcid":false,"given":"Shuai","family":"Liu","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3102-6425","authenticated-orcid":false,"given":"Jie","family":"Li","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4355-5711","authenticated-orcid":false,"given":"Guannan","family":"Jiang","sequence":"additional","affiliation":[{"name":"Contemporary Amperex Technology Co. Limited (CATL), Ningde, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4739-8936","authenticated-orcid":false,"given":"Xianming","family":"Lin","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9163-2932","authenticated-orcid":false,"given":"Rongrong","family":"Ji","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume":"202","author":"Azizi Shekoofeh","unstructured":"Shekoofeh Azizi, Simon Kornblith, Chitwan Saharia, Mohammad Norouzi, and David J Fleet. 2023. Synthetic data from diffusion models improves imagenet classification. arXiv preprint arXiv:2304.08466 (2023).","journal-title":"David J Fleet."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2967301"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2018.00048"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58610-2_41"},{"key":"e_1_3_2_1_6_1","volume-title":"British Machine Vision Conference.","author":"Gao Chen","year":"2018","unstructured":"Chen Gao, Yuliang Zou, and Jia-Bin Huang. 2018. iCAN: Instance-centric attention network for human-object interaction detection. In British Machine Vision Conference."},{"key":"e_1_3_2_1_7_1","volume-title":"Laurent Itti, and Vibhav Vineet.","author":"Ge Yunhao","year":"2022","unstructured":"Yunhao Ge, Jiashu Xu, Brian Nlong Zhao, Laurent Itti, and Vibhav Vineet. 2022. Dall-e for detection: Language-driven context image synthesis for object detection. arXiv preprint arXiv:2206.09592 (2022)."},{"key":"e_1_3_2_1_8_1","volume-title":"Visual semantic role labeling. arXiv preprint arXiv:1505.04474","author":"Gupta Saurabh","year":"2015","unstructured":"Saurabh Gupta and Jitendra Malik. 2015. Visual semantic role labeling. arXiv preprint arXiv:1505.04474 (2015)."},{"key":"e_1_3_2_1_9_1","volume-title":"International Conference on Computer Vision.","author":"He Ruifei","year":"2023","unstructured":"Ruifei He, Shuyang Sun, Xin Yu, Chuhui Xue, Wenqing Zhang, Philip Torr, Song Bai, and Xiaojuan Qi. 2023. Is synthetic data from generative models ready for image recognition?. In International Conference on Computer Vision."},{"key":"e_1_3_2_1_10_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58555-6_35"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Zhi Hou Baosheng Yu Yu Qiao Xiaojiang Peng and Dacheng Tao. 2021. Affordance transfer learning for human-object interaction detection. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR46437.2021.00056"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Zhi Hou Baosheng Yu Yu Qiao Xiaojiang Peng and Dacheng Tao. 2021. Detecting human-object interaction via fabricated compositional learning. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR46437.2021.01441"},{"key":"e_1_3_2_1_14_1","volume-title":"Istr: End-to-end instance segmentation with transformers. arXiv preprint arXiv:2105.00637","author":"Hu Jie","year":"2021","unstructured":"Jie Hu, Liujuan Cao, Yao Lu, ShengChuan Zhang, YanWang, Ke Li, Feiyue Huang, Ling Shao, and Rongrong Ji. 2021. Istr: End-to-end instance segmentation with transformers. arXiv preprint arXiv:2105.00637 (2021)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Jie Hu Linyan Huang Tianhe Ren Shengchuan Zhang Rongrong Ji and Liujuan Cao. 2023. You Only Segment Once: Towards Real-Time Panoptic Segmentation. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52729.2023.01709"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"ASM Iftekhar Hao Chen Kaustav Kundu Xinyu Li Joseph Tighe and Davide Modolo. 2022. What to look at and where: Semantic and spatial refined transformer for detecting human-object interactions. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52688.2022.00528"},{"volume-title":"SGAP-Net: Semanticguided attentive prototypes network for few-shot human-object interaction recognition","author":"Ji Zhong","key":"e_1_3_2_1_17_1","unstructured":"Zhong Ji, Xiyao Liu, Yanwei Pang, and Xuelong Li. 2020. SGAP-Net: Semanticguided attentive prototypes network for few-shot human-object interaction recognition. In Association for the Advancement of Artificial Intelligence."},{"key":"e_1_3_2_1_18_1","volume-title":"The Hungarian method for the assignment problem. Naval research logistics quarterly","author":"Kuhn Harold W","year":"1955","unstructured":"Harold W Kuhn. 1955. The Hungarian method for the assignment problem. Naval research logistics quarterly (1955)."},{"key":"e_1_3_2_1_19_1","volume-title":"Ppdm: Parallel point detection and matching for real-time human-object interaction detection. In Computer Vision and Pattern Recognition.","author":"Liao Yue","year":"2020","unstructured":"Yue Liao, Si Liu, Fei Wang, Yanjie Chen, Chen Qian, and Jiashi Feng. 2020. Ppdm: Parallel point detection and matching for real-time human-object interaction detection. In Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_20_1","volume-title":"Gen-vlkt: Simplify association and enhance interaction understanding for hoi detection. In Computer Vision and Pattern Recognition.","author":"Liao Yue","year":"2022","unstructured":"Yue Liao, Aixi Zhang, Miao Lu, Yongliang Wang, Xiaobo Li, and Si Liu. 2022. Gen-vlkt: Simplify association and enhance interaction understanding for hoi detection. In Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_1_22_1","volume-title":"Visual question answering with dense inter-and intra-modality interactions","author":"Liu Fei","year":"2020","unstructured":"Fei Liu, Jing Liu, Zhiwei Fang, Richang Hong, and Hanqing Lu. 2020. Visual question answering with dense inter-and intra-modality interactions. IEEE Transactions on Multimedia (2020)."},{"key":"e_1_3_2_1_23_1","unstructured":"Xinpeng Liu Yong-Lu Li Xiaoqian Wu Yu-Wing Tai Cewu Lu and Chi-Keung Tang. 2022. Interactiveness Field in Human-Object Interactions. In Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_24_1","volume":"202","author":"Park Jihwan","unstructured":"Jihwan Park, SeungJun Lee, Hwan Heo, Hyeong Kyu Choi, and Hyunwoo J Kim. 2022. Consistency learning via decoding path augmentation for transformers in human object interaction detection. In Computer Vision and Pattern Recognition.","journal-title":"Hyunwoo J Kim."},{"key":"e_1_3_2_1_25_1","unstructured":"Xian Qu Changxing Ding Xingao Li Xubin Zhong and Dacheng Tao. 2022. Distillation using oracle queries for transformer-based human-object interaction detection. In Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_26_1","volume-title":"International Conference on Machine Learning.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Hamid Rezatofighi Nathan Tsoi JunYoung Gwak Amir Sadeghian Ian Reid and Silvio Savarese. 2019. Generalized intersection over union: A metric and a loss for bounding box regression. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2019.00075"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2022. High-resolution image synthesis with latent diffusion models. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Mert Bulent Sariyildiz Karteek Alahari Diane Larlus and Yannis Kalantidis. 2023. Fake it till you make it: Learning transferable representations from synthetic ImageNet clones. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52729.2023.00774"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00852"},{"key":"e_1_3_2_1_31_1","volume-title":"QPIC: Query- Based Pairwise Human-Object Interaction Detection with Image-Wide Contextual Information. In Computer Vision and Pattern Recognition.","author":"Tamura Masato","year":"2021","unstructured":"Masato Tamura, Hiroki Ohashi, and Tomoaki Yoshinaga. 2021. QPIC: Query- Based Pairwise Human-Object Interaction Detection with Image-Wide Contextual Information. In Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_32_1","unstructured":"Antti Tarvainen and Harri Valpola. 2017. Mean teachers are better role models: Weight-averaged consistency targets improve semi-supervised deep learning results. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_33_1","volume-title":"Effective data augmentation with diffusion models. arXiv preprint arXiv:2302.07944","author":"Trabucco Brandon","year":"2023","unstructured":"Brandon Trabucco, Kyle Doherty, Max Gurinas, and Ruslan Salakhutdinov. 2023. Effective data augmentation with diffusion models. arXiv preprint arXiv:2302.07944 (2023)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2454332"},{"key":"e_1_3_2_1_35_1","volume-title":"Vsgnet: Spatial attention network for detecting human object interactions using graph convolutions. In Computer Vision and Pattern Recognition.","author":"Ulutan Oytun","year":"2020","unstructured":"Oytun Ulutan, ASM Iftekhar, and Bangalore S Manjunath. 2020. Vsgnet: Spatial attention network for detecting human object interactions using graph convolutions. In Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20053-3_38"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547793"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"PeiWang Zhaowei Cai Hao Yang Gurumurthy Swaminathan Nuno Vasconcelos Bernt Schiele and Stefano Soatto. 2022. Omni-DETR: Omni-supervised object detection with transformers. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52688.2022.00915"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00579"},{"key":"e_1_3_2_1_40_1","unstructured":"Bingjie Xu Yongkang Wong Junnan Li Qi Zhao and Mohan S Kankanhalli. 2019. Learning to detect human-object interactions with knowledge. In Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547862"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.144"},{"volume-title":"Detecting humanobject interactions with object-guided cross-modal calibrated semantics","author":"Yuan Hangjie","key":"e_1_3_2_1_43_1","unstructured":"Hangjie Yuan, MangWang, Dong Ni, and Liangpeng Xu. 2022. Detecting humanobject interactions with object-guided cross-modal calibrated semantics. In Association for the Advancement of Artificial Intelligence."},{"key":"e_1_3_2_1_44_1","unstructured":"Aixi Zhang Yue Liao Si Liu Miao Lu Yongliang Wang Chen Gao and Xiaobo Li. 2021. Mining the benefits of two-stage and one-stage hoi detection. Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01307"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","unstructured":"Frederic Z Zhang Dylan Campbell and Stephen Gould. 2022. Efficient two-stage detection of human-object interactions with a novel unary-pairwise transformer. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52688.2022.01947"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19812-0_26"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"Xubin Zhong Xian Qu Changxing Ding and Dacheng Tao. 2021. Glance and Gaze: Inferring Action-aware Points for One-Stage Human-Object Interaction Detection. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR46437.2021.01303"},{"key":"e_1_3_2_1_49_1","volume-title":"Places: A 10 million image database for scene recognition","author":"Zhou Bolei","year":"2017","unstructured":"Bolei Zhou, Agata Lapedriza, Aditya Khosla, Aude Oliva, and Antonio Torralba. 2017. Places: A 10 million image database for scene recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence (2017)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"crossref","unstructured":"Desen Zhou Zhichao Liu Jian Wang Leshan Wang Tao Hu Errui Ding and Jingdong Wang. 2022. Human-object interaction detection via disentangled transformer. In Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR52688.2022.01896"},{"key":"e_1_3_2_1_51_1","unstructured":"Shangchen Zhou Kelvin Chan Chongyi Li and Chen Change Loy. 2022. Towards robust blind face restoration with codebook lookup transformer. In Advances in Neural Information Processing Systems."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611735","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3611735","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:10:20Z","timestamp":1755821420000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611735"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":51,"alternative-id":["10.1145\/3581783.3611735","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3611735","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}