{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:06:35Z","timestamp":1784268395914,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763954","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:27:29Z","timestamp":1765211249000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Uni-Inter: Unifying 3D Human Motion Synthesis Across Diverse Interaction Contexts"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6612-5023","authenticated-orcid":false,"given":"Sheng","family":"Liu","sequence":"first","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3429-9798","authenticated-orcid":false,"given":"Yuanzhi","family":"Liang","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, China Telecom (TeleAI), Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6049-4458","authenticated-orcid":false,"given":"Jiepeng","family":"Wang","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, China Telecom (TeleAI), Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6432-3704","authenticated-orcid":false,"given":"Sidan","family":"Du","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3514-2490","authenticated-orcid":false,"given":"Chi","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, China Telecom (TeleAI), Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0019-4197","authenticated-orcid":false,"given":"Xuelong","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, China Telecom (TeleAI), Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02032"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_34"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_23"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Francesca Capozzi and Jelena Ristic. 2018. How attention gates social interactions. Annals of the New York Academy of Sciences 1426 1 (2018) 179\u2013198.","DOI":"10.1111\/nyas.13854"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01636"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00702"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01880"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.14739"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00222"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01118"},{"key":"e_1_3_3_2_12_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems 33 (2020) 6840\u20136851."},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01607"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01915"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00859"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00171"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.105"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00096"},{"key":"e_1_3_3_2_19_1","first-page":"54","volume-title":"European Conference on Computer Vision","author":"Li Jiaman","year":"2024","unstructured":"Jiaman Li, Alexander Clegg, Roozbeh Mottaghi, Jiajun Wu, Xavier Puig, and C\u00a0Karen Liu. 2024. Controllable human-object interaction synthesis. In European Conference on Computer Vision. Springer, 54\u201372."},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"crossref","unstructured":"Jiaman Li Jiajun Wu and C\u00a0Karen Liu. 2023a. Object motion guided human motion synthesis. ACM Transactions on Graphics (TOG) 42 6 (2023) 1\u201311.","DOI":"10.1145\/3618333"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01265"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00877"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"crossref","unstructured":"Han Liang Wenqian Zhang Wenxuan Li Jingyi Yu and Lan Xu. 2024. Intergen: Diffusion-based multi-human motion generation under complex interactions. International Journal of Computer Vision 132 9 (2024) 3463\u20133483.","DOI":"10.1007\/s11263-024-02042-6"},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"crossref","unstructured":"Sheng Liu Jianghai Shuai Yang Li and Sidan Du. 2023. MMDA: Multi-person marginal distribution awareness for monocular 3D pose estimation. IET Image Processing 17 7 (2023) 2182\u20132191.","DOI":"10.1049\/ipr2.12783"},{"key":"e_1_3_3_2_26_1","first-page":"1","volume-title":"European Conference on Computer Vision","author":"Liu Xinpeng","year":"2024","unstructured":"Xinpeng Liu, Haowen Hou, Yanchao Yang, Yong-Lu Li, and Cewu Lu. 2024. Revisit human-scene interaction via space occupancy. In European Conference on Computer Vision. Springer, 1\u201319."},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"crossref","unstructured":"Matthew Loper Naureen Mahmood Javier Romero Gerard Pons-Moll and Michael\u00a0J Black. 2015. SMPL: A Skinned Multi-Person Linear Model. ACM Transactions on Graphics 34 6 (2015).","DOI":"10.1145\/2816795.2818013"},{"key":"e_1_3_3_2_28_1","unstructured":"Jintao Lu He Zhang Yuting Ye Takaaki Shiratori Sebastian Starke and Taku Komura. 2024. CHOICE: Coordinated human-object interaction in cluttered environments for pick-and-place actions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.06702 (2024)."},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV62453.2024.00061"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"crossref","unstructured":"Eley Ng Ziang Liu and Monroe Kennedy. 2023. Diffusion co-policy for synergistic human-robot collaborative tasks. IEEE Robotics and Automation Letters 9 1 (2023) 215\u2013222.","DOI":"10.1109\/LRA.2023.3330663"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV62453.2024.00149"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2004.434"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"crossref","unstructured":"Brian Parkinson. 2008. Emotions in direct and remote social interaction: Getting through the spaces between us. Computers in Human Behavior 24 4 (2008) 1510\u20131529.","DOI":"10.1016\/j.chb.2007.05.006"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW67362.2025.00271"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01383"},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-05311-5_9"},{"key":"e_1_3_3_2_37_1","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748\u20138763."},{"key":"e_1_3_3_2_38_1","first-page":"8821","volume-title":"International conference on machine learning","author":"Ramesh Aditya","year":"2021","unstructured":"Aditya Ramesh, Mikhail Pavlov, Gabriel Goh, Scott Gray, Chelsea Voss, Alec Radford, Mark Chen, and Ilya Sutskever. 2021. Zero-shot text-to-image generation. In International conference on machine learning. Pmlr, 8821\u20138831."},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"crossref","unstructured":"Roel Rutten. 2017. Beyond proximities: The socio-spatial dynamics of knowledge creation. Progress in human geography 41 2 (2017) 159\u2013177.","DOI":"10.1177\/0309132516629003"},{"key":"e_1_3_3_2_41_1","unstructured":"Chitwan Saharia William Chan Saurabh Saxena Lala Li Jay Whang Emily\u00a0L Denton Kamyar Ghasemipour Raphael Gontijo\u00a0Lopes Burcu Karagol\u00a0Ayan Tim Salimans et\u00a0al. 2022. Photorealistic text-to-image diffusion models with deep language understanding. Advances in neural information processing systems 35 (2022) 36479\u201336494."},{"key":"e_1_3_3_2_42_1","unstructured":"Yonatan Shafir Guy Tevet Roy Kapon and Amit\u00a0H Bermano. 2023. Human motion diffusion as a generative prior. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.01418 (2023)."},{"key":"e_1_3_3_2_43_1","unstructured":"Kihyuk Sohn Honglak Lee and Xinchen Yan. 2015. Learning structured output representation using deep conditional generative models. Advances in neural information processing systems 28 (2015)."},{"key":"e_1_3_3_2_44_1","unstructured":"Jiaming Song Chenlin Meng and Stefano Ermon. 2020. Denoising diffusion implicit models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.02502 (2020)."},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00242"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"crossref","unstructured":"Sebastian Starke Yiwei Zhao Taku Komura and Kazi\u00a0A Zaman. 2020. Local motion phases for learning multi-contact character movements. ACM Trans. Graph. 39 4 (2020) 54.","DOI":"10.1145\/3386569.3392450"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"crossref","unstructured":"Sebastian Starke Yiwei Zhao Fabio Zinno and Taku Komura. 2021. Neural animation layering for synthesizing martial arts movements. ACM Transactions on Graphics (TOG) 40 4 (2021) 1\u201316.","DOI":"10.1145\/3450626.3459881"},{"key":"e_1_3_3_2_48_1","unstructured":"Guy Tevet Sigal Raab Brian Gordon Yonatan Shafir Daniel Cohen-Or and Amit\u00a0H Bermano. 2022. Human motion diffusion model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.14916 (2022)."},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01981"},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00928"},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"crossref","unstructured":"Yiwei Wang Yixuan Sheng Ji Wang and Wenlong Zhang. 2017. Optimal collision-free robot trajectory generation based on time series prediction of human motion. IEEE Robotics and Automation Letters 3 1 (2017) 226\u2013233.","DOI":"10.1109\/LRA.2017.2737486"},{"key":"e_1_3_3_2_52_1","unstructured":"Zan Wang Yixin Chen Tengyu Liu Yixin Zhu Wei Liang and Siyuan Huang. 2022a. Humanise: Language-conditioned human motion generation in 3d scenes. Advances in Neural Information Processing Systems 35 (2022) 14959\u201314971."},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01986"},{"key":"e_1_3_3_2_54_1","unstructured":"Qianyang Wu Ye Shi Xiaoshui Huang Jingyi Yu Lan Xu and Jingya Wang. 2024b. Thor: Text to human-object interaction diffusion via relation intervention. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.11208 (2024)."},{"key":"e_1_3_3_2_55_1","unstructured":"Zhen Wu Jiaman Li Pei Xu and C\u00a0Karen Liu. 2024a. Human-object interaction from human-level instructions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.17840 (2024)."},{"key":"e_1_3_3_2_56_1","unstructured":"Zeqi Xiao Tai Wang Jingbo Wang Jinkun Cao Wenwei Zhang Bo Dai Dahua Lin and Jiangmiao Pang. 2023. Unified human-scene interaction via prompted chain-of-contacts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.07918 (2023)."},{"key":"e_1_3_3_2_57_1","unstructured":"Yiming Xie Varun Jampani Lei Zhong Deqing Sun and Huaizu Jiang. 2023. Omnicontrol: Control any joint at any time for human motion generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.08580 (2023)."},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00212"},{"key":"e_1_3_3_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00173"},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01371"},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"crossref","unstructured":"Sirui Xu Yu-Xiong Wang Liangyan Gui et\u00a0al. 2024a. Interdreamer: Zero-shot text to 3d dynamic human-object interaction. Advances in Neural Information Processing Systems 37 (2024) 52858\u201352890.","DOI":"10.52202\/079017-1675"},{"key":"e_1_3_3_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10610944"},{"key":"e_1_3_3_2_63_1","first-page":"246","volume-title":"European Conference on Computer Vision","author":"Yi Hongwei","year":"2024","unstructured":"Hongwei Yi, Justus Thies, Michael\u00a0J Black, Xue\u00a0Bin Peng, and Davis Rempe. 2024. Generating human interaction motions in scenes with text control. In European Conference on Computer Vision. Springer, 246\u2013263."},{"key":"e_1_3_3_2_64_1","doi-asserted-by":"crossref","unstructured":"Mingyuan Zhang Zhongang Cai Liang Pan Fangzhou Hong Xinying Guo Lei Yang and Ziwei Liu. 2024. Motiondiffuse: Text-driven human motion generation with diffusion model. IEEE transactions on pattern analysis and machine intelligence 46 6 (2024) 4115\u20134128.","DOI":"10.1109\/TPAMI.2024.3355414"},{"key":"e_1_3_3_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00040"},{"key":"e_1_3_3_2_66_1","volume-title":"International Conference on 3D Vision (3DV)","volume":"2","author":"Zhang Siwei","year":"2020","unstructured":"Siwei Zhang, Yan Zhang, Qianli Ma, Michael\u00a0J Black, and Siyu Tang. 2020b. Generating person-scene interactions in 3d scenes. In International Conference on 3D Vision (3DV) , Vol.\u00a02."},{"key":"e_1_3_3_2_67_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20065-6_30"},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00623"},{"key":"e_1_3_3_2_69_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20068-7_18"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763954","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:25:05Z","timestamp":1765250705000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763954"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":68,"alternative-id":["10.1145\/3757377.3763954","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763954","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}