{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:02:33Z","timestamp":1784268153408,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,13]],"date-time":"2024-07-13T00:00:00Z","timestamp":1720828800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Shenzhen Science and Technology Program","award":["RCYX20210609103121030"],"award-info":[{"award-number":["RCYX20210609103121030"]}]},{"name":"Guangdong Natural Science Foundation","award":["2021B1515020085"],"award-info":[{"award-number":["2021B1515020085"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62322207"],"award-info":[{"award-number":["62322207"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,13]]},"DOI":"10.1145\/3641519.3657422","type":"proceedings-article","created":{"date-parts":[[2024,7,12]],"date-time":"2024-07-12T10:39:28Z","timestamp":1720780768000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":19,"title":["LGTM: Local-to-Global Text-Driven Human Motion Diffusion Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6450-7746","authenticated-orcid":false,"given":"Haowen","family":"Sun","sequence":"first","affiliation":[{"name":"Shenzhen University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8178-3886","authenticated-orcid":false,"given":"Ruikun","family":"Zheng","sequence":"additional","affiliation":[{"name":"Shenzhen University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7787-6428","authenticated-orcid":false,"given":"Haibin","family":"Huang","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8243-9513","authenticated-orcid":false,"given":"Chongyang","family":"Ma","sequence":"additional","affiliation":[{"name":"ByteDance Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3212-0544","authenticated-orcid":false,"given":"Hui","family":"Huang","sequence":"additional","affiliation":[{"name":"Shenzhen University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6798-0336","authenticated-orcid":false,"given":"Ruizhen","family":"Hu","sequence":"additional","affiliation":[{"name":"Shenzhen University, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,13]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2019.00084"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592458"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV57658.2022.00053"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591487"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.visinf.2021.10.003"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01726"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","unstructured":"Xin Chen Biao Jiang Wen Liu Zilong Huang Bin Fu Tao Chen Jingyi Yu and Gang Yu. 2023b. Executing Your Commands via Motion Diffusion in Latent Space. https:\/\/doi.org\/10.48550\/arXiv.2212.04048 arxiv:2212.04048\u00a0[cs]","DOI":"10.48550\/arXiv.2212.04048"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00143"},{"key":"e_1_3_2_2_9_1","volume-title":"Conformer: Convolution-augmented Transformer for Speech Recognition. arxiv:2005.08100\u00a0[cs, eess]","author":"Gulati Anmol","year":"2020","unstructured":"Anmol Gulati, James Qin, Chung-Cheng Chiu, Niki Parmar, Yu Zhang, Jiahui Yu, Wei Han, Shibo Wang, Zhengdong Zhang, Yonghui Wu, and Ruoming Pang. 2020. Conformer: Convolution-augmented Transformer for Speech Recognition. arxiv:2005.08100\u00a0[cs, eess]"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00509"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_34"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1360612.1360626"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3516429"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-007-0200-1"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","unstructured":"Biao Jiang Xin Chen Wen Liu Jingyi Yu Gang Yu and Tao Chen. 2023. MotionGPT: Human Motion as a Foreign Language. https:\/\/doi.org\/10.48550\/arXiv.2306.14795 arxiv:2306.14795\u00a0[cs]","DOI":"10.48550\/arXiv.2306.14795"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.3390\/fi15090300"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3550454.3555489"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2816795.2818013"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20047-2_28"},{"key":"e_1_3_2_2_20_1","volume-title":"TMR: Text-to-Motion Retrieval Using Contrastive 3D Human Motion Synthesis. arxiv:2305.00976\u00a0[cs]","author":"Petrovich Mathis","year":"2023","unstructured":"Mathis Petrovich, Michael\u00a0J. Black, and G\u00fcl Varol. 2023. TMR: Text-to-Motion Retrieval Using Contrastive 3D Human Motion Synthesis. arxiv:2305.00976\u00a0[cs]"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","unstructured":"Ben Poole Ajay Jain Jonathan\u00a0T. Barron and Ben Mildenhall. 2022. DreamFusion: Text-to-3D Using 2D Diffusion. https:\/\/doi.org\/10.48550\/arXiv.2209.14988 arxiv:2209.14988\u00a0[cs stat]","DOI":"10.48550\/arXiv.2209.14988"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","unstructured":"Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark Gretchen Krueger and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. https:\/\/doi.org\/10.48550\/arXiv.2103.00020 arxiv:2103.00020\u00a0[cs]","DOI":"10.48550\/arXiv.2103.00020"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.fsidi.2023.301609"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2945078.2945107"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","unstructured":"Jiaming Song Chenlin Meng and Stefano Ermon. 2022. Denoising Diffusion Implicit Models. https:\/\/doi.org\/10.48550\/arXiv.2010.02502 arxiv:2010.02502\u00a0[cs]","DOI":"10.48550\/arXiv.2010.02502"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3355089.3356505"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3386569.3392450"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3450626.3459881"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20047-2_21"},{"key":"e_1_3_2_2_31_1","volume-title":"Human Motion Diffusion Model. In The Eleventh International Conference on Learning Representations.","author":"Tevet Guy","year":"2022","unstructured":"Guy Tevet, Sigal Raab, Brian Gordon, Yoni Shafir, Daniel Cohen-or, and Amit\u00a0Haim Bermano. 2022b. Human Motion Diffusion Model. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_32_1","volume-title":"Advances in Neural Information Processing Systems, Vol.\u00a030. Curran Associates","author":"Vaswani Ashish","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention Is All You Need. In Advances in Neural Information Processing Systems, Vol.\u00a030. Curran Associates, Inc."},{"key":"e_1_3_2_2_33_1","unstructured":"Heyuan Yao Zhenhua Song Yuyang Zhou Tenglong Ao Baoquan Chen and Libin Liu. [n. d.]. MoConVQ: Unified Physics-Based Motion Control via Scalable Discrete Representations. ([n. d.])."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","unstructured":"Ye Yuan Jiaming Song Umar Iqbal Arash Vahdat and Jan Kautz. 2022. PhysDiff: Physics-Guided Human Motion Diffusion Model. https:\/\/doi.org\/10.48550\/arXiv.2212.02500 arxiv:2212.02500\u00a0[cs]","DOI":"10.48550\/arXiv.2212.02500"},{"key":"e_1_3_2_2_35_1","volume-title":"SmoothNet: A Plug-and-Play Network for Refining Human Poses in Videos. In European Conference on Computer Vision. Springer.","author":"Zeng Ailing","year":"2022","unstructured":"Ailing Zeng, Lei Yang, Xuan Ju, Jiefeng Li, Jianyi Wang, and Qiang Xu. 2022. SmoothNet: A Plug-and-Play Network for Refining Human Poses in Videos. In European Conference on Computer Vision. Springer."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201366"},{"key":"e_1_3_2_2_37_1","unstructured":"Mingyuan Zhang Zhongang Cai Liang Pan Fangzhou Hong Xinying Guo Lei Yang and Ziwei Liu. 2022. MotionDiffuse: Text-Driven Human Motion Generation with Diffusion Model. arxiv:2208.15001\u00a0[cs]"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.visinf.2022.03.002"}],"event":{"name":"SIGGRAPH '24: Special Interest Group on Computer Graphics and Interactive Techniques Conference","location":"Denver CO USA","acronym":"SIGGRAPH '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657422","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641519.3657422","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:09:36Z","timestamp":1750295376000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657422"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,13]]},"references-count":38,"alternative-id":["10.1145\/3641519.3657422","10.1145\/3641519"],"URL":"https:\/\/doi.org\/10.1145\/3641519.3657422","relation":{},"subject":[],"published":{"date-parts":[[2024,7,13]]},"assertion":[{"value":"2024-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}