{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:02:04Z","timestamp":1784268124980,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,13]],"date-time":"2024-07-13T00:00:00Z","timestamp":1720828800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,13]]},"DOI":"10.1145\/3641519.3657447","type":"proceedings-article","created":{"date-parts":[[2024,7,12]],"date-time":"2024-07-12T10:39:28Z","timestamp":1720780768000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":28,"title":["Iterative Motion Editing with Natural Language"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2618-092X","authenticated-orcid":false,"given":"Purvi","family":"Goel","sequence":"first","affiliation":[{"name":"Stanford University, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6785-8146","authenticated-orcid":false,"given":"Kuan-Chieh","family":"Wang","sequence":"additional","affiliation":[{"name":"Snap Inc., United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5926-0905","authenticated-orcid":false,"given":"C. Karen","family":"Liu","sequence":"additional","affiliation":[{"name":"Stanford University, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8754-0429","authenticated-orcid":false,"given":"Kayvon","family":"Fatahalian","sequence":"additional","affiliation":[{"name":"Stanford University, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,13]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3386569.3392469"},{"key":"e_1_3_2_2_2_1","unstructured":"Maneesh Agrawala. 2023. Unpredictable Black Boxes are Terrible Interfaces. https:\/\/magrawala.substack.com\/p\/unpredictable-black-boxes-are-terrible."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3272127.3275038"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"e_1_3_2_2_5_1","volume-title":"Computer Vision \u2013 ECCV 2016(Lecture Notes in Computer Science)","author":"Bogo Federica","unstructured":"Federica Bogo, Angjoo Kanazawa, Christoph Lassner, Peter Gehler, Javier Romero, and Michael\u00a0J. Black. 2016. Keep it SMPL: Automatic Estimation of 3D Human Pose and Shape from a Single Image. In Computer Vision \u2013 ECCV 2016(Lecture Notes in Computer Science). Springer International Publishing."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Tim Brooks Aleksander Holynski and Alexei\u00a0A. Efros. 2023. InstructPix2Pix: Learning to Follow Image Editing Instructions. In CVPR.","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01379"},{"key":"e_1_3_2_2_8_1","volume-title":"Motion Question Answering via Modular Motion Programs. ICML","author":"Endo Mark","year":"2023","unstructured":"Mark Endo, Joy Hsu, Jiaman Li, and Jiajun Wu. 2023. Motion Question Answering via Modular Motion Programs. ICML (2023)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Yao Feng Jing Lin Sai\u00a0Kumar Dwivedi Yu Sun Priyanka Patel and Michael\u00a0J. Black. 2023. PoseGPT: Chatting about 3D Human Pose. arxiv:2311.18836\u00a0[cs.CV]","DOI":"10.1109\/CVPR52733.2024.00204"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/253284.253321"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/364338.364400"},{"key":"e_1_3_2_2_12_1","unstructured":"Tao Gong Chengqi Lyu Shilong Zhang Yudong Wang Miao Zheng Qian Zhao Kuikun Liu Wenwei Zhang Ping Luo and Kai Chen. 2023. MultiModal-GPT: A Vision and Language Model for Dialogue with Humans. arxiv:2305.04790\u00a0[cs.CV]"},{"key":"e_1_3_2_2_13_1","unstructured":"Deepak Gopinath and Jungdam Won. 2020. fairmotion - Tools to load process and visualize motion capture data. Github. https:\/\/github.com\/facebookresearch\/fairmotion"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Chuan Guo Yuxuan Mu Muhammad\u00a0Gohar Javed Sen Wang and Li Cheng. 2023. MoMask: Generative Masked Modeling of 3D Human Motions. (2023). arxiv:2312.00063\u00a0[cs.CV]","DOI":"10.1109\/CVPR52733.2024.00186"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00509"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2682626"},{"key":"e_1_3_2_2_17_1","unstructured":"Amir Hertz Ron Mokady Jay Tenenbaum Kfir Aberman Yael Pritch and Daniel Cohen-Or. 2022. Prompt-to-prompt image editing with cross attention control. (2022)."},{"key":"e_1_3_2_2_18_1","volume-title":"Denoising Diffusion Probabilistic Models. arXiv preprint arxiv:2006.11239","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. arXiv preprint arxiv:2006.11239 (2020)."},{"key":"e_1_3_2_2_19_1","volume-title":"Language Models as Zero-Shot Planners: Extracting Actionable Knowledge for Embodied Agents. CoRR abs\/2201.07207","author":"Huang Wenlong","year":"2022","unstructured":"Wenlong Huang, Pieter Abbeel, Deepak Pathak, and Igor Mordatch. 2022. Language Models as Zero-Shot Planners: Extracting Actionable Knowledge for Embodied Agents. CoRR abs\/2201.07207 (2022). arXiv:2201.07207https:\/\/arxiv.org\/abs\/2201.07207"},{"key":"e_1_3_2_2_20_1","volume-title":"MotionGPT: Human Motion as a Foreign Language. arXiv preprint arXiv:2306.14795","author":"Jiang Biao","year":"2023","unstructured":"Biao Jiang, Xin Chen, Wen Liu, Jingyi Yu, Gang Yu, and Tao Chen. 2023. MotionGPT: Human Motion as a Foreign Language. arXiv preprint arXiv:2306.14795 (2023)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00603"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Korrawe Karunratanakul Konpat Preechakul Emre Aksan Thabo Beeler Supasorn Suwajanakorn and Siyu Tang. 2023. Optimizing Diffusion Noise Can Serve As Universal Motion Priors. In arxiv:2312.11994.","DOI":"10.1109\/CVPR52733.2024.00133"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3596711.3596788"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"crossref","unstructured":"Sumith Kulal Jiayuan Mao Alex Aiken and Jiajun Wu. 2021. Hierarchical Motion Understanding via Motion Programs. In CVPR.","DOI":"10.1109\/CVPR46437.2021.00650"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/311535.311539"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3618333"},{"key":"e_1_3_2_2_27_1","unstructured":"Ruilong Li Shan Yang David\u00a0A. Ross and Angjoo Kanazawa. 2021. AI Choreographer: Music Conditioned 3D Dance Generation with AIST++. arxiv:2101.08779\u00a0[cs.CV]"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00554"},{"key":"e_1_3_2_2_30_1","unstructured":"Chenlin Meng Yutong He Yang Song Jiaming Song Jiajun Wu Jun-Yan Zhu and Stefano Ermon. 2022. SDEdit: Guided Image Synthesis and Editing with Stochastic Differential Equations. arxiv:2108.01073\u00a0[cs.CV]"},{"key":"e_1_3_2_2_31_1","volume-title":"Generative Proxemics: A Prior for 3D Social Interaction from Images. arXiv preprint arXiv:2306.09337","author":"M\u00fcller Lea","year":"2023","unstructured":"Lea M\u00fcller, Vickie Ye, Georgios Pavlakos, Michael Black, and Angjoo Kanazawa. 2023. Generative Proxemics: A Prior for 3D Social Interaction from Images. arXiv preprint arXiv:2306.09337 (2023)."},{"key":"e_1_3_2_2_32_1","volume-title":"International Conference on Learning Representations.","author":"Oreshkin N.","year":"2022","unstructured":"Boris\u00a0N. Oreshkin, Florent Bocquelet, F\u00e9lix\u00a0G. Harvey, Bay Raitt, and Dominic Laflamme. 2022. ProtoRes: Proto-Residual Network for Pose Authoring via Learned Inverse Kinematics. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3550454.3555454"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01322"},{"key":"e_1_3_2_2_35_1","volume-title":"InsActor: Instruction-driven Physics-based Characters. NeurIPS","author":"Ren Jiawei","year":"2023","unstructured":"Jiawei Ren, Mingyuan Zhang, Cunjun Yu, Xiao Ma, Liang Pan, and Ziwei Liu. 2023. InsActor: Instruction-driven Physics-based Characters. NeurIPS (2023)."},{"key":"e_1_3_2_2_36_1","unstructured":"Vishnu Sarukkai Linden Li Arden Ma Christopher R\u00e9 and Kayvon Fatahalian. 2023. Collage Diffusion. arxiv:2303.00262\u00a0[cs.CV]"},{"key":"e_1_3_2_2_37_1","volume-title":"Human motion diffusion as a generative prior. arXiv preprint arXiv:2303.01418","author":"Shafir Yonatan","year":"2023","unstructured":"Yonatan Shafir, Guy Tevet, Roy Kapon, and Amit\u00a0H Bermano. 2023. Human motion diffusion as a generative prior. arXiv preprint arXiv:2303.01418 (2023)."},{"key":"e_1_3_2_2_38_1","volume-title":"Reflexion: Language Agents with Verbal Reinforcement Learning. arxiv:2303.11366\u00a0[cs.AI]","author":"Shinn Noah","year":"2023","unstructured":"Noah Shinn, Federico Cassano, Edward Berman, Ashwin Gopinath, Karthik Narasimhan, and Shunyu Yao. 2023. Reflexion: Language Agents with Verbal Reinforcement Learning. arxiv:2303.11366\u00a0[cs.AI]"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161317"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01092"},{"key":"e_1_3_2_2_41_1","volume-title":"Human Motion Diffusion Model. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SJ1kSyO2jwu","author":"Tevet Guy","year":"2023","unstructured":"Guy Tevet, Sigal Raab, Brian Gordon, Yoni Shafir, Daniel Cohen-or, and Amit\u00a0Haim Bermano. 2023. Human Motion Diffusion Model. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SJ1kSyO2jwu"},{"key":"e_1_3_2_2_42_1","volume-title":"EDGE: Editable Dance Generation From Music. arXiv preprint arXiv:2211.10658","author":"Tseng Jonathan","year":"2022","unstructured":"Jonathan Tseng, Rodrigo Castellon, and C\u00a0Karen Liu. 2022. EDGE: Editable Dance Generation From Music. arXiv preprint arXiv:2211.10658 (2022)."},{"key":"e_1_3_2_2_43_1","volume-title":"Voyager: An Open-Ended Embodied Agent with Large Language Models. arxiv:2305.16291\u00a0[cs.AI]","author":"Wang Guanzhi","year":"2023","unstructured":"Guanzhi Wang, Yuqi Xie, Yunfan Jiang, Ajay Mandlekar, Chaowei Xiao, Yuke Zhu, Linxi Fan, and Anima Anandkumar. 2023b. Voyager: An Open-Ended Embodied Agent with Large Language Models. arxiv:2305.16291\u00a0[cs.AI]"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02119"},{"key":"e_1_3_2_2_45_1","volume-title":"Weiqing Li, and Jian-Zhou Lu.","author":"Wei Dong","year":"2023","unstructured":"Dong Wei, Xiaoning Sun, Huaijiang Sun, Bin Li, Sheng liang Hu, Weiqing Li, and Jian-Zhou Lu. 2023a. Understanding Text-driven Motion Synthesis with Keyframe Collaboration via Diffusion Models. ArXiv abs\/2305.13773 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258841591"},{"key":"e_1_3_2_2_46_1","volume-title":"Chi, Quoc Le, and Denny Zhou","author":"Wei Jason","year":"2023","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed Chi, Quoc Le, and Denny Zhou. 2023b. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. arxiv:2201.11903\u00a0[cs.CL]"},{"key":"e_1_3_2_2_47_1","volume-title":"Proceedings of the 22nd annual conference on Computer graphics and interactive techniques","author":"P.","year":"1995","unstructured":"Andrew\u00a0P. Witkin and Zoran Popovic. 1995. Motion warping. Proceedings of the 22nd annual conference on Computer graphics and interactive techniques (1995). https:\/\/api.semanticscholar.org\/CorpusID:1497012"},{"key":"e_1_3_2_2_48_1","unstructured":"Wilson Yan Yunzhi Zhang Pieter Abbeel and Aravind Srinivas. 2021. VideoGPT: Video Generation using VQ-VAE and Transformers. arxiv:2104.10157\u00a0[cs.CV]"},{"key":"e_1_3_2_2_49_1","unstructured":"Shunyu Yao Jeffrey Zhao Dian Yu Nan Du Izhak Shafran Karthik Narasimhan and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. arxiv:2210.03629\u00a0[cs.CL]"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00503"},{"key":"e_1_3_2_2_51_1","volume-title":"MotionDiffuse: Text-Driven Human Motion Generation with Diffusion Model. arXiv preprint arXiv:2208.15001","author":"Zhang Mingyuan","year":"2022","unstructured":"Mingyuan Zhang, Zhongang Cai, Liang Pan, Fangzhou Hong, Xinying Guo, Lei Yang, and Ziwei Liu. 2022. MotionDiffuse: Text-Driven Human Motion Generation with Diffusion Model. arXiv preprint arXiv:2208.15001 (2022)."},{"key":"e_1_3_2_2_52_1","volume-title":"FineMoGen: Fine-Grained Spatio-Temporal Motion Generation and Editing. NeurIPS","author":"Zhang Mingyuan","year":"2023","unstructured":"Mingyuan Zhang, Huirong Li, Zhongang Cai, Jiawei Ren, Lei Yang, and Ziwei Liu. 2023. FineMoGen: Fine-Grained Spatio-Temporal Motion Generation and Editing. NeurIPS (2023)."}],"event":{"name":"SIGGRAPH '24: Special Interest Group on Computer Graphics and Interactive Techniques Conference","location":"Denver CO USA","acronym":"SIGGRAPH '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657447","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641519.3657447","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:09:36Z","timestamp":1750295376000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641519.3657447"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,13]]},"references-count":52,"alternative-id":["10.1145\/3641519.3657447","10.1145\/3641519"],"URL":"https:\/\/doi.org\/10.1145\/3641519.3657447","relation":{},"subject":[],"published":{"date-parts":[[2024,7,13]]},"assertion":[{"value":"2024-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}