{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T19:05:59Z","timestamp":1784228759304,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":86,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,19]]},"DOI":"10.1145\/3799902.3811093","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T16:15:27Z","timestamp":1784218527000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Go-with-the-Track: Video Compositing and Motion Control with Point Tracking"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-6146-5659","authenticated-orcid":false,"given":"Koichi","family":"Namekata","sequence":"first","affiliation":[{"name":"Eyeline Labs, Oxford, United Kingdom and University of Oxford, Oxford, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8347-4895","authenticated-orcid":false,"given":"Yash","family":"Kant","sequence":"additional","affiliation":[{"name":"Eyeline Labs, Los Angeles, USA and Netflix, Los Angeles, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9426-3718","authenticated-orcid":false,"given":"Zhizheng","family":"Liu","sequence":"additional","affiliation":[{"name":"Eyeline Labs, Los Angeles, USA and University of California, Los Angeles, Los Angeles, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5947-2076","authenticated-orcid":false,"given":"Ryan","family":"Burgert","sequence":"additional","affiliation":[{"name":"Eyeline Labs, New York, USA and Stony Brook University, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2254-5752","authenticated-orcid":false,"given":"Yuancheng","family":"Xu","sequence":"additional","affiliation":[{"name":"Eyeline Labs, Los Angeles, USA and Netflix, Los Angeles, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0605-5862","authenticated-orcid":false,"given":"Kuan Heng","family":"Lin","sequence":"additional","affiliation":[{"name":"Eyeline Labs, New York, USA and Columbia University, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2852-4444","authenticated-orcid":false,"given":"Emmett","family":"Steven","sequence":"additional","affiliation":[{"name":"Netflix, Los Angeles, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3125-1614","authenticated-orcid":false,"given":"Julien","family":"Philip","sequence":"additional","affiliation":[{"name":"Eyeline Labs, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6992-0089","authenticated-orcid":false,"given":"Li","family":"Ma","sequence":"additional","affiliation":[{"name":"Eyeline Labs, Los Angeles, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1374-2858","authenticated-orcid":false,"given":"Andrea","family":"Vedaldi","sequence":"additional","affiliation":[{"name":"University of Oxford, Oxford, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7381-2323","authenticated-orcid":false,"given":"Paul","family":"Debevec","sequence":"additional","affiliation":[{"name":"Eyeline Labs, Los Angeles, USA and Netflix, Los Angeles, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6865-1325","authenticated-orcid":false,"given":"Ning","family":"Yu","sequence":"additional","affiliation":[{"name":"Eyeline Labs, Los Angeles, USA and Netflix, Los Angeles, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_3_2_2_1","unstructured":"Artiprocher and ModelScope\u00a0AIGC Team. 2024. DiffSynth-Studio: An Open-Source Diffusion Engine for Image and Video Generation. https:\/\/github.com\/modelscope\/DiffSynth-Studio. Version 0.1.0 and later."},{"key":"e_1_3_3_2_3_1","unstructured":"Jianhong Bai Menghan Xia Xiao Fu Xintao Wang Lianrui Mu Jinwen Cao Zuozhu Liu Haoji Hu Xiang Bai Pengfei Wan et\u00a0al. 2025. ReCamMaster: Camera-Controlled Generative Rendering from A Single Video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.11647 (2025)."},{"key":"e_1_3_3_2_4_1","unstructured":"Yuxuan Bian Xin Chen Zenan Li Tiancheng Zhi Shen Sang Linjie Luo and Qiang Xu. 2025. Video-As-Prompt: Unified Semantic Control for Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.20888 (2025)."},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02161"},{"key":"e_1_3_3_2_6_1","unstructured":"Ryan Burgert Charles Herrmann Forrester Cole Michael\u00a0S Ryoo Neal Wadhwa Andrey Voynov and Nataniel Ruiz. 2025a. MotionV2V: Editing Motion in a Video. (2025). arxiv:https:\/\/arXiv.org\/abs\/2511.20640\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2511.20640"},{"key":"e_1_3_3_2_7_1","unstructured":"Ryan Burgert Yuancheng Xu Wenqi Xian Oliver Pilarski Pascal Clausen Mingming He Li Ma Yitong Deng Lingxiao Li Mohsen Mousavi Michael Ryoo Paul Debevec and Ning Yu. 2025b. Go-with-the-Flow: Motion-Controllable Video Diffusion Models Using Real-Time Warped Noise. arxiv:https:\/\/arXiv.org\/abs\/2501.08331\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2501.08331"},{"key":"e_1_3_3_2_8_1","unstructured":"Chenjie Cao Jingkai Zhou Shikai Li Jingyun Liang Chaohui Yu Fan Wang Xiangyang Xue and Yanwei Fu. 2025. Uni3C: Unifying Precisely 3D-Enhanced Camera and Human Motion Controls for Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.14899 (2025)."},{"key":"e_1_3_3_2_9_1","unstructured":"Nicolas Carion Laura Gustafson Yuan-Ting Hu Shoubhik Debnath Ronghang Hu Didac Suris Chaitanya Ryali Kalyan\u00a0Vasudev Alwala Haitham Khedr Andrew Huang Jie Lei Tengyu Ma Baishan Guo Arpit Kalla Markus Marks Joseph Greer Meng Wang Peize Sun Roman R\u00e4dle Triantafyllos Afouras Effrosyni Mavroudi Katherine Xu Tsung-Han Wu Yu Zhou Liliane Momeni Rishi Hazra Shuangrui Ding Sagar Vaze Francois Porcher Feng Li Siyuan Li Aishwarya Kamath Ho\u00a0Kei Cheng Piotr Doll\u00e1r Nikhila Ravi Kate Saenko Pengchuan Zhang and Christoph Feichtenhofer. 2025. SAM 3: Segment Anything with Concepts. arxiv:https:\/\/arXiv.org\/abs\/2511.16719\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2511.16719"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"crossref","unstructured":"Hila Chefer Shiran Zada Roni Paiss Ariel Ephrat Omer Tov Michael Rubinstein Lior Wolf Tali Dekel Tomer Michaeli and Inbar Mosseri. 2024. Still-Moving: Customized Video Generation without Customized Video Data. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.08674 (2024).","DOI":"10.1145\/3687945"},{"key":"e_1_3_3_2_11_1","unstructured":"Hong Chen Xin Wang Guanning Zeng Yipeng Zhang Yuwei Zhou Feilin Han and Wenwu Zhu. 2023. VideoDreamer: Customized Multi-Subject Text-to-Video Generation with Disen-Mix Finetuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.00990 (2023)."},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00572"},{"key":"e_1_3_3_2_13_1","unstructured":"Gang Cheng Xin Gao Li Hu Siqi Hu Mingyang Huang Chaonan Ji Ju Li Dechao Meng Jinwei Qi Penchong Qiao et\u00a0al. 2025. Wan-Animate: Unified Character Animation and Replacement with Holistic Replication. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.14055 (2025)."},{"key":"e_1_3_3_2_14_1","unstructured":"Ruihang Chu Yefei He Zhekai Chen Shiwei Zhang Xiaogang Xu Bin Xia Dingdong Wang Hongwei Yi Xihui Liu Hengshuang Zhao et\u00a0al. 2025. Wan-move: Motion-controllable video generation via latent trajectory guidance. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2512.08765 (2025)."},{"key":"e_1_3_3_2_15_1","unstructured":"Yufan Deng Xun Guo Yuanyang Yin Jacob\u00a0Zhiyuan Fang Yiding Yang Yizhi Wang Shenghai Yuan Angtian Wang Bo Liu Haibin Huang and Chongyang Ma. 2025. MAGREF: Masked Guidance for Any-Reference Video Generation with Subject Disentanglement. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.23742 (2025)."},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"crossref","unstructured":"Carl Doersch Pauline Luc Yi Yang Dilara Gokay Skanda Koppula Ankush Gupta Joseph Heyward Ignacio Rocco Ross Goroshin Jo\u00e3o Carreira and Andrew Zisserman. 2024. BootsTAP: Bootstrapped Training for Tracking-Any-Point. arxiv:https:\/\/arXiv.org\/abs\/2402.00847\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2402.00847","DOI":"10.1007\/978-981-96-0901-7_28"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00010"},{"key":"e_1_3_3_2_18_1","unstructured":"Genmo Team. 2024. Mochi 1. https:\/\/github.com\/genmoai\/models."},{"key":"e_1_3_3_2_19_1","unstructured":"Michal Geyer Omer Bar-Tal Shai Bagon and Tali Dekel. 2023. TokenFlow: Consistent Diffusion Features for Consistent Video Editing. arXiv preprint arxiv:https:\/\/arXiv.org\/abs\/2307.10373 (2023)."},{"key":"e_1_3_3_2_20_1","unstructured":"Simon Giebenhain Tobias Kirschstein R\u00fcnz R\u00fcnz Lourdes Agapito and Matthias Nie\u00dfner. 2025. Pixel3DMM: Versatile Screen-Space Priors for Single-Image 3D Face Reconstruction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.00615 (2025)."},{"key":"e_1_3_3_2_21_1","volume-title":"Gemini 2.5 Flash and Gemini 2.5 Flash Image Model Card","author":"DeepMind Google","year":"2025","unstructured":"Google DeepMind. 2025. Gemini 2.5 Flash and Gemini 2.5 Flash Image Model Card. Technical Report. Google DeepMind. https:\/\/storage.googleapis.com\/deepmind-media\/Model-Cards\/Gemini-2-5-Flash-Model-Card.pdf Also known as \u201cNano Banana\u201d (Gemini 2.5 Flash Image)."},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00373"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3721238.3730607"},{"key":"e_1_3_3_2_24_1","unstructured":"Zekai Gu Rui Yan Jiahao Lu Peng Li Zhiyang Dou Chenyang Si Zhen Dong Qifeng Liu Cheng Lin Ziwei Liu Wenping Wang and Yuan Liu. 2025b. Diffusion as Shader: 3D-aware Video Diffusion for Versatile Video Generation Control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.03847 (2025)."},{"key":"e_1_3_3_2_25_1","unstructured":"Yoav HaCohen Benny Brazowski Nisan Chiprut Yaki Bitterman Andrew Kvochko Avishai Berkowitz Daniel Shalem Daphna Lifschitz Dudu Moshe Eitan Porat Eitan Richardson Guy Shiran Itay Chachy Jonathan Chetboun Michael Finkelson Michael Kupchick Nir Zabari Nitzan Guetta Noa Kotler Ofir Bibi Ori Gordon Poriya Panet Roi Benita Shahar Armon Victor Kulikov Yaron Inger Yonatan Shiftan Zeev Melumian and Zeev Farbman. 2026. LTX-2: Efficient Joint Audio-Visual Foundation Model. (2026). arxiv:https:\/\/arXiv.org\/abs\/2601.03233\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2601.03233"},{"key":"e_1_3_3_2_26_1","unstructured":"Yoav HaCohen Nisan Chiprut Benny Brazowski Daniel Shalem Dudu Moshe Eitan Richardson Eran Levin Guy Shiran Nir Zabari Ori Gordon Poriya Panet Sapir Weissbuch Victor Kulikov Yaki Bitterman Zeev Melumian and Ofir Bibi. 2024. LTX-Video: Realtime Video Latent Diffusion. (2024). arxiv:https:\/\/arXiv.org\/abs\/2501.00103\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2501.00103"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01246"},{"key":"e_1_3_3_2_28_1","unstructured":"Kai He Ruofan Liang Jacob Munkberg Jon Hasselgren Nandita Vijaykumar Alexander Keller Sanja Fidler Igor Gilitschenski Zan Gojcic and Zian Wang. 2025a. UniRelight: Learning Joint Decomposition and Synthesis for Video Relighting. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.15673 (2025)."},{"key":"e_1_3_3_2_29_1","unstructured":"Cheng Hou et\u00a0al. 2024. Training-free Camera Control for Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.10126 (2024)."},{"key":"e_1_3_3_2_30_1","unstructured":"Tao Hu Haoyang Peng Xiao Liu and Yuewen Ma. 2025a. EX-4D: EXtreme Viewpoint 4D Video Synthesis via Depth Watertight Mesh. arxiv:https:\/\/arXiv.org\/abs\/2506.05554\u00a0[cs.CV]"},{"key":"e_1_3_3_2_31_1","unstructured":"Teng Hu Zhentao Yu Zhengguang Zhou Sen Liang Yuan Zhou Qin Lin and Qinglin Lu. 2025b. HunyuanCustom: A Multimodal-Driven Architecture for Customized Video Generation. arxiv:https:\/\/arXiv.org\/abs\/2505.04512\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2505.04512"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02121"},{"key":"e_1_3_3_2_33_1","unstructured":"Yuming Jiang Tianxing Wu Shuai Yang Chenyang Si Dahua Lin Yu Qiao Chen\u00a0Change Loy and Ziwei Liu. 2023. VideoBooth: Diffusion-based Video Generation with Image Prompts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.00777 (2023)."},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01597"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00196"},{"key":"e_1_3_3_2_36_1","unstructured":"Xuan Ju Yiming Gao Zhaoyang Zhang Ziyang Yuan Xintao Wang Ailing Zeng Yu Xiong Qiang Xu and Ying Shan. 2024. MiraData: A Large-Scale Video Dataset with Long Durations and Structured Captions. arxiv:https:\/\/arXiv.org\/abs\/2407.06358\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2407.06358"},{"key":"e_1_3_3_2_37_1","unstructured":"Xuan Ju Tianyu Wang Yuqian Zhou He Zhang Qing Liu Nanxuan Zhao Zhifei Zhang Yijun Li Yuanhao Cai Shaoteng Liu Daniil Pakhomov Zhe Lin Soo\u00a0Ye Kim and Qiang Xu. 2025. EditVerse: Unifying Image and Video Editing and Generation with In-Context Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.20360 (2025). https:\/\/arxiv.org\/abs\/2509.20360"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01531"},{"key":"e_1_3_3_2_39_1","unstructured":"Nikita Karaev Ignacio Rocco et\u00a0al. 2024a. CoTracker3: Simpler and Better Point Tracking by Pseudo Videos. arXiv:https:\/\/arXiv.org\/abs\/2410.11831 (2024)."},{"key":"e_1_3_3_2_40_1","volume-title":"ECCV","author":"Karaev Nikita","year":"2024","unstructured":"Nikita Karaev, Ignacio Rocco, Benjamin Graham, Natalia Neverova, Andrea Vedaldi, and Christian Rupprecht. 2024b. CoTracker: It is Better to Track Together. In ECCV."},{"key":"e_1_3_3_2_41_1","unstructured":"Weijie Kong et\u00a0al. 2025. HunyuanVideo: A Systematic Framework For Large Video Generative Models. arxiv:https:\/\/arXiv.org\/abs\/2412.03603\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2412.03603"},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"crossref","unstructured":"Skanda Koppula Ignacio Rocco Yi Yang Joe Heyward Jo\u00e3o Carreira Andrew Zisserman Gabriel Brostow and Carl Doersch. 2024. TAPVid-3D: A Benchmark for Tracking Any Point in 3D. arxiv:https:\/\/arXiv.org\/abs\/2407.05921\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2407.05921","DOI":"10.52202\/079017-2611"},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"crossref","unstructured":"Cheng Lei Jiayu Zhang Yue Ma Xinyu Wang Long Chen Liang Tang Yiqiang Yan Fei Su and Zhicheng Zhao. 2025. DiTraj: Training-free Trajectory Control for Video Diffusion Transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.21839 (2025).","DOI":"10.2139\/ssrn.6947474"},{"key":"e_1_3_3_2_44_1","unstructured":"Baolu Li Yiming Zhang Qinghe Wang Liqian Ma Xiaoyu Shi Xintao Wang Pengfei Wan Zhenfei Yin Yunzhi Zhuge Huchuan Lu et\u00a0al. 2025b. VFXMaster: Unlocking Dynamic Visual Effect Generation via In-Context Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.25772 (2025)."},{"key":"e_1_3_3_2_45_1","unstructured":"Hui Li Mingwang Xu Yun Zhan Shan Mu Jiaye Li Kaihui Cheng Yuxuan Chen Tan Chen Mao Ye Jingdong Wang et\u00a0al. 2024. OpenHumanVid: A Large-Scale High-Quality Dataset for Enhancing Human-Centric Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.00115 (2024)."},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00349"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02092"},{"key":"e_1_3_3_2_48_1","unstructured":"Yuchen Liu et\u00a0al. 2025. Kaleido: Open-Sourced Multi-Subject Reference Video Generation Model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.18573 (2025)."},{"key":"e_1_3_3_2_49_1","unstructured":"I Loshchilov. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1711.05101 (2017)."},{"key":"e_1_3_3_2_50_1","unstructured":"Luma AI. 2025. Ray3: AI Video Generation with HDR Motion & Visual Reasoning. https:\/\/lumalabs.ai\/ray. Accessed: 2025-11-06."},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3757377.3763841"},{"key":"e_1_3_3_2_52_1","unstructured":"Wan-Duo\u00a0Kurt Ma John\u00a0P Lewis and W\u00a0Bastiaan Kleijn. 2023. TrailBlazer: Trajectory Control for Diffusion-Based Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.00896 (2023)."},{"key":"e_1_3_3_2_53_1","unstructured":"Jinjie Mai Chaoyang Wang Guocheng\u00a0Gordon Qian Willi Menapace Sergey Tulyakov Bernard Ghanem Peter Wonka and Ashkan Mirzaei. 2025. EasyV2V: A High-quality Instruction-based Video Editing Framework. arxiv:https:\/\/arXiv.org\/abs\/2512.16920\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2512.16920"},{"key":"e_1_3_3_2_54_1","unstructured":"Koichi Namekata Sherwin Bahmani Ziyi Wu Yash Kant Igor Gilitschenski and David\u00a0B Lindell. 2024. SG-I2V: Self-Guided Trajectory Control in Image-to-Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.04989 (2024)."},{"key":"e_1_3_3_2_55_1","unstructured":"Kepan Nan Rui Xie Penghao Zhou Tiehan Fan Zhenheng Yang Zhijie Chen Xiang Li Jian Yang and Ying Tai. 2024. OpenVid-1M: A Large-Scale High-Quality Dataset for Text-to-video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.02371 (2024)."},{"key":"e_1_3_3_2_56_1","unstructured":"Tuan\u00a0Duc Ngo Peiye Zhuang Chuang Gan Evangelos Kalogerakis Sergey Tulyakov Hsin-Ying Lee and Chaoyang Wang. 2024. DELTA: Dense Efficient Long-range 3D Tracking for Any video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.24211 (2024)."},{"key":"e_1_3_3_2_57_1","unstructured":"NVIDIA. 2025. World Simulation with Video Foundation Models for Physical AI. arxiv:https:\/\/arXiv.org\/abs\/2511.00062\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2511.00062"},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_2_59_1","unstructured":"Jordi Pont-Tuset Federico Perazzi Sergi Caelles Pablo Arbel\u00e1ez Alex Sorkine-Hornung and Luc\u00a0Van Gool. 2018. The 2017 DAVIS Challenge on Video Object Segmentation. arxiv:https:\/\/arXiv.org\/abs\/1704.00675\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1704.00675"},{"key":"e_1_3_3_2_60_1","first-page":"652","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","author":"Qi Charles\u00a0R","year":"2017","unstructured":"Charles\u00a0R Qi, Hao Su, Kaichun Mo, and Leonidas\u00a0J Guibas. 2017. Pointnet: Deep learning on point sets for 3d classification and segmentation. In Proceedings of the IEEE conference on computer vision and pattern recognition. 652\u2013660."},{"key":"e_1_3_3_2_61_1","unstructured":"Haonan Qiu Zhaoxi Chen Zhouxia Wang Yingqing He Menghan Xia and Ziwei Liu. 2024. FreeTraj: Tuning-Free Trajectory Control in Video Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.16863 (2024)."},{"key":"e_1_3_3_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00574"},{"key":"e_1_3_3_2_63_1","unstructured":"Shen Sang et\u00a0al. 2025. Lynx: Towards High-Fidelity Personalized Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.15496 (2025)."},{"key":"e_1_3_3_2_64_1","unstructured":"Joonghyuk Shin Zhengqi Li Richard Zhang Jun-Yan Zhu Jaesik Park Eli Schechtman and Xun Huang. 2025. MotionStream: Real-Time Video Generation with Interactive Motion Controls. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2511.01266 (2025)."},{"key":"e_1_3_3_2_65_1","volume-title":"Neurocomputing","author":"Su Jianlin","year":"2024","unstructured":"Jianlin Su, Murtadha Ahmed, Yu Lu, Shengfeng Pan, Wen Bo, and Yunfeng Liu. 2024. Roformer: Enhanced transformer with rotary position embedding. In Neurocomputing."},{"key":"e_1_3_3_2_66_1","unstructured":"Team Wan. 2025. Wan: Open and advanced large-scale video generative models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.20314 (2025)."},{"key":"e_1_3_3_2_67_1","unstructured":"Angtian Wang Haibin Huang Jacob\u00a0Zhiyuan Fang Yiding Yang and Chongyang Ma. 2025a. ATI: Any Trajectory Instruction for Controllable Video Generation. arxiv:https:\/\/arXiv.org\/abs\/2505.22944\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2505.22944"},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"crossref","unstructured":"Fan-Yun Wang et\u00a0al. 2024a. AnimateLCM: Computation-Efficient Personalized Style Video Generation. ACM Transactions on Graphics (2024).","DOI":"10.1145\/3681758.3698013"},{"key":"e_1_3_3_2_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01165"},{"key":"e_1_3_3_2_70_1","doi-asserted-by":"crossref","unstructured":"Wenshan Wang Delong Zhu Xiangwei Wang Yaoyu Hu Yuheng Qiu Chen Wang Yafei Hu Ashish Kapoor and Sebastian Scherer. 2020. TartanAir: A Dataset to Push the Limits of Visual SLAM. (2020).","DOI":"10.1109\/IROS45743.2020.9341801"},{"key":"e_1_3_3_2_71_1","unstructured":"Yifan Wang Jianjun Zhou Haoyi Zhu Wenzheng Chang Yang Zhou Zizun Li Junyi Chen Jiangmiao Pang Chunhua Shen and Tong He. 2025c. Pi-3: Permutation-Equivariant Visual Geometry Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.13347 (2025)."},{"key":"e_1_3_3_2_72_1","unstructured":"Zhouxia Wang Haonan Qiu Yingqing He Zhaoxi Chen et\u00a0al. 2023. MotionCtrl: A Unified and Flexible Motion Controller for Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.03641 (2023)."},{"key":"e_1_3_3_2_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657518"},{"key":"e_1_3_3_2_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00463"},{"key":"e_1_3_3_2_75_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2415"},{"key":"e_1_3_3_2_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01929"},{"key":"e_1_3_3_2_77_1","unstructured":"Bowen Xue Qixin Yan Wenjing Wang Hao Liu and Chen Li. 2025. Stand-In: A Lightweight and Plug-and-Play Identity Control for Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.07901 (2025)."},{"key":"e_1_3_3_2_78_1","unstructured":"Xitong Yang Devansh Kukreja Don Pinkus Anushka Sagar Taosha Fan Jinhyung Park Soyong Shin Jinkun Cao Jiawei Liu Nicolas Ugrinovic Matt Feiszli Jitendra Malik Piotr Dollar and Kris Kitani. 2026. SAM 3D Body: Robust Full-Body Human Mesh Recovery. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.15989 (2026)."},{"key":"e_1_3_3_2_79_1","unstructured":"Zhuoyi Yang Jiayan Teng Wendi Zheng Ming Ding Shiyu Huang Jiazheng Xu Yuanming Yang Wenyi Hong Xiaohan Zhang Guanyu Feng et\u00a0al. 2024. Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.06072 (2024)."},{"key":"e_1_3_3_2_80_1","unstructured":"Danah Yatim Rafail Fridman Omer Bar-Tal Yoni Kasten and Tali Dekel. 2023. Space-Time Diffusion Features for Zero-Shot Text-Driven Motion Transfer. arXiv preprint arxiv:https:\/\/arXiv.org\/abs\/2311.17009 (2023)."},{"key":"e_1_3_3_2_81_1","unstructured":"Zixuan Ye Xuanhua He Quande Liu Qiulin Wang Xintao Wang Pengfei Wan Di Zhang Kun Gai Qifeng Chen and Wenhan Luo. 2025. UNIC: Unified In-Context Video Editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.04216 (2025)."},{"key":"e_1_3_3_2_82_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00017"},{"key":"e_1_3_3_2_83_1","unstructured":"Zhenghao Zhang Junchao Liao Menghao Li Long Qin and Weizhi Wang. 2024. Tora: Trajectory-oriented Diffusion Transformer for Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21705 (2024)."},{"key":"e_1_3_3_2_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754869"},{"key":"e_1_3_3_2_85_1","unstructured":"Zhiyuan Zhang Can Wang Dongdong Chen and Jing Liao. 2025b. FlexTraj: Image-to-Video Generation with Flexible Point Trajectory Control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.08527 (2025)."},{"key":"e_1_3_3_2_86_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01818"},{"key":"e_1_3_3_2_87_1","volume-title":"NeurIPS D&B","author":"Zi Bojia","year":"2025","unstructured":"Bojia Zi, Penghui Ruan, Marco Chen, Xianbiao Qi, Shaozhe Hao, Shihao Zhao, Youze Huang, Bin Liang, Rong Xiao, and Kam-Fai Wong. 2025. Se\u00f1orita-2M: A High-Quality Instruction-based Dataset for General Video Editing by Video Specialists. In NeurIPS D&B."}],"event":{"name":"SIGGRAPH Conference Papers '26: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers","location":"Los Angeles CA USA","acronym":"SIGGRAPH Conference Papers '26","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers"],"original-title":[],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T18:21:56Z","timestamp":1784226116000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3799902.3811093"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":86,"alternative-id":["10.1145\/3799902.3811093","10.1145\/3799902"],"URL":"https:\/\/doi.org\/10.1145\/3799902.3811093","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}