{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:59:33Z","timestamp":1776931173150,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763896","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:30:41Z","timestamp":1765211441000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Cut2Next: Generating Next Shot via In-Context Tuning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9021-1167","authenticated-orcid":false,"given":"Jingwen","family":"He","sequence":"first","affiliation":[{"name":"Chinese University of Hong Kong, Hongkong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0753-1652","authenticated-orcid":false,"given":"Hongbo","family":"Liu","sequence":"additional","affiliation":[{"name":"Shanghai Artificial Intelligence Laboratory, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1785-8133","authenticated-orcid":false,"given":"Jiajun","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong University of Science and Technology, Hongkong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8008-5873","authenticated-orcid":false,"given":"Ziqi","family":"Huang","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1889-2567","authenticated-orcid":false,"given":"Qiao","family":"Yu","sequence":"additional","affiliation":[{"name":"Shanghai Artificial Intelligence Laboratory, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9163-2761","authenticated-orcid":false,"given":"Wanli","family":"Ouyang","sequence":"additional","affiliation":[{"name":"Chinese University of Hong Kong, Hongkong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4220-5958","authenticated-orcid":false,"given":"Ziwei","family":"Liu","sequence":"additional","affiliation":[{"name":"Nanyang Technological University (NTU), Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_1_2_1","unstructured":"2024. Kling. Accessed December 9 2024 [Online] https:\/\/klingai.kuaishou.com\/. https:\/\/klingai.kuaishou.com\/"},{"key":"e_1_3_3_1_3_1","unstructured":"2024. Sora. Accessed February 15 2024 [Online] https:\/\/sora.com\/library. https:\/\/sora.com\/library"},{"key":"e_1_3_3_1_4_1","unstructured":"Yuval Atzmon Rinon Gal Yoad Tewel Yoni Kasten and Gal Chechik. 2024. Multi-Shot Character Consistency for Text-to-Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.07750 (2024)."},{"key":"e_1_3_3_1_5_1","unstructured":"Jianhong Bai Menghan Xia Xintao Wang Ziyang Yuan Xiao Fu Zuozhu Liu Haoji Hu Pengfei Wan and Di Zhang. 2024. SynCamMaster: Synchronizing Multi-Camera Video Generation from Diverse Viewpoints. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.07760 (2024)."},{"key":"e_1_3_3_1_6_1","unstructured":"Hritik Bansal Yonatan Bitton Michal Yarom Idan Szpektor Aditya Grover and Kai-Wei Chang. 2024. Talc: Time-aligned captions for multi-scene text-to-video generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.04682 (2024)."},{"key":"e_1_3_3_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687614"},{"key":"e_1_3_3_1_8_1","unstructured":"Black Forest Labs. 2024. Flux: Official inference repository for flux.1 models. https:\/\/github.com\/BlackForestLabs\/flux Accessed: 2024-11-12."},{"key":"e_1_3_3_1_9_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts et\u00a0al. 2023. Stable video diffusion: Scaling latent video diffusion models to large datasets. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.15127 (2023)."},{"key":"e_1_3_3_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02062"},{"key":"e_1_3_3_1_11_1","unstructured":"Guibin Chen Dixuan Lin Jiangping Yang Chunze Lin Juncheng Zhu Mingyuan Fan Hao Zhang Sheng Chen Zheng Chen Chengchen Ma et\u00a0al. 2025. Skyreels-v2: Infinite-length film generative model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.13074 (2025)."},{"key":"e_1_3_3_1_12_1","volume-title":"Forty-first international conference on machine learning","author":"Esser Patrick","year":"2024","unstructured":"Patrick Esser, Sumith Kulal, Andreas Blattmann, Rahim Entezari, Jonas M\u00fcller, Harry Saini, Yam Levi, Dominik Lorenz, Axel Sauer, Frederic Boesel, et\u00a0al. 2024. Scaling rectified flow transformers for high-resolution image synthesis. In Forty-first international conference on machine learning."},{"key":"e_1_3_3_1_13_1","unstructured":"FFmpeg Developers. [n. d.]. FFmpeg. https:\/\/ffmpeg.org."},{"key":"e_1_3_3_1_14_1","unstructured":"Rinon Gal Yuval Alaluf Yuval Atzmon Or Patashnik Amit\u00a0H Bermano Gal Chechik and Daniel Cohen-Or. 2022. An image is worth one word: Personalizing text-to-image generation using textual inversion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2208.01618 (2022)."},{"key":"e_1_3_3_1_15_1","first-page":"322","volume-title":"European Conference on Computer Vision","author":"Gal Rinon","year":"2024","unstructured":"Rinon Gal, Or Lichter, Elad Richardson, Or Patashnik, Amit\u00a0H Bermano, Gal Chechik, and Daniel Cohen-Or. 2024. Lcm-lookahead for encoder-based text-to-image personalization. In European Conference on Computer Vision. Springer, 322\u2013340."},{"key":"e_1_3_3_1_16_1","unstructured":"Gemini Google Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew\u00a0M Dai Anja Hauth Katie Millican et\u00a0al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.11805 (2023)."},{"key":"e_1_3_3_1_17_1","unstructured":"Yuwei Guo Ceyuan Yang Ziyan Yang Zhibei Ma Zhijie Lin Zhenheng Yang Dahua Lin and Lu Jiang. 2025. Long Context Tuning for Video Generation. arxiv:https:\/\/arXiv.org\/abs\/2503.10589\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2503.10589"},{"key":"e_1_3_3_1_18_1","unstructured":"Xuanhua He Quande Liu Shengju Qian Xin Wang Tao Hu Ke Cao Keyu Yan and Jie Zhang. 2024. Id-animator: Zero-shot identity-preserving human video generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.15275 (2024)."},{"key":"e_1_3_3_1_19_1","unstructured":"Edward\u00a0J Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang Weizhu Chen et\u00a0al. 2022. Lora: Low-rank adaptation of large language models. ICLR 1 2 (2022) 3."},{"key":"e_1_3_3_1_20_1","unstructured":"Lianghua Huang Wei Wang Zhi-Fan Wu Yupeng Shi Huanzhang Dou Chen Liang Yutong Feng Yu Liu and Jingren Zhou. 2024. In-context lora for diffusion transformers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.23775 (2024)."},{"key":"e_1_3_3_1_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_41"},{"key":"e_1_3_3_1_22_1","unstructured":"Yuzhou Huang Ziyang Yuan Quande Liu Qiulin Wang Xintao Wang Ruimao Zhang Pengfei Wan Di Zhang and Kun Gai. 2025. ConceptMaster: Multi-Concept Video Customization on Diffusion Transformer Models Without Test-Time Tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.04698 (2025)."},{"key":"e_1_3_3_1_23_1","unstructured":"Jaided AI. 2024. EasyOCR. https:\/\/github.com\/JaidedAI\/EasyOCR."},{"key":"e_1_3_3_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00639"},{"key":"e_1_3_3_1_25_1","doi-asserted-by":"crossref","unstructured":"Ozgur Kara Krishna\u00a0Kumar Singh Feng Liu Duygu Ceylan James\u00a0M Rehg and Tobias Hinz. 2025. ShotAdapter: Text-to-Multi-Shot Video Generation with Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.07652 (2025).","DOI":"10.1109\/CVPR52734.2025.02645"},{"key":"e_1_3_3_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00510"},{"key":"e_1_3_3_1_27_1","unstructured":"Diederik\u00a0P Kingma. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1412.6980 (2014)."},{"key":"e_1_3_3_1_28_1","unstructured":"Weijie Kong Qi Tian Zijian Zhang Rox Min Zuozhuo Dai Jin Zhou Jiangfeng Xiong Xin Li Bo Wu Jianwei Zhang et\u00a0al. 2024. Hunyuanvideo: A systematic framework for large video generative models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.03603 (2024)."},{"key":"e_1_3_3_1_29_1","unstructured":"LAION-AI. 2022. aesthetic-predictor. https:\/\/github.com\/LAION-AI\/aesthetic-predictor."},{"key":"e_1_3_3_1_30_1","unstructured":"Zhong-Yu Li Ruoyi Du Juncheng Yan Le Zhuo Zhen Li Peng Gao Zhanyu Ma and Ming-Ming Cheng. 2025. VisualCloze: A Universal Image Generation Framework via Visual In-Context Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.07960 (2025)."},{"key":"e_1_3_3_1_31_1","unstructured":"Tao Liu Kai Wang Senmao Li Joost van\u00a0de Weijer Fahad\u00a0Shahbaz Khan Shiqi Yang Yaxing Wang Jian Yang and Ming-Ming Cheng. 2025. One-Prompt-One-Story: Free-Lunch Consistent Text-to-Image Generation Using a Single Prompt. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.13554 (2025)."},{"key":"e_1_3_3_1_32_1","first-page":"468","volume-title":"European Conference on Computer Vision","author":"Long Fuchen","year":"2024","unstructured":"Fuchen Long, Zhaofan Qiu, Ting Yao, and Tao Mei. 2024. VideoStudio: Generating Consistent-Content and Multi-Scene Videos. In European Conference on Computer Vision. Springer, 468\u2013485."},{"key":"e_1_3_3_1_33_1","unstructured":"Chaojie Mao Jingfeng Zhang Yulin Pan Zeyinzi Jiang Zhen Han Yu Liu and Jingren Zhou. 2025. Ace++: Instruction-based image creation and editing via context-aware content filling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.02487 (2025)."},{"key":"e_1_3_3_1_34_1","unstructured":"Maxime Oquab Timoth\u00e9e Darcet Th\u00e9o Moutakanni Huy Vo Marc Szafraniec Vasil Khalidov Pierre Fernandez Daniel Haziza Francisco Massa Alaaeldin El-Nouby et\u00a0al. 2023. Dinov2: Learning robust visual features without supervision. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.07193 (2023)."},{"key":"e_1_3_3_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20071-7_39"},{"key":"e_1_3_3_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_1_37_1","unstructured":"Quynh Phung Long Mai Fabian David\u00a0Caba Heilbron Feng Liu Jia-Bin Huang and Cusuh Ham. 2025. CineVerse: Consistent Keyframe Synthesis for Cinematic Scene Composition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.19894 (2025)."},{"key":"e_1_3_3_1_38_1","unstructured":"Tianhao Qi Jianlong Yuan Wanquan Feng Shancheng Fang Jiawei Liu SiYu Zhou Qian He Hongtao Xie and Yongdong Zhang. 2025. Mask2DiT: Dual Mask-based Diffusion Transformer for Multi-Scene Long Video Generation. arxiv:https:\/\/arXiv.org\/abs\/2503.19881\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2503.19881"},{"key":"e_1_3_3_1_39_1","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PmLR, 8748\u20138763."},{"key":"e_1_3_3_1_40_1","unstructured":"Colin Raffel Noam Shazeer Adam Roberts Katherine Lee Sharan Narang Michael Matena Yanqi Zhou Wei Li and Peter\u00a0J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research 21 140 (2020) 1\u201367."},{"key":"e_1_3_3_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_3_1_42_1","unstructured":"Tom\u00e1\u0161 Sou\u010dek and Jakub Loko\u010d. 2020. TransNet V2: An effective deep network architecture for fast shot transition detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2008.04838 (2020)."},{"key":"e_1_3_3_1_43_1","unstructured":"Zhenxiong Tan Songhua Liu Xingyi Yang Qiaochu Xue and Xinchao Wang. 2024. Ominicontrol: Minimal and universal control for diffusion transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.15098 (2024)."},{"key":"e_1_3_3_1_44_1","doi-asserted-by":"crossref","unstructured":"Yoad Tewel Omri Kaduri Rinon Gal Yoni Kasten Lior Wolf Gal Chechik and Yuval Atzmon. 2024. Training-free consistent text-to-image generation. ACM Transactions on Graphics (TOG) 43 4 (2024) 1\u201318.","DOI":"10.1145\/3658157"},{"key":"e_1_3_3_1_45_1","unstructured":"TostAI. 2024. NSFW Image Detection (Large). https:\/\/huggingface.co\/TostAI\/nsfw-image-detection-large."},{"key":"e_1_3_3_1_46_1","unstructured":"Bo Wang Haoyang Huang Zhiyin Lu Fengyuan Liu Guoqing Ma Jianlong Yuan Yuan Zhang and Nan Duan. 2025. STORYANCHORS: Generating Consistent Multi-Scene Story Frames for Long-Form Narratives. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.08350 (2025)."},{"key":"e_1_3_3_1_47_1","first-page":"6537","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Wei Yujie","year":"2024","unstructured":"Yujie Wei, Shiwei Zhang, Zhiwu Qing, Hangjie Yuan, Zhiheng Liu, Yu Liu, Yingya Zhang, Jingren Zhou, and Hongming Shan. 2024. Dreamvideo: Composing your dream videos with customized subject and motion. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6537\u20136549."},{"key":"e_1_3_3_1_48_1","unstructured":"Shaojin Wu Mengqi Huang Wenxu Wu Yufeng Cheng Fei Ding and Qian He. 2025. Less-to-more generalization: Unlocking more controllability by in-context generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.02160 (2025)."},{"key":"e_1_3_3_1_49_1","unstructured":"Junfei Xiao Feng Cheng Lu Qi Liangke Gui Jiepeng Cen Zhibei Ma Alan Yuille and Lu Jiang. 2025. VideoAuteur: Towards Long Narrative Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.06173 (2025)."},{"key":"e_1_3_3_1_50_1","first-page":"399","volume-title":"European Conference on Computer Vision","author":"Xing Jinbo","year":"2024","unstructured":"Jinbo Xing, Menghan Xia, Yong Zhang, Haoxin Chen, Wangbo Yu, Hanyuan Liu, Gongye Liu, Xintao Wang, Ying Shan, and Tien-Tsin Wong. 2024. Dynamicrafter: Animating open-domain images with video diffusion priors. In European Conference on Computer Vision. Springer, 399\u2013417."},{"key":"e_1_3_3_1_51_1","unstructured":"Zhuoyi Yang Jiayan Teng Wendi Zheng Ming Ding Shiyu Huang Jiazheng Xu Yuanming Yang Wenyi Hong Xiaohan Zhang Guanyu Feng et\u00a0al. 2024. Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.06072 (2024)."},{"key":"e_1_3_3_1_52_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. Ip-adapter: Text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.06721 (2023)."},{"key":"e_1_3_3_1_53_1","unstructured":"Shilong Zhang Lianghua Huang Xi Chen Yifei Zhang Zhi-Fan Wu Yutong Feng Wei Wang Yujun Shen Yu Liu and Ping Luo. 2024. Flashface: Human image personalization with high-fidelity identity preservation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.17008 (2024)."},{"key":"e_1_3_3_1_54_1","unstructured":"Shiwei Zhang Jiayu Wang Yingya Zhang Kang Zhao Hangjie Yuan Zhiwu Qin Xiang Wang Deli Zhao and Jingren Zhou. 2023. I2vgen-xl: High-quality image-to-video synthesis via cascaded diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.04145 (2023)."},{"key":"e_1_3_3_1_55_1","unstructured":"Canyu Zhao Mingyu Liu Wen Wang Weihua Chen Fan Wang Hao Chen Bo Zhang and Chunhua Shen. 2024. Moviedreamer: Hierarchical generation for coherent long visual sequence. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.16655 (2024)."},{"key":"e_1_3_3_1_56_1","unstructured":"Mingzhe Zheng Yongqi Xu Haojian Huang Xuran Ma Yexin Liu Wenjie Shu Yatian Pang Feilong Tang Qifeng Chen Harry Yang et\u00a0al. 2024. VideoGen-of-Thought: A Collaborative Framework for Multi-Shot Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.02259 (2024)."},{"key":"e_1_3_3_1_57_1","doi-asserted-by":"crossref","unstructured":"Yupeng Zhou Daquan Zhou Ming-Ming Cheng Jiashi Feng and Qibin Hou. 2024. Storydiffusion: Consistent self-attention for long-range image and video generation. Advances in Neural Information Processing Systems 37 (2024) 110315\u2013110340.","DOI":"10.52202\/079017-3501"}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763896","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:25:55Z","timestamp":1765250755000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763896"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":56,"alternative-id":["10.1145\/3757377.3763896","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763896","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}