{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,22]],"date-time":"2026-01-22T08:23:15Z","timestamp":1769070195860,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","funder":[{"name":"Beijing Municipal Science and Technology Project","award":["Z241100001324002"],"award-info":[{"award-number":["Z241100001324002"]}]},{"DOI":"10.13039\/501100005090","name":"Beijing Nova Program","doi-asserted-by":"publisher","award":["20240484681"],"award-info":[{"award-number":["20240484681"]}],"id":[{"id":"10.13039\/501100005090","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3761987","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:51Z","timestamp":1761377211000},"page":"13737-13742","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Identity-Preserving Video Generation Challenge"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1940-6137","authenticated-orcid":false,"given":"Yiheng","family":"Zhang","sequence":"first","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7485-9198","authenticated-orcid":false,"given":"Zhaofan","family":"Qiu","sequence":"additional","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4606-8926","authenticated-orcid":false,"given":"Qi","family":"Cai","sequence":"additional","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9603-1113","authenticated-orcid":false,"given":"Yehao","family":"Li","sequence":"additional","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0818-0985","authenticated-orcid":false,"given":"Fuchen","family":"Long","sequence":"additional","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4344-8898","authenticated-orcid":false,"given":"Yingwei","family":"Pan","sequence":"additional","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7587-101X","authenticated-orcid":false,"given":"Ting","family":"Yao","sequence":"additional","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5990-7307","authenticated-orcid":false,"given":"Tao","family":"Mei","sequence":"additional","affiliation":[{"name":"HiDream.ai Inc., Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space. arXiv e-prints","author":"Batifol Stephen","year":"2025","unstructured":"Stephen Batifol, Andreas Blattmann, Frederic Boesel, et al., 2025. FLUX. 1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space. arXiv e-prints (2025), arXiv-2506."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Qi Cai Jingwen Chen Yang Chen Yehao Li et al. 2025. HiDream-I1: A High-Efficient Image Generative Foundation Model with Sparse Diffusion Transformer. arXiv preprint arXiv:2505.22705 (2025).","DOI":"10.1145\/3746027.3756870"},{"key":"e_1_3_2_1_3_1","unstructured":"Brandon Castellano. 2025. PySceneDetect: Video Cut Detection and Analysis Tool. https:\/\/github.com\/Breakthrough\/PySceneDetect"},{"key":"e_1_3_2_1_4_1","volume-title":"Controlstyle: Text-driven stylized image generation using diffusion priors. In ACM Multimedia.","author":"Chen Jingwen","year":"2023","unstructured":"Jingwen Chen, Yingwei Pan, Ting Yao, and Tao Mei. 2023. Controlstyle: Text-driven stylized image generation using diffusion priors. In ACM Multimedia."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"J. S. Chung A. Nagrani and A. Zisserman. 2018. VoxCeleb2: Deep Speaker Recognition. In INTERSPEECH.","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"e_1_3_2_1_6_1","volume-title":"Arcface: Additive angular margin loss for deep face recognition. In CVPR.","author":"Deng Jiankang","year":"2019","unstructured":"Jiankang Deng, Jia Guo, Niannan Xue, and Stefanos Zafeiriou. 2019. Arcface: Additive angular margin loss for deep face recognition. In CVPR."},{"key":"e_1_3_2_1_7_1","volume-title":"Zhaofan Qiu and Tao Mei","author":"Fuchen Long Ting Yao","year":"2024","unstructured":"Ting Yao Fuchen Long, Zhaofan Qiu and Tao Mei. 2024. VideoStudio: Generating Consistent-Content and Multi-Scene Videos. In ECCV."},{"key":"e_1_3_2_1_8_1","unstructured":"Rinon Gal Yuval Alaluf Yuval Atzmon Or Patashnik Amit H. Bermano Gal Chechik and Daniel Cohen-Or. 2023. An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion. In ICLR."},{"key":"e_1_3_2_1_9_1","unstructured":"Zinan Guo Yanze Wu Zhuowei Chen Lang Chen Peng Zhang and Qian He. 2024. PuLID: Pure and Lightning ID Customization via Contrastive Alignment. In NeurIPS."},{"key":"e_1_3_2_1_10_1","volume-title":"COVER: A Comprehensive Video Quality Evaluator. In CVPR Workshops.","author":"He Chenlong","year":"2024","unstructured":"Chenlong He, Qi Zheng, Ruoxi Zhu, Xiaoyang Zeng, Yibo Fan, and Zhengzhong Tu. 2024b. COVER: A Comprehensive Video Quality Evaluator. In CVPR Workshops."},{"key":"e_1_3_2_1_11_1","volume-title":"Id-animator: Zero-shot identity-preserving human video generation. arXiv preprint arXiv:2404.15275","author":"He Xuanhua","year":"2024","unstructured":"Xuanhua He, Quande Liu, Shengju Qian, Xin Wang, Tao Hu, Ke Cao, Keyu Yan, and Jie Zhang. 2024a. Id-animator: Zero-shot identity-preserving human video generation. arXiv preprint arXiv:2404.15275 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"Ronan Le Bras, and Yejin Choi","author":"Hessel Jack","year":"2021","unstructured":"Jack Hessel, Ari Holtzman, Maxwell Forbes, Ronan Le Bras, and Yejin Choi. 2021. Clipscore: A reference-free evaluation metric for image captioning. arXiv preprint arXiv:2104.08718 (2021)."},{"key":"e_1_3_2_1_13_1","unstructured":"Martin Heusel Hubert Ramsauer Thomas Unterthiner Bernhard Nessler and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. In NeurIPS."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Yuge Huang Yuhan Wang Ying Tai Xiaoming Liu Pengcheng Shen Shaoxin Li and Feiyue Huang Jilin Li. 2020. CurricularFace: Adaptive Curriculum Learning Loss for Deep Face Recognition. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00594"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Ziqi Huang Yinan He Jiashuo Yu Fan Zhang Chenyang Si Yuming Jiang Yuanhan Zhang Tianxing Wu Qingyang Jin Nattapol Chanpaisit Yaohui Wang Xinyuan Chen Limin Wang Dahua Lin Yu Qiao and Ziwei Liu. 2024. VBench: Comprehensive Benchmark Suite for Video Generative Models. In CVPR.","DOI":"10.1109\/CVPR52733.2024.02060"},{"key":"e_1_3_2_1_16_1","unstructured":"Zhen Li Mingdeng Cao Xintao Wang Zhongang Qi Ming-Ming Cheng and Ying Shan. 2024. PhotoMaker: Customizing Realistic Human Photos via Stacked ID Embedding. In CVPR."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"A. Nagrani J. S. Chung and A. Zisserman. 2017. VoxCeleb: a large-scale speaker identification dataset. In INTERSPEECH.","DOI":"10.21437\/Interspeech.2017-950"},{"key":"e_1_3_2_1_18_1","unstructured":"Yingwei Pan Zhaofan Qiu Ting Yao Houqiang Li and Tao Mei. 2017. To create what you tell: Generating videos from captions. In ACM Multimedia."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2022. High-resolution image synthesis with latent diffusion models. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Andreas R\u00f6ssler Davide Cozzolino Luisa Verdoliva Christian Riess Justus Thies and Matthias Nie\u00dfner. 2019. FaceForensics: Learning to Detect Manipulated Facial Images. In ICCV.","DOI":"10.1109\/ICCV.2019.00009"},{"key":"e_1_3_2_1_21_1","volume-title":"DreamBooth: Fine Tuning Text-to-image Diffusion Models for Subject-Driven Generation. arXiv preprint arxiv:2208.12242","author":"Ruiz Nataniel","year":"2022","unstructured":"Nataniel Ruiz, Yuanzhen Li, Varun Jampani, Yael Pritch, Michael Rubinstein, and Kfir Aberman. 2022. DreamBooth: Fine Tuning Text-to-image Diffusion Models for Subject-Driven Generation. arXiv preprint arxiv:2208.12242 (2022)."},{"key":"e_1_3_2_1_22_1","volume-title":"Instantbooth: Personalized text-to-image generation without test-time finetuning. In CVPR.","author":"Shi Jing","year":"2024","unstructured":"Jing Shi, Wei Xiong, Zhe Lin, and Hyun Joon Jung. 2024. Instantbooth: Personalized text-to-image generation without test-time finetuning. In CVPR."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Vincent Vanhoucke Sergey Ioffe Jon Shlens and Zbigniew Wojna. 2016. Rethinking the inception architecture for computer vision. In CVPR.","DOI":"10.1109\/CVPR.2016.308"},{"key":"e_1_3_2_1_24_1","volume-title":"MEAD: A Large-scale Audio-visual Dataset for Emotional Talking-face Generation. In ECCV.","author":"Wang Kaisiyuan","year":"2020","unstructured":"Kaisiyuan Wang, Qianyi Wu, Linsen Song, Zhuoqian Yang, Wayne Wu, Chen Qian, Ran He, Yu Qiao, and Chen Change Loy. 2020. MEAD: A Large-scale Audio-visual Dataset for Emotional Talking-face Generation. In ECCV."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Yujie Wei Shiwei Zhang Zhiwu Qing Hangjie Yuan Zhiheng Liu Yu Liu Yingya Zhang Jingren Zhou and Hongming Shan. 2024. DreamVideo: Composing Your Dream Videos with Customized Subject and Motion. In CVPR.","DOI":"10.1109\/CVPR52733.2024.00625"},{"key":"e_1_3_2_1_26_1","unstructured":"Ting Yao Yehao Li Yingwei Pan Zhaofan Qiu and Tao Mei. 2025. Denoising token prediction in masked autoregressive models. In ICCV."},{"key":"e_1_3_2_1_27_1","unstructured":"Yuan Yao Tianyu Yu Ao Zhang Chongyi Wang Junbo Cui Hongji Zhu Tianchi Cai Haoyu Li Weilin Zhao Zhihui He et al. 2024. MiniCPM-V: A GPT-4V Level MLLM on Your Phone. arXiv preprint arXiv:2408.01800 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. arXiv preprint arxiv:2308.06721","author":"Ye Hu","year":"2023","unstructured":"Hu Ye, Jun Zhang, Sibo Liu, Xiao Han, and Wei Yang. 2023. IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. arXiv preprint arxiv:2308.06721 (2023)."},{"key":"e_1_3_2_1_29_1","volume-title":"Weidong Cai, and Wayne Wu.","author":"Yu Jianhui","year":"2023","unstructured":"Jianhui Yu, Hao Zhu, Liming Jiang, Chen Change Loy, Weidong Cai, and Wayne Wu. 2023. CelebV-Text: A Large-Scale Facial Text-Video Dataset. In CVPR."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Shenghai Yuan Jinfa Huang Xianyi He Yunyang Ge Yujun Shi Liuhan Chen Jiebo Luo and Li Yuan. 2025. Identity-preserving text-to-video generation by frequency decomposition. In CVPR.","DOI":"10.32388\/TZIID6"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Hao Zhu Wayne Wu Wentao Zhu Liming Jiang Siwei Tang Li Zhang Ziwei Liu and Chen Change Loy. 2022. CelebV-HQ: A Large-Scale Video Facial Attributes Dataset. In ECCV.","DOI":"10.1007\/978-3-031-20071-7_38"},{"key":"e_1_3_2_1_32_1","volume-title":"Sd-dit: Unleashing the power of self-supervised discrimination in diffusion transformer. In CVPR.","author":"Zhu Rui","year":"2024","unstructured":"Rui Zhu, Yingwei Pan, Yehao Li, Ting Yao, Zhenglong Sun, Tao Mei, and Chang Wen Chen. 2024. Sd-dit: Unleashing the power of self-supervised discrimination in diffusion transformer. In CVPR."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3761987","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:06:49Z","timestamp":1765339609000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3761987"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":32,"alternative-id":["10.1145\/3746027.3761987","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3761987","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}