{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:57:49Z","timestamp":1781539069164,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":91,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"New Generation Artificial Intelligence-National Science and Technology Major Project","award":["2025ZD0123502"],"award-info":[{"award-number":["2025ZD0123502"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810615","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"2438-2447","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Ivy-Fake: A Unified Explainable Framework and Benchmark for Image and Video AIGC Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6813-8669","authenticated-orcid":false,"given":"Changjiang","family":"Jiang","sequence":"first","affiliation":[{"name":"Nanjing University, Nanjing, China and Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3154-9087","authenticated-orcid":false,"given":"Wenhui","family":"Dong","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China and Pi3Lab, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6491-5598","authenticated-orcid":false,"given":"Zhonghao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Ningxia University, Yinchuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6503-4688","authenticated-orcid":false,"given":"Fengchang","family":"Yu","sequence":"additional","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2892-5764","authenticated-orcid":false,"given":"Wei","family":"Peng","sequence":"additional","affiliation":[{"name":"Stanford University, Stanford, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8715-5497","authenticated-orcid":false,"given":"Xinbin","family":"Yuan","sequence":"additional","affiliation":[{"name":"Nankai University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-0900-6834","authenticated-orcid":false,"given":"Yifei","family":"Bi","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology, Atlanta, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9963-4296","authenticated-orcid":false,"given":"Ming","family":"Zhao","sequence":"additional","affiliation":[{"name":"Jilin University, Changchun, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7380-3608","authenticated-orcid":false,"given":"Zian","family":"Zhou","sequence":"additional","affiliation":[{"name":"Zhejiang University, Zhejiang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1725-5766","authenticated-orcid":false,"given":"Chenyang","family":"Si","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2131-1671","authenticated-orcid":false,"given":"Caifeng","family":"Shan","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal et\u00a0al. 2023. Gpt-4 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.08774 (2023)."},{"key":"e_1_3_3_2_3_2","first-page":"460","volume-title":"Chinese Conference on Pattern Recognition and Computer Vision (PRCV)","author":"Bai Jianfa","year":"2024","unstructured":"Jianfa Bai, Man Lin, Gang Cao, and Zijie Lou. 2024. AI-Generated Video Detection via Spatial-Temporal Anomaly Learning. In Chinese Conference on Pattern Recognition and Computer Vision (PRCV). Springer, 460\u2013470."},{"key":"e_1_3_3_2_4_2","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et\u00a0al. 2025. Qwen2.5-vl technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13923 (2025)."},{"key":"e_1_3_3_2_5_2","unstructured":"James Betker Gabriel Goh Li Jing Tim Brooks Jianfeng Wang Linjie Li Long Ouyang Juntang Zhuang Joyce Lee Yufei Guo et\u00a0al. 2023. Improving image generation with better captions. Computer Science 2 3 (2023) 8."},{"key":"e_1_3_3_2_6_2","unstructured":"Rohit Bharadwaj Hanan Gani Muzammal Naseer Fahad\u00a0Shahbaz Khan and Salman Khan. 2024. VANE-Bench: Video Anomaly Evaluation Benchmark for Conversational LMMs. arxiv:https:\/\/arXiv.org\/abs\/2406.10326"},{"key":"e_1_3_3_2_7_2","unstructured":"Simone Bonechi Paolo Andreini and Barbara\u00a0Toniella Corradini. 2025. Who Made This? Fake Detection and Source Attribution with Diffusion Features. arxiv:https:\/\/arXiv.org\/abs\/2510.27602\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2510.27602"},{"key":"e_1_3_3_2_8_2","unstructured":"Andrew Brock Jeff Donahue and Karen Simonyan. 2018. Large scale GAN training for high fidelity natural image synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1809.11096 (2018)."},{"key":"e_1_3_3_2_9_2","unstructured":"Tim Brooks Bill Peebles Connor Holmes Will DePue Yufei Guo Li Jing David Schnurr Joe Taylor Troy Luhman Eric Luhman Clarence Ng Ricky Wang and Aditya Ramesh. 2024. Video generation models as world simulators. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.19707 (2024). https:\/\/openai.com\/research\/video-generation-models-as-world-simulators"},{"key":"e_1_3_3_2_10_2","unstructured":"Bin Cao Jianhao Yuan Yexin Liu et\u00a0al. 2024. Synartifact: Classifying and alleviating artifacts in synthetic images via vision-language model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.18068 (2024)."},{"key":"e_1_3_3_2_11_2","series-title":"(ICML\u201924)","volume-title":"ICML","author":"Chen Baoying","year":"2024","unstructured":"Baoying Chen, Jishen Zeng, Jianquan Yang, and Rui Yang. 2024. DRCT: diffusion reconstruction contrastive training towards universal detection of diffusion generated images. In ICML (Vienna, Austria) (ICML\u201924). JMLR.org, Article 297, 19\u00a0pages."},{"key":"e_1_3_3_2_12_2","unstructured":"Haoxing Chen Yan Hong Zizheng Huang et\u00a0al. 2024. DeMamba: AI-Generated Video Detection on Million-Scale GenVideo Benchmark. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.19707 (2024)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02283"},{"key":"e_1_3_3_2_14_2","volume-title":"arxiv","author":"Contributors LLaVA\u00a0Community","year":"2025","unstructured":"LLaVA\u00a0Community Contributors. 2025. LLaVA-OneVision-1.5: Fully Open Framework for Democratized Multimodal Training. In arxiv."},{"key":"e_1_3_3_2_15_2","unstructured":"Jingyi Deng Chenhao Lin Zhengyu Zhao Shuai Liu Qian Wang and Chao Shen. 2024. A survey of defenses against ai-generated visual media: Detection disruption and authentication. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.10575 (2024)."},{"key":"e_1_3_3_2_16_2","unstructured":"Bo Du Xuekang Zhu Xiaochen Ma et\u00a0al. 2025. Forensichub: A unified benchmark & codebase for all-domain fake image detection and localization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.11003 (2025)."},{"key":"e_1_3_3_2_17_2","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et\u00a0al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12948 (2025)."},{"key":"e_1_3_3_2_18_2","volume-title":"International Conference on Learning Representations","author":"He Pengcheng","year":"2021","unstructured":"Pengcheng He, Xiaodong Liu, Jianfeng Gao, and Weizhu Chen. 2021. DEBERTA: DECODING-ENHANCED BERT WITH DISENTANGLED ATTENTION. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=XPZIaotutsD"},{"key":"e_1_3_3_2_19_2","unstructured":"Amir Hertz Ron Mokady Jay Tenenbaum et\u00a0al. 2022. Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2208.01626 (2022)."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","unstructured":"Yan Hong Jianming Feng Haoxing Chen et\u00a0al. 2025. WildFake: A Large-Scale and Hierarchical Dataset for AI-Generated Images Detection. Proceedings of the AAAI Conference on Artificial Intelligence 39 4 (Apr. 2025) 3500\u20133508. 10.1609\/aaai.v39i4.32363","DOI":"10.1609\/aaai.v39i4.32363"},{"key":"e_1_3_3_2_21_2","unstructured":"Qing Huang Zhipei Xu Xuanyu Zhang and Jian Zhang. 2025. UniShield: An Adaptive Multi-Agent Framework for Unified Forgery Image Detection and Localization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.03161 (2025)."},{"key":"e_1_3_3_2_22_2","unstructured":"Tai-Ming Huang Wei-Tung Lin Kai-Lung Hua et\u00a0al. 2025. ThinkFake: Reasoning in Multimodal Large Language Models for AI-Generated Image Detection. arxiv:https:\/\/arXiv.org\/abs\/2509.19841\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2509.19841"},{"key":"e_1_3_3_2_23_2","unstructured":"Yikun Ji Yan Hong Bowen Deng et\u00a0al. 2025. Zoom-In to Sort AI-Generated Images Out. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.04225 (2025)."},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP55912.2026.11462736"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.169"},{"key":"e_1_3_3_2_26_2","unstructured":"Wan Jiang Jing Yan Ruixuan Zhang et\u00a0al. 2025. Revisiting Reconstruction-based AI-generated Image Detection: A Geometric Perspective. arXiv e-prints (2025) arXiv\u20132510."},{"key":"e_1_3_3_2_27_2","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev Mustafa Suleyman and Andrew Zisserman. 2017. The Kinetics Human Action Video Dataset. arxiv:https:\/\/arXiv.org\/abs\/1705.06950\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/1705.06950"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Mamadou Keita Wassim Hamidouche Hessen Bougueffa\u00a0Eutamene Abdelmalik Taleb-Ahmed David Camacho and Abdenour Hadid. 2025. Bi-LORA: A Vision-Language Approach for Synthetic Image Detection. Expert Systems 42 2 (2025) e13829.","DOI":"10.1111\/exsy.13829"},{"key":"e_1_3_3_2_29_2","unstructured":"Zhenglun Kong Yize Li Fanhu Zeng et\u00a0al. 2025. Token Reduction Should Go Beyond Efficiency in Generative Models\u2013From Vision Language to Multimodality. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.18227 (2025)."},{"key":"e_1_3_3_2_30_2","unstructured":"Black\u00a0Forest Labs Stephen Batifol Andreas Blattmann Frederic Boesel Saksham Consul Cyril Diagne Tim Dockhorn Jack English Zion English Patrick Esser Sumith Kulal Kyle Lacey Yam Levi Cheng Li Dominik Lorenz Jonas M\u00fcller Dustin Podell Robin Rombach Harry Saini Axel Sauer and Luke Smith. 2025. FLUX.1 Kontext: Flow Matching for In-Context Image Generation and Editing in Latent Space. arxiv:https:\/\/arXiv.org\/abs\/2506.15742\u00a0[cs.GR] https:\/\/arxiv.org\/abs\/2506.15742"},{"key":"e_1_3_3_2_31_2","unstructured":"Bo Li Yuanhan Zhang Dong Guo Renrui Zhang Feng Li Hao Zhang Kaichen Zhang Peiyuan Zhang Yanwei Li Ziwei Liu and Chunyuan Li. 2024. LLaVA-OneVision: Easy Visual Task Transfer. Transactions on Machine Learning Research (2024)."},{"key":"e_1_3_3_2_32_2","unstructured":"Ziqiang Li Jiazhen Yan Ziwen He Kai Zeng Weiwei Jiang Lizhi Xiong and Zhangjie Fu. 2025. Is artificial intelligence generated image detection a solved problem?arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.12335 (2025)."},{"key":"e_1_3_3_2_33_2","first-page":"74","volume-title":"Text summarization branches out","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74\u201381."},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754982"},{"key":"e_1_3_3_2_35_2","unstructured":"Jiaxin Liu Jia Wang Saihui Hou et\u00a0al. 2025. Beyond Face Swapping: A Diffusion-Based Digital Human Benchmark for Multimodal Deepfake Detection. arxiv:https:\/\/arXiv.org\/abs\/2505.16512\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2505.16512"},{"key":"e_1_3_3_2_36_2","unstructured":"Ruiqi Liu Manni Cui Ziheng Qin et\u00a0al. 2026. MIRROR: Manifold Ideal Reference ReconstructOR for Generalizable AI-Generated Image Detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.02222 (2026)."},{"key":"e_1_3_3_2_37_2","volume-title":"ICLR","author":"Liu Xuannan","year":"2025","unstructured":"Xuannan Liu, Zekun Li, Peipei Li, Huaibo Huang, Shuhan Xia, Xing Cui, Linzhi Huang, Weihong Deng, and Zhaofeng He. 2025. MMFakeBench: A Mixed-Source Multimodal Misinformation Detection Benchmark for LVLMs. In ICLR."},{"key":"e_1_3_3_2_38_2","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Fixing Weight Decay Regularization in Adam. ArXiv abs\/1711.05101 (2017). https:\/\/api.semanticscholar.org\/CorpusID:3312944"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"Xiaochen Ma Xuekang Zhu Lei Su et\u00a0al. 2025. Imdl-benco: A comprehensive benchmark and codebase for image manipulation detection & localization. Advances in Neural Information Processing Systems 37 (2025) 134591\u2013134613.","DOI":"10.52202\/079017-4277"},{"key":"e_1_3_3_2_40_2","unstructured":"Alex Nichol Prafulla Dhariwal Aditya Ramesh Pranav Shyam Pamela Mishkin Bob McGrew Ilya Sutskever and Mark Chen. 2021. Glide: Towards photorealistic image generation and editing with text-guided diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2112.10741 (2021)."},{"key":"e_1_3_3_2_41_2","unstructured":"Kyoungjun Park Yifan Yang Juheon Yi et\u00a0al. 2025. Vidguard-r1: Ai-generated video detection and explanation via reasoning mllms and rl. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2510.02282 (2025)."},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58610-2_6"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00575"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i1.32051"},{"key":"e_1_3_3_2_45_2","volume-title":"The Fourteenth International Conference on Learning Representations","author":"Qu Chenfan","year":"2026","unstructured":"Chenfan Qu, Yiwu Zhong, Fengjun Guo, and Lianwen Jin. 2026. Omni-IML: Towards Unified Interpretable Image Manipulation Localization. In The Fourteenth International Conference on Learning Representations."},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01025"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v40i10.37814"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"crossref","unstructured":"Shavez\u00a0Mushtaq Qureshi Atif Saeed Sultan\u00a0H. Almotiri Farooq Ahmad and Mohammed A.\u00a0Al Ghamdi. 2024. Deepfake forensics: a survey of digital forensic methods for multimodal deepfake identification on social media. PeerJ Computer Science 10 (2024). https:\/\/api.semanticscholar.org\/CorpusID:270088699","DOI":"10.7717\/peerj-cs.2037"},{"key":"e_1_3_3_2_49_2","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, et\u00a0al. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:231591445"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","unstructured":"Olga Russakovsky Jia Deng Hao Su Jonathan Krause Sanjeev Satheesh Sean Ma Zhiheng Huang Andrej Karpathy Aditya Khosla Michael Bernstein Alexander\u00a0C. Berg and Li Fei-Fei. 2015. ImageNet Large Scale Visual Recognition Challenge. International Journal of Computer Vision (IJCV) 115 3 (2015) 211\u2013252. 10.1007\/s11263-015-0816-y","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"crossref","unstructured":"Chitwan Saharia William Chan Saurabh Saxena Lala Li Jay Whang Emily\u00a0L Denton Kamyar Ghasemipour Raphael Gontijo\u00a0Lopes Burcu Karagol\u00a0Ayan Tim Salimans et\u00a0al. 2022. Photorealistic text-to-image diffusion models with deep language understanding. Advances in neural information processing systems 35 (2022) 36479\u201336494.","DOI":"10.52202\/068431-2643"},{"key":"e_1_3_3_2_53_2","unstructured":"Zhihong Shao Peiyi Wang Qihao Zhu Runxin Xu Junxiao Song Xiao Bi Haowei Zhang Mingchuan Zhang Y.\u00a0K. Li Y. Wu and Daya Guo. 2024. DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. arxiv:https:\/\/arXiv.org\/abs\/2402.03300\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2402.03300"},{"key":"e_1_3_3_2_54_2","unstructured":"Chao Shuai Zhenguang Liu Shaojing Fan et\u00a0al. 2026. When Detectors Forget Forensics: Blocking Semantic Shortcuts for Generalizable AI-Generated Image Detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2603.09242 (2026)."},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295110"},{"key":"e_1_3_3_2_56_2","unstructured":"Hao Tan Jun Lan Senyuan Shi et\u00a0al. 2026. VideoVeritas: AI-Generated Video Detection via Perception Pretext Reinforcement Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.08828 (2026)."},{"key":"e_1_3_3_2_57_2","volume-title":"International Conference on Learning Representations","author":"Tan Hao","year":"2026","unstructured":"Hao Tan, Jun Lan, Zichang Tan, et\u00a0al. 2026. Veritas: Generalizable Deepfake Detection via Pattern-Aware Reasoning. In International Conference on Learning Representations."},{"key":"e_1_3_3_2_58_2","unstructured":"Core Team Zihao Yue Zhenru Lin et\u00a0al. 2025. MiMo-VL Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2506.03569\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2506.03569"},{"key":"e_1_3_3_2_59_2","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew\u00a0M Dai Anja Hauth Katie Millican et\u00a0al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.11805 (2023)."},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00872"},{"key":"e_1_3_3_2_61_2","unstructured":"Youqi Wang Shen Chen Haowei Wang et\u00a0al. 2026. ForgeryVCR: Visual-Centric Reasoning via Efficient Forensic Tools in MLLMs for Image Forgery Detection and Localization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.14098 (2026)."},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02051"},{"key":"e_1_3_3_2_63_2","unstructured":"Siwei Wen Junyan Ye Peilin Feng Hengrui Kang Zichen Wen Yize Chen Jiang Wu Wenjun Wu Conghui He and Weijia Li. 2025. Spot the fake: Large multimodal model-based synthetic image detection with artifact explanation. NeurIPS (2025)."},{"key":"e_1_3_3_2_64_2","unstructured":"Chenfei Wu Jiahao Li Jingren Zhou et\u00a0al. 2025. Qwen-Image Technical Report. arxiv:https:\/\/arXiv.org\/abs\/2508.02324\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2508.02324"},{"key":"e_1_3_3_2_65_2","unstructured":"Lei Xin Caiyun Huang Hao Li et\u00a0al. 2024. Artificial intelligence for central dogma-centric multi-omics: Challenges and breakthroughs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.12668 (2024)."},{"key":"e_1_3_3_2_66_2","first-page":"302","volume-title":"AAAI Bridge Program on AI for Medicine and Healthcare","author":"Xin Lei","year":"2026","unstructured":"Lei Xin, Zhenglun Kong, Fukang Chen, et\u00a0al. 2026. DualCPT: Dual-branch Modeling for Cellular Phenotype Transition. In AAAI Bridge Program on AI for Medicine and Healthcare. PMLR, 302\u2013312."},{"key":"e_1_3_3_2_67_2","unstructured":"Lei Xin Yuhao Zheng Ke Cheng Changjiang Jiang Zifan Zhang and Fanhu Zeng. 2026. HyTRec: A Hybrid Temporal-Aware Attention Architecture for Long Behavior Sequential Recommendation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.18283 (2026). https:\/\/arxiv.org\/abs\/2602.18283"},{"key":"e_1_3_3_2_68_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_3_2_69_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Yan Shilin","year":"2025","unstructured":"Shilin Yan, Ouxiang Li, Jiayin Cai, et\u00a0al. 2025. A Sanity Check for AI-generated Image Detection. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=ODRHZrkOQM"},{"key":"e_1_3_3_2_70_2","volume-title":"ICML","author":"Yan Zhiyuan","year":"2024","unstructured":"Zhiyuan Yan, Jiangming Wang, Zhendong Wang, Peng Jin, Ke-Yue Zhang, Shen Chen, Taiping Yao, Shouhong Ding, Baoyuan Wu, and Li Yuan. 2024. Effort: Efficient Orthogonal Modeling for Generalizable AI-Generated Image Detection. In ICML."},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"crossref","unstructured":"Yuan Yao Tianyu Yu Ao Zhang et\u00a0al. 2025. MiniCPM-V: A GPT-4V Level MLLM on Your Phone. Nat Commun 16 5509 (2025) (2025).","DOI":"10.1038\/s41467-025-61040-5"},{"key":"e_1_3_3_2_72_2","unstructured":"Junyan Ye Baichuan Zhou Zilong Huang et\u00a0al. 2025. LOKI: A Comprehensive Synthetic Data Detection Benchmark using Large Multimodal Models. ICLR (2025)."},{"key":"e_1_3_3_2_73_2","unstructured":"Dehao Ying Fengchang Yu Haihua Chen Changjiang Jiang Yurong Li and Wei Lu. 2026. Beyond Human Annotation: Recent Advances in Data Generation Methods for Document Intelligence. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2601.12318 (2026)."},{"key":"e_1_3_3_2_74_2","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681609"},{"key":"e_1_3_3_2_75_2","doi-asserted-by":"publisher","DOI":"10.1145\/3677389.3702534"},{"key":"e_1_3_3_2_76_2","unstructured":"Tianyu Yu Zefan Wang Chongyi Wang et\u00a0al. 2025. MiniCPM-V 4.5: Cooking Efficient MLLMs via Architecture Data and Training Recipe. arxiv:https:\/\/arXiv.org\/abs\/2509.18154\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2509.18154"},{"key":"e_1_3_3_2_77_2","unstructured":"Yangxin Yu Yue Zhou Bin Li et\u00a0al. 2026. AgentFoX: LLM Agent-Guided Fusion with eXplainability for AI-Generated Image Detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2603.23115 (2026)."},{"key":"e_1_3_3_2_78_2","volume-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems","author":"Yuan Xinbin","year":"2025","unstructured":"Xinbin Yuan, Jian Zhang, Kaixin Li, et\u00a0al. 2025. SE-GUI: Enhancing Visual Grounding for GUI Agents via Self-Evolutionary Reinforcement Learning. In The Thirty-ninth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_3_2_79_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00697"},{"key":"e_1_3_3_2_80_2","volume-title":"International Conference on Learning Representations","author":"Zhang Tianyi","year":"2020","unstructured":"Tianyi Zhang, Varsha Kishore, Felix Wu, Kilian\u00a0Q. Weinberger, and Yoav Artzi. 2020. BERTScore: Evaluating Text Generation with BERT. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SkeHuCVFDr"},{"key":"e_1_3_3_2_81_2","unstructured":"Yonggang Zhang Jun Nie Xinmei Tian et\u00a0al. 2025. Detecting Generated Images by Fitting Natural Image Distributions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2511.01293 (2025)."},{"key":"e_1_3_3_2_82_2","volume-title":"The Fourteenth International Conference on Learning Representations","author":"Zhao Ming","year":"2026","unstructured":"Ming Zhao, Wenhui Dong, Yang Zhang, et\u00a0al. 2026. SpineBench: A Clinically Salient, Level-Aware Benchmark Powered by the SpineMed-450k Corpus. In The Fourteenth International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/2601.12318"},{"key":"e_1_3_3_2_83_2","unstructured":"Qiannian Zhao Chen Yang Jinhao Jing et\u00a0al. 2026. Know What You Know: Metacognitive Entropy Calibration for Verifiable RL Reasoning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2602.22751 (2026)."},{"key":"e_1_3_3_2_84_2","doi-asserted-by":"crossref","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric Xing et\u00a0al. 2023. Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems 36 (2023) 46595\u201346623.","DOI":"10.52202\/075280-2020"},{"key":"e_1_3_3_2_85_2","unstructured":"Nan Zhong Yiran Xu Sheng Li Zhenxing Qian and Xinpeng Zhang. 2023. Patchcraft: Exploring texture patch for efficient ai-generated image detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.12397 (2023)."},{"key":"e_1_3_3_2_86_2","unstructured":"Jiang Zhou Xiaohu Zhao Xinwei Wu Tianyu Dong Hao Wang Yangyang Liu Heng Liu Linlong Xu Longyue Wang Weihua Luo and Deyi Xiong. 2026. Incentivizing Parametric Knowledge via Reinforcement Learning with Verifiable Rewards for Cross-Cultural Entity Translation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2604.16881 (2026)."},{"key":"e_1_3_3_2_87_2","unstructured":"Yue Zhou Xinan He KaiQing Lin et\u00a0al. 2025. Breaking latent prior bias in detectors for generalizable aigc image detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2506.00874 (2025)."},{"key":"e_1_3_3_2_88_2","doi-asserted-by":"crossref","unstructured":"Ziyin Zhou Yunpeng Luo Yuanchen Wu et\u00a0al. 2025. AIGI-Holmes: Towards Explainable and Generalizable AI-Generated Image Detection via Multimodal Large Language Models. ArXiv abs\/2507.02664 (2025). https:\/\/api.semanticscholar.org\/CorpusID:280141523","DOI":"10.1109\/ICCV51701.2025.01742"},{"key":"e_1_3_3_2_89_2","unstructured":"Mingjian Zhu Hanting Chen Mouxiao Huang Wei Li Hailin Hu Jie Hu and Yunhe Wang. 2023. Gendet: Towards good generalizations for ai-generated image detection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.08880 (2023)."},{"key":"e_1_3_3_2_90_2","series-title":"(NIPS \u201923)","volume-title":"NeurIPS","author":"Zhu Mingjian","year":"2023","unstructured":"Mingjian Zhu, Hanting Chen, Qiangyu Yan, Xudong Huang, Guanyu Lin, Wei Li, Zhijun Tu, Hailin Hu, Jie Hu, and Yunhe Wang. 2023. GenImage: a million-scale benchmark for detecting AI-generated image. In NeurIPS (New Orleans, LA, USA) (NIPS \u201923). Curran Associates Inc., Red Hook, NY, USA, Article 3398, 12\u00a0pages."},{"key":"e_1_3_3_2_91_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i10.33198"},{"key":"e_1_3_3_2_92_2","unstructured":"Xuekang Zhu Ji-Zhe Zhou Kaiwen Feng et\u00a0al. 2025. Does the Manipulation Process Matter? RITA: Reasoning Composite Image Manipulations via Reversely-Ordered Incremental-Transition Autoregression. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.20006 (2025)."}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:47:45Z","timestamp":1781538465000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810615"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":91,"alternative-id":["10.1145\/3805622.3810615","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810615","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}