{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:57:35Z","timestamp":1781539055068,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810619","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"2428-2437","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Anime-2026: A Large-scale Anime Character Dataset for Anime-related AI Tasks"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-6144-0325","authenticated-orcid":false,"given":"Shijie","family":"Xuyang","sequence":"first","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6495-7894","authenticated-orcid":false,"given":"Bingzhe","family":"Yu","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7720-806X","authenticated-orcid":false,"given":"Minyi","family":"Zhao","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3238-7657","authenticated-orcid":false,"given":"Guangze","family":"Li","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2313-7635","authenticated-orcid":false,"given":"Jihong","family":"Guan","sequence":"additional","affiliation":[{"name":"Tongji University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1949-2768","authenticated-orcid":false,"given":"Shuigeng","family":"Zhou","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia\u00a0Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et\u00a0al. 2023. Gpt-4 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.08774 (2023)."},{"key":"e_1_3_3_2_3_2","unstructured":"Anonymous Danbooru community and Gwern Branwen. 2022. Danbooru2021: A Large-Scale Crowdsourced and Tagged Anime Illustration Dataset. https:\/\/gwern.net\/danbooru2021. https:\/\/gwern.net\/danbooru2021 Accessed: DATE."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Jes\u00fas Armenta-Segura and Grigori Sidorov. 2025. Anime popularity prediction before huge investments: a multimodal approach using deep learning. PeerJ Computer Science 11 (2025) e2715.","DOI":"10.7717\/peerj-cs.2715"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/1101149.1101243"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01267"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Ritendra Datta Dhiraj Joshi Jia Li and James\u00a0Z Wang. 2008. Image retrieval: Ideas influences and trends of the new age. ACM Computing Surveys (Csur) 40 2 (2008) 1\u201360.","DOI":"10.1145\/1348246.1348248"},{"key":"e_1_3_3_2_8_2","unstructured":"Jacob Devlin Ming-Wei Chang Kenton Lee and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1810.04805 (2018)."},{"key":"e_1_3_3_2_9_2","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et\u00a0al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.11929 (2020)."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3011549.3011551"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"crossref","unstructured":"Venkat\u00a0N Gudivada and Vijay\u00a0V Raghavan. 1995. Content based image retrieval systems. Computer 28 9 (1995) 18\u201322.","DOI":"10.1109\/2.410145"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i14.17537"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Steven\u00a0CH Hoi Wei Liu and Shih-Fu Chang. 2010. Semi-supervised distance metric learning for collaborative image retrieval and clustering. ACM Transactions on Multimedia Computing Communications and Applications (TOMM) 6 3 (2010) 1\u201326.","DOI":"10.1145\/1823746.1823752"},{"key":"e_1_3_3_2_15_2","unstructured":"Jing Huo Wenbin Li Yinghuan Shi Yang Gao and Hujun Yin. 2017. Webcaricature: a benchmark for caricature recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1703.03230 (2017)."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/2557642.2563670"},{"key":"e_1_3_3_2_17_2","unstructured":"Yanghua Jin Jiakai Zhang Minjun Li Yingtao Tian Huachun Zhu and Zhihao Fang. 2017. Towards the automatic anime characters creation with generative adversarial networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1708.05509 (2017)."},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"De Li Wenying Xu and Xun Jin. 2024. Anime Audio Retrieval Based on Audio Separation and Feature Recognition. International Journal of Intelligent Systems 2024 1 (2024) 6668582.","DOI":"10.1155\/2024\/6668582"},{"key":"e_1_3_3_2_19_2","first-page":"19730","volume-title":"International conference on machine learning","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In International conference on machine learning. PMLR, 19730\u201319742."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19775-8_33"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Zechao Li Jinhui Tang Liyan Zhang and Jian Yang. 2020. Weakly-supervised semantic guided hashing for social image retrieval. International Journal of Computer Vision 128 (2020) 2265\u20132278.","DOI":"10.1007\/s11263-020-01331-0"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Zhansheng Li Yangyang Xu Nanxuan Zhao Yang Zhou Yongtuo Liu Dahua Lin and Shengfeng He. 2023. Parsing-Conditioned Anime Translation: A New Dataset and Method. ACM Transactions on Graphics 42 3 (2023) 1\u201314.","DOI":"10.1145\/3585002"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2015.7301269"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"crossref","unstructured":"Chao Liu Jingjing Ma Xu Tang Fang Liu Xiangrong Zhang and Licheng Jiao. 2020. Deep hash learning for remote sensing image retrieval. IEEE Transactions on Geoscience and Remote Sensing 59 4 (2020) 3420\u20133443.","DOI":"10.1109\/TGRS.2020.3007533"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58529-7_42"},{"key":"e_1_3_3_2_26_2","unstructured":"Haotian Liu Chunyuan Li Yuheng Li and Yong\u00a0Jae Lee. 2023. Improved baselines with visual instruction tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.03744 (2023)."},{"key":"e_1_3_3_2_27_2","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong\u00a0Jae Lee. 2024. Visual instruction tuning. Advances in neural information processing systems 36 (2024)."},{"key":"e_1_3_3_2_28_2","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1711.05101 (2017)."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"crossref","unstructured":"Yusuke Matsui Kota Ito Yuji Aramaki Azuma Fujimoto Toru Ogawa Toshihiko Yamasaki and Kiyoharu Aizawa. 2017. Sketch-based manga retrieval using manga109 dataset. Multimedia Tools and Applications 76 (2017) 21811\u201321838.","DOI":"10.1007\/s11042-016-4020-z"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2017.291"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-86331-9_27"},{"key":"e_1_3_3_2_32_2","unstructured":"OpenAI. [n. d.]. GPT Image 1 model. https:\/\/platform.openai.com\/docs\/models\/gpt-image-1. Accessed: 2026-02-14."},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2017.178"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3321408.3322624"},{"key":"e_1_3_3_2_35_2","first-page":"8748","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1873987"},{"key":"e_1_3_3_2_37_2","unstructured":"Edwin\u00a0Arkel Rios Wen-Huang Cheng and Bo-Cheng Lai. 2021. DAF: Re: A challenging crowd-sourced large-scale long-tailed dataset for anime character recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2101.08674 (2021)."},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS48785.2022.9937519"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_40_2","unstructured":"Stability AI. 2024. Introducing Stable Diffusion 3.5. https:\/\/stability.ai\/news\/introducing-stable-diffusion-3-5. Accessed: 2026-02-14."},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i3.16364"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"crossref","unstructured":"Ahmed Talib Massudi Mahmuddin Husniza Husni and Loay\u00a0E George. 2013. A weighted dominant color descriptor for content-based image retrieval. Journal of Visual Communication and Image Representation 24 3 (2013) 345\u2013360.","DOI":"10.1016\/j.jvcir.2013.01.007"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"crossref","unstructured":"Jinhui Tang and Zechao Li. 2017. Weakly supervised multimodal hashing for scalable social image retrieval. IEEE Transactions on Circuits and Systems for Video Technology 28 10 (2017) 2730\u20132741.","DOI":"10.1109\/TCSVT.2017.2715227"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5539994"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.541"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00586"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"crossref","unstructured":"Xiu-Shen Wei Jian-Hao Luo Jianxin Wu and Zhi-Hua Zhou. 2017. Selective convolutional descriptor aggregation for fine-grained image retrieval. IEEE Transactions on Image Processing 26 6 (2017) 2868\u20132881.","DOI":"10.1109\/TIP.2017.2688133"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"crossref","unstructured":"Heather\u00a0Lynn Wipfli Minji Kim Julia Vassey and Cassandra Stanton. 2022. Vaping and anime: a growing area of concern. Tobacco Control (2022).","DOI":"10.1136\/tobaccocontrol-2021-057195"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v28i1.8952"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN54540.2023.10191980"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"crossref","unstructured":"Peter Young Alice Lai Micah Hodosh and Julia Hockenmaier. 2014. From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Transactions of the Association for Computational Linguistics 2 (2014) 67\u201378.","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"crossref","unstructured":"Jun Yu Dongquan Liu Dacheng Tao and Hock\u00a0Soon Seah. 2012. On combining multiple features for cartoon character retrieval and clip synthesis. IEEE Transactions on Systems Man and Cybernetics Part B: Cybernetics 42 5 (2012) 1413\u20131427.","DOI":"10.1109\/TSMCB.2012.2192108"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"crossref","unstructured":"Yawen Zeng Yiru Wang Dongliang Liao Gongfu Li Weijie Huang Jin Xu Da Cao and Hong Man. 2022. Keyword-Based Diverse Image Retrieval With Variational Multiple Instance Graph. IEEE Transactions on Neural Networks and Learning Systems (2022).","DOI":"10.1109\/TNNLS.2022.3168431"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"crossref","unstructured":"Xiaohua Zhai Yuxin Peng and Jianguo Xiao. 2014. Modeling Information Retrieval by Formal Logic: A Survey. IEEE Transactions on Circuits and Systems for Video Technology 24 6 (2014) 965\u2013978.","DOI":"10.1109\/TCSVT.2013.2276704"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557678"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"crossref","unstructured":"Yang Zhao Xiaohan Yu Yongsheng Gao and Chunhua Shen. 2022. Learning discriminative region representation for person retrieval. Pattern Recognition 121 (2022) 108229.","DOI":"10.1016\/j.patcog.2021.108229"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413726"},{"key":"e_1_3_3_2_58_2","unstructured":"Chenyang Zhu Xing Zhang Yuyang Sun Ching-Chun Chang and Isao Echizen. 2025. AnimeDL-2M: Million-Scale AI-Generated Anime Image Detection and Localization in Diffusion Era. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.11015 (2025)."},{"key":"e_1_3_3_2_59_2","unstructured":"Deyao Zhu Jun Chen Xiaoqian Shen Xiang Li and Mohamed Elhoseiny. 2023. Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.10592 (2023)."},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"crossref","unstructured":"Lei Zhu Jialie Shen Liang Xie and Zhiyong Cheng. 2016. Unsupervised visual hashing with semantic assistant for content-based image retrieval. IEEE Transactions on Knowledge and Data Engineering 29 2 (2016) 472\u2013486.","DOI":"10.1109\/TKDE.2016.2562624"}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:42:56Z","timestamp":1781538176000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810619"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":59,"alternative-id":["10.1145\/3805622.3810619","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810619","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}