{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:30:15Z","timestamp":1765308615578,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":72,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755041","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:50:47Z","timestamp":1761371447000},"page":"3300-3309","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Character-Centric Understanding of Animated Movies"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-8005-0471","authenticated-orcid":false,"given":"Zhongrui","family":"Gui","sequence":"first","affiliation":[{"name":"VGG, University of Oxford, Oxford, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1123-493X","authenticated-orcid":false,"given":"Junyu","family":"Xie","sequence":"additional","affiliation":[{"name":"VGG, University of Oxford, Oxford, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1874-9664","authenticated-orcid":false,"given":"Tengda","family":"Han","sequence":"additional","affiliation":[{"name":"VGG, University of Oxford, Oxford, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8609-6826","authenticated-orcid":false,"given":"Weidi","family":"Xie","sequence":"additional","affiliation":[{"name":"SAI, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8945-8573","authenticated-orcid":false,"given":"Andrew","family":"Zisserman","sequence":"additional","affiliation":[{"name":"VGG, University of Oxford, Oxford, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58523-5_13"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2020.2987895"},{"key":"e_1_3_2_1_3_1","volume-title":"Sajid Javed, Abdulhadi Shoufan, Yahya Zweiri, and Naoufel Werghi.","author":"Alansari Mohamad","year":"2023","unstructured":"Mohamad Alansari, Oussama Abdul Hay, Sajid Javed, Abdulhadi Shoufan, Yahya Zweiri, and Naoufel Werghi. 2023. GhostFaceNets: Lightweight Face Recognition Model From Cheap Operations. IEEE Access (2023)."},{"key":"e_1_3_2_1_4_1","unstructured":"AudioVault. 2025. AudioVault. https:\/\/audiovault.net\/."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-78"},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the Asian Conference on Computer Vision (ACCV).","author":"Bain Max","year":"2020","unstructured":"Max Bain, Arsha Nagrani, Andrew Brown, and Andrew Zisserman. 2020. Condensed Movies: Story Based Retrieval with Contextual Embeddings. In Proceedings of the Asian Conference on Computer Vision (ACCV)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-105"},{"key":"e_1_3_2_1_8_1","volume-title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs. arXiv preprint arXiv:2406.07476","author":"Cheng Zesen","year":"2024","unstructured":"Zesen Cheng, Sicong Leng, Hang Zhang, Yifei Xin, Xin Li, Guanzheng Chen, Yongxin Zhu, Wenqi Zhang, Ziyang Luo, Deli Zhao, and Lidong Bing. 2024. VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs. arXiv preprint arXiv:2406.07476 (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"LLM-AD: Large Language Model based Audio Description System. arXiv preprint arXiv:2405.00983","author":"Chu Peng","year":"2024","unstructured":"Peng Chu, Jiang Wang, and Andre Abrantes. 2024. LLM-AD: Large Language Model based Audio Description System. arXiv preprint arXiv:2405.00983 (2024)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2337"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3116"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00525"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00482"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDARW.2019.00012"},{"key":"e_1_3_2_1_16_1","unstructured":"Fandom. 2025. Fandom. https:\/\/www.fandom.com\/."},{"key":"e_1_3_2_1_17_1","volume-title":"Chan","author":"Fang Bo","year":"2024","unstructured":"Bo Fang, Wenhao Wu, Qiangqiang Wu, Yuxin Song, and Antoni B. Chan. 2024. DistinctAD: Distinctive Audio Description Generation in Contexts. arXiv preprint arXiv:2411.18180 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2899"},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the 10th International Conference on Communication and Information Processing (ICCIP).","author":"Gong Yue","year":"2024","unstructured":"Yue Gong and Weifeng Zhong. 2024. A Survey of Faical Detection and Recognition methods for Cartoon Characters. In Proceedings of the 10th International Conference on Communication and Information Processing (ICCIP)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01255"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01815"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01720"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10106"},{"key":"e_1_3_2_1_24_1","volume-title":"Quality-Aware End-to-End Audio-Visual Neural Speaker Diarization. arXiv preprint arXiv:2410.22350","author":"He Mao-Kui","year":"2024","unstructured":"Mao-Kui He, Jun Du, Shu-Tong Niu, Qing-Feng Liu, and Chin-Hui Lee. 2024. Quality-Aware End-to-End Audio-Visual Neural Speaker Diarization. arXiv preprint arXiv:2410.22350 (2024)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1022"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_41"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the Asian Conference on Computer Vision (ACCV).","author":"Huh Jaesung","year":"2024","unstructured":"Jaesung Huh and Andrew Zisserman. 2024. Character-aware Audio-visual Subtitling in Context. In Proceedings of the Asian Conference on Computer Vision (ACCV)."},{"key":"e_1_3_2_1_28_1","unstructured":"IMDB. 2025. IMDb: Internet Movie Database. https:\/\/www.imdb.com\/."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/DIIA62678.2024.10871249"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.3006372"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01819"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446480"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383502"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3177952"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV).","author":"Lin Kevin Qinghong","year":"2024","unstructured":"Kevin Qinghong Lin, Pengchuan Zhang, Difei Gao, Xide Xia, Joya Chen, Ziteng Gao, Jinheng Xie, Xuhong Xiao, and Mike Zheng Shou. 2024. Learning Video Context as Interleaved Multimodal Sequences. In Proceedings of the European Conference on Computer Vision (ECCV)."},{"volume-title":"The Llama 3 Herd of Models. arXiv preprint arXiv: 2407.21783","year":"2024","key":"e_1_3_2_1_36_1","unstructured":"Meta. 2024. The Llama 3 Herd of Models. arXiv preprint arXiv: 2407.21783 (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Minderer Matthias","year":"2023","unstructured":"Matthias Minderer, Alexey Gritsenko, and Neil Houlsby. 2023. Scaling Open-Vocabulary Object Detection. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_38_1","volume-title":"Audio-Visual Speaker Diarization: Current Databases, Approaches and Challenges. arXiv preprint arXiv:2409.05659","author":"Mingote Victoria","year":"2024","unstructured":"Victoria Mingote, Alfonso Ortega, Antonio Miguel, and Eduardo Lleida. 2024. Audio-Visual Speaker Diarization: Current Databases, Approaches and Challenges. arXiv preprint arXiv:2409.05659 (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"Jason Sebastian Sulistyawan, and Kelvin Julian","author":"Naftali Martinus Grady","year":"2023","unstructured":"Martinus Grady Naftali, Jason Sebastian Sulistyawan, and Kelvin Julian. 2023. AniWho : A Quick and Accurate Way to Classify Anime Character Faces in Images. arXiv preprint arXiv:2208.11012 (2023)."},{"key":"e_1_3_2_1_40_1","volume-title":"CAST: Character labeling in Animation using Self-supervision by Tracking. Computer Graphics Forum","author":"Nir Oron","year":"2022","unstructured":"Oron Nir, Gal Rapoport, and Ariel Shamir. 2022. CAST: Character labeling in Animation using Self-supervision by Tracking. Computer Graphics Forum (2022)."},{"key":"e_1_3_2_1_41_1","volume-title":"Object Detection for Comics using Manga109 Annotations. arXiv preprint arXiv:1803.08670","author":"Ogawa Toru","year":"2018","unstructured":"Toru Ogawa, Atsushi Otsubo, Rei Narita, Yusuke Matsui, Toshihiko Yamasaki, and Kiyoharu Aizawa. 2018. Object Detection for Comics using Manga109 Annotations. arXiv preprint arXiv:1803.08670 (2018)."},{"key":"e_1_3_2_1_42_1","volume-title":"DINOv2: Learning Robust Visual Features without Supervision. Transactions on Machine Learning Research (TMLR)","author":"Oquab Maxime","year":"2024","unstructured":"Maxime Oquab, Timoth\u00e9e Darcet, Th\u00e9o Moutakanni, Huy V. Vo, Marc Szafraniec, Vasil Khalidov, Pierre Fernandez, Daniel HAZIZA, Francisco Massa, Alaaeldin El-Nouby, Mido Assran, Nicolas Ballas, Wojciech Galuba, Russell Howes, Po-Yao Huang, Shang-Wen Li, Ishan Misra, Michael Rabbat, Vasu Sharma, Gabriel Synnaeve, Hu Xu, Herve Jegou, Julien Mairal, Patrick Labatut, Armand Joulin, and Piotr Bojanowski. 2024. DINOv2: Learning Robust Visual Features without Supervision. Transactions on Machine Learning Research (TMLR) (2024)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3321408.3322624"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3531232.3531250"},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML).","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In Proceedings of the International Conference on Machine Learning (ICML)."},{"volume-title":"Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP).","author":"Radhakrishnan Srijith","key":"e_1_3_2_1_46_1","unstructured":"Srijith Radhakrishnan, Chao-Han Huck Yang, Sumeer Ahmad Khan, Rohit Kumar, Narsis A. Kiani, David Gomez-Cabrero, and Jesper N. Tegner. 2023. Whispering LLaMA: A Cross-Modal Generative Error Correction Framework for Speech Recognition. In Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP)."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR).","author":"Ravi Nikhila","year":"2025","unstructured":"Nikhila Ravi, Valentin Gabeur, Yuan-Ting Hu, Ronghang Hu, Chaitanya Ryali, Tengyu Ma, Haitham Khedr, Roman R\u00e4dle, Chloe Rolland, Laura Gustafson, Eric Mintun, Junting Pan, Kalyan Vasudev Alwala, Nicolas Carion, Chao-Yuan Wu, Ross Girshick, Piotr Doll\u00e1r, and Christoph Feichtenhofer. 2025. SAM 2: Segment Anything in Images and Videos. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS).","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. In Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_49_1","volume-title":"Crowd-Sourced, Large-Scale, Long-Tailed Dataset For Anime Character Recognition. arXiv preprint arXiv:2101.08674","author":"Rios Edwin Arkel","year":"2021","unstructured":"Edwin Arkel Rios, Wen-Huang Cheng, and Bo-Cheng Lai. 2021. DAF:re: A Challenging, Crowd-Sourced, Large-Scale, Long-Tailed Dataset For Anime Character Recognition. arXiv preprint arXiv:2101.08674 (2021)."},{"key":"e_1_3_2_1_50_1","volume-title":"Understanding inverse document frequency: on theoretical arguments for IDF. Journal of Documentation","author":"Robertson Stephen","year":"2004","unstructured":"Stephen Robertson. 2004. Understanding inverse document frequency: on theoretical arguments for IDF. Journal of Documentation (2004)."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-322"},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the Asian Conference on Computer Vision (ACCV).","author":"Sachdeva Ragav","year":"2024","unstructured":"Ragav Sachdeva, Gyungin Shin, and Andrew Zisserman. 2024. Tails Tell Tales: Chapter-Wide Manga Transcriptions with Character Names. In Proceedings of the Asian Conference on Computer Vision (ACCV)."},{"key":"e_1_3_2_1_53_1","unstructured":"SceneDetect. 2025. SceneDetect. https:\/\/www.scenedetect.com\/."},{"key":"e_1_3_2_1_54_1","volume-title":"Narayanan","author":"Somandepalli Krishna","year":"2018","unstructured":"Krishna Somandepalli, Naveen Kumar, Tanaya Guha, and Shrikanth S. Narayanan. 2018. Unsupervised Discovery of Character Dictionaries in Animation Movies. IEEE Transactions on Multimedia (2018)."},{"key":"e_1_3_2_1_55_1","volume-title":"Identity-Aware Semi-Supervised Learning for Comic Character Re-Identification. arXiv preprint arXiv:2308.09096","author":"Soykan G\u00fcrkan","year":"2023","unstructured":"G\u00fcrkan Soykan, Deniz Yuret, and Tevfik Metin Sezgin. 2023. Identity-Aware Semi-Supervised Learning for Comic Character Re-Identification. arXiv preprint arXiv:2308.09096 (2023)."},{"key":"e_1_3_2_1_56_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale Dan Bikel Lukas Blecher Cristian Canton Ferrer Moya Chen Guillem Cucurull David Esiobu Jude Fernandes Jeremy Fu Wenyin Fu Brian Fuller Cynthia Gao Vedanuj Goswami Naman Goyal Anthony Hartshorn Saghar Hosseini Rui Hou Hakan Inan Marcin Kardas Viktor Kerkez Madian Khabsa Isabel Kloumann Artem Korenev Punit Singh Koura Marie-Anne Lachaux Thibaut Lavril Jenya Lee Diana Liskovich Yinghai Lu Yuning Mao Xavier Martinet Todor Mihaylov Pushkar Mishra Igor Molybog Yixin Nie Andrew Poulton Jeremy Reizenstein Rashi Rungta Kalyan Saladi Alan Schelten Ruan Silva Eric Michael Smith Ranjan Subramanian Xiaoqing Ellen Tan Binh Tang Ross Taylor Adina Williams Jian Xiang Kuan Puxin Xu Zheng Yan Iliyan Zarov Yuchen Zhang Angela Fan Melanie Kambadur Sharan Narang Aurelien Rodriguez Robert Stojnic Sergey Edunov and Thomas Scialom. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv preprint arXiv: 2307.09288 (2023)."},{"key":"e_1_3_2_1_57_1","volume-title":"Representation Learning with Contrastive Predictive Coding. arXiv preprint arXiv:1807.03748","author":"van den Oord A\u00e4ron","year":"2018","unstructured":"A\u00e4ron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation Learning with Contrastive Predictive Coding. arXiv preprint arXiv:1807.03748 (2018)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_1_59_1","volume-title":"Contextual AD Narration with Interleaved Multimodal Sequence. arXiv preprint arXiv:2403.12922","author":"Wang Hanlin","year":"2024","unstructured":"Hanlin Wang, Zhan Tong, Kecheng Zheng, Yujun Shen, and Limin Wang. 2024c. Contextual AD Narration with Interleaved Multimodal Sequence. arXiv preprint arXiv:2403.12922 (2024)."},{"key":"e_1_3_2_1_60_1","volume-title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191","author":"Wang Peng","year":"2024","unstructured":"Peng Wang, Shuai Bai, Sinan Tan, Shijie Wang, Zhihao Fan, Jinze Bai, Keqin Chen, Xuejing Liu, Jialin Wang, Wenbin Ge, Yang Fan, Kai Dang, Mengfei Du, Xuancheng Ren, Rui Men, Dayiheng Liu, Chang Zhou, Jingren Zhou, and Junyang Lin. 2024a. Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution. arXiv preprint arXiv:2409.12191 (2024)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462628"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-209"},{"key":"e_1_3_2_1_63_1","unstructured":"Yan Wang. 2019. Danbooru 2018 Anime Character Recognition Dataset. https:\/\/github.com\/grapeot\/Danbooru2018AnimeCharacterRecognitionDataset"},{"key":"e_1_3_2_1_64_1","volume-title":"Shot-by-Shot: Film-Grammar-Aware Training-Free Audio Description Generation. arXiv preprint arXiv:2504.01020","author":"Xie Junyu","year":"2025","unstructured":"Junyu Xie, Tengda Han, Max Bain, Arsha Nagrani, Eshika Khandelwal, G\u00fcl Varol, Weidi Xie, and Andrew Zisserman. 2025. Shot-by-Shot: Film-Grammar-Aware Training-Free Audio Description Generation. arXiv preprint arXiv:2504.01020 (2025)."},{"key":"e_1_3_2_1_65_1","volume-title":"Proceedings of the Asian Conference on Computer Vision (ACCV).","author":"Xie Junyu","year":"2024","unstructured":"Junyu Xie, Tengda Han, Max Bain, Arsha Nagrani, G\u00fcl Varol, Weidi Xie, and Andrew Zisserman. 2024. AutoAD-Zero: A Training-Free Framework for Zero-Shot Audio Description. In Proceedings of the Asian Conference on Computer Vision (ACCV)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548027"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN54540.2023.10191980"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683892"},{"key":"e_1_3_2_1_69_1","volume-title":"ACFD: Asymmetric Cartoon Face Detector. arXiv preprint arXiv:2007.00899","author":"Zhang Bin","year":"2020","unstructured":"Bin Zhang, Jian Li, Yabiao Wang, Zhipeng Cui, Yili Xia, Chengjie Wang, Jilin Li, and Feiyue Huang. 2020. ACFD: Asymmetric Cartoon Face Detector. arXiv preprint arXiv:2007.00899 (2020)."},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01295"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413892"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413726"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755041","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:25:27Z","timestamp":1765308327000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755041"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":72,"alternative-id":["10.1145\/3746027.3755041","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755041","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}