{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:57:37Z","timestamp":1781539057032,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":66,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810730","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"2246-2255","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["A-MAR: Agent-based Multimodal Art Retrieval for Fine-Grained Artwork Understanding"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-4318-8505","authenticated-orcid":false,"given":"Shuai","family":"Wang","sequence":"first","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0298-0905","authenticated-orcid":false,"given":"Hongyi","family":"Zhu","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7943-2591","authenticated-orcid":false,"given":"Jiahong","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands and Amazon AGI, Seattle, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8447-872X","authenticated-orcid":false,"given":"Yixian","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0872-2054","authenticated-orcid":false,"given":"Chengxi","family":"Zeng","sequence":"additional","affiliation":[{"name":"University of Bristol, Bristol, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1904-8736","authenticated-orcid":false,"given":"Stevan","family":"Rudinac","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7423-3902","authenticated-orcid":false,"given":"Monika","family":"Kackovic","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8070-8719","authenticated-orcid":false,"given":"Nachoem","family":"Wijnberg","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands and College of Business and Economics, University of Johannesburg, Johannesburg, South Africa"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4097-4136","authenticated-orcid":false,"given":"Marcel","family":"Worring","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01140"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"e_1_3_3_2_4_2","unstructured":"Anthropic. 2025. Introducing Claude Haiku 4.5. https:\/\/www.anthropic.com\/news\/claude-haiku-4-5. 15-10-2025."},{"key":"e_1_3_3_2_5_2","unstructured":"Anthropic. 2025. Introducing Claude Sonnet 4.5. https:\/\/www.anthropic.com\/news\/claude-sonnet-4-5. 29-09-2025."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Zechen Bai Yuta Nakashima and Noa Garc\u00eda. 2021. Explain Me the Painting: Multi-Topic Knowledgeable Art Description Generation. 2021 IEEE\/CVF International Conference on Computer Vision (ICCV) (2021) 5402\u20135412. https:\/\/api.semanticscholar.org\/CorpusID:237490413","DOI":"10.1109\/ICCV48922.2021.00537"},{"key":"e_1_3_3_2_7_2","volume-title":"Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An Automatic Metric for MT Evaluation with Improved Correlation with Human Judgments. In Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization."},{"key":"e_1_3_3_2_8_2","volume-title":"Painting and experience in fifteenth century Italy: a primer in the social history of pictorial style","author":"Baxandall Michael","year":"1988","unstructured":"Michael Baxandall. 1988. Painting and experience in fifteenth century Italy: a primer in the social history of pictorial style. Oxford Paperbacks."},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681656"},{"key":"e_1_3_3_2_10_2","volume-title":"Advances in Neural Information Processing Systems","author":"Brown Tom et\u00a0al.","year":"2020","unstructured":"Tom et\u00a0al. Brown. 2020. Language Models are Few-Shot Learners. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33765-9_11"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-68796-0_36"},{"key":"e_1_3_3_2_13_2","volume-title":"Forty-first International Conference on Machine Learning","author":"Chen Dongping","year":"2024","unstructured":"Dongping Chen, Ruoxi Chen, Shilin Zhang, Yaochen Wang, Yinuo Liu, Huichi Zhou, Qihui Zhang, Yao Wan, Pan Zhou, and Lichao Sun. 2024. Mllm-as-a-judge: Assessing multimodal llm-as-a-judge with vision-language benchmark. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_3_2_14_2","volume-title":"European Conference on Computer Vision","author":"Chen Lin","year":"2024","unstructured":"Lin Chen, Jinsong Li, Xiaoyi Dong, Pan Zhang, Conghui He, Jiaqi Wang, Feng Zhao, and Dahua Lin. 2024. Sharegpt4v: Improving large multi-modal models with better captions. In European Conference on Computer Vision."},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00444"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.5244\/C.28.38"},{"key":"e_1_3_3_2_17_2","unstructured":"Darren Edge Ha Trinh Newman Cheng Joshua Bradley Alex Chao Apurva Mody Steven Truitt Dasha Metropolitansky Robert\u00a0Osazuwa Ness and Jonathan Larson. 2025. From Local to Global: A Graph RAG Approach to Query-Focused Summarization. https:\/\/arxiv.org\/abs\/2404.16130"},{"key":"e_1_3_3_2_18_2","unstructured":"Athanasios Efthymiou Stevan Rudinac Monika Kackovic Nachoem Wijnberg and Marcel Worring. 2026. VL-KGE: Vision-Language Models Meet Knowledge Graph Embeddings. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2603.02435 (2026)."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475586"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671470"},{"key":"e_1_3_3_2_21_2","volume-title":"Proceedings of the European Conference in Computer Vision Workshops","author":"Garcia Noa","year":"2018","unstructured":"Noa Garcia and George Vogiatzis. 2018. How to Read Paintings: Semantic Art Understanding with Multi-Modal Retrieval. In Proceedings of the European Conference in Computer Vision Workshops."},{"key":"e_1_3_3_2_22_2","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et\u00a0al. 2024. A survey on llm-as-a-judge. The Innovation (2024)."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.568"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-short.65"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730351"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-96-2071-5_30"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Lei Huang Weijiang Yu Weitao Ma Weihong Zhong Zhangyin Feng Haotian Wang Qianglong Chen Weihua Peng Xiaocheng Feng Bing Qin and Ting Liu. 2025. A Survey on Hallucination in Large Language Models: Principles Taxonomy Challenges and Open Questions. ACM Trans. Inf. Syst. (2025).","DOI":"10.1145\/3703155"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.74"},{"key":"e_1_3_3_2_30_2","unstructured":"Ziwei Ji Nayeon Lee Rita Frieske Tiezheng Yu Dan Su Yan Xu Etsuko Ishii Ye\u00a0Jin Bang Andrea Madotto and Pascale Fung. 2023. Survey of Hallucination in Natural Language Generation. ACM Comput. Surv. Article 248 (March 2023) 38\u00a0pages."},{"key":"e_1_3_3_2_31_2","volume-title":"Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI-24","author":"Jiang Yanbei","year":"2024","unstructured":"Yanbei Jiang, Krista\u00a0A. Ehinger, and Jey\u00a0Han Lau. 2024. KALE: An Artwork Image Captioning System Augmented with Heterogeneous Graph. In Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI-24. International Joint Conferences on Artificial Intelligence Organization."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.5244\/C.28.122"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/3643489.3661132"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-53302-0_31"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.579"},{"key":"e_1_3_3_2_36_2","unstructured":"Dan Li Shuai Wang Jie Zou Chang Tian Elisha Nieuwburg Fengyuan Sun and Evangelos Kanoulas. 2021. Paint4Poem: A Dataset for Artistic Visualization of Classical Chinese Poems. arxiv:https:\/\/arXiv.org\/abs\/2109.11682\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2109.11682"},{"key":"e_1_3_3_2_37_2","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In Proceedings of the 40th International Conference on Machine Learning."},{"key":"e_1_3_3_2_38_2","volume-title":"Text Summarization Branches Out","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. ROUGE: A Package for Automatic Evaluation of Summaries. In Text Summarization Branches Out. Association for Computational Linguistics."},{"key":"e_1_3_3_2_39_2","volume-title":"Advances in Neural Information Processing Systems","author":"Liu Haotian","year":"2023","unstructured":"Haotian Liu, Chunyuan Li, Qingyang Wu, and Yong\u00a0Jae Lee. 2023. Visual Instruction Tuning. In Advances in Neural Information Processing Systems , Vol.\u00a036."},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.153"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Mistral. 2025. Introducing Mistral 3. https:\/\/mistral.ai\/news\/mistral-3. 25-12-2025.","DOI":"10.5840\/mr2025312"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02058"},{"key":"e_1_3_3_2_43_2","unstructured":"OpenAI. 2025. Introducing GPT\u20115.2. https:\/\/openai.com\/index\/introducing-gpt-5-2\/. 11-12-2025."},{"key":"e_1_3_3_2_44_2","volume-title":"Meaning in the Visual Arts","author":"Panofsky E.","year":"1955","unstructured":"E. Panofsky. 1955. Meaning in the Visual Arts. University of Chicago Press. https:\/\/books.google.nl\/books?id=Qsa00QEACAAJ"},{"key":"e_1_3_3_2_45_2","volume-title":"Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a Method for Automatic Evaluation of Machine Translation. In Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics."},{"key":"e_1_3_3_2_46_2","unstructured":"Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark Gretchen Krueger and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. arxiv:https:\/\/arXiv.org\/abs\/2103.00020\u00a0[cs.CV]"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"crossref","unstructured":"Cynthia Rudin. 2019. Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead. Nature machine intelligence 1 5 (2019) 206\u2013215.","DOI":"10.1038\/s42256-019-0048-x"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350972"},{"key":"e_1_3_3_2_49_2","unstructured":"Aditi Singh Abul Ehtesham Saket Kumar and Tala\u00a0Talaei Khoei. 2025. Agentic Retrieval-Augmented Generation: A Survey on Agentic RAG. ArXiv abs\/2501.09136 (2025)."},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-30645-8_66"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3414445"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"crossref","unstructured":"Gjorgji Strezoski and Marcel Worring. 2018. OmniArt: A Large-scale Artistic Benchmark. ACM Trans. Multimedia Comput. Commun. Appl. (2018).","DOI":"10.1145\/3273022"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2016.7533051"},{"key":"e_1_3_3_2_54_2","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Yang Fan Kai Dang Mengfei Du Xuancheng Ren Rui Men Dayiheng Liu Chang Zhou Jingren Zhou and Junyang Lin. 2024. Qwen2-VL: Enhancing Vision-Language Model\u2019s Perception of the World at Any Resolution. https:\/\/arxiv.org\/abs\/2409.12191"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3755673"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-53311-2_34"},{"key":"e_1_3_3_2_57_2","unstructured":"Shuai Wang David\u00a0W Zhang Jia-Hong Huang Stevan Rudinac Monika Kackovic Nachoem Wijnberg and Marcel Worring. 2024. Ada-hgnn: Adaptive sampling for scalable hypergraph neural networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.13372 (2024)."},{"key":"e_1_3_3_2_58_2","unstructured":"Zixun Wu. 2022. Artwork interpretation. Master\u2019s thesis University of Melbourne (2022)."},{"key":"e_1_3_3_2_59_2","volume-title":"International Conference on Learning Representations (ICLR)","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_3_2_60_2","unstructured":"Xinlei Yu Changmiao Wang Hui Jin Ahmed Elazab Gangyong Jia Xiang Wan Changqing Zou and Ruiquan Ge. 2025. CRISP-SAM2: SAM2 with Cross-Modal Interaction and Semantic Prompting for Multi-Organ Segmentation. arxiv:https:\/\/arXiv.org\/abs\/2506.23121\u00a0[eess.IV] https:\/\/arxiv.org\/abs\/2506.23121"},{"key":"e_1_3_3_2_61_2","volume-title":"Proceedings of conference on language modelling","author":"Yuan Zheng","year":"2024","unstructured":"Zheng Yuan, HU Xue, Xinyi Wang, Yongming Liu, Zhuanzhe Zhao, and Kun Wang. 2024. ArtGPT-4: Towards Artistic-understanding Large Vision-Language Models with Enhanced Adapter. In Proceedings of conference on language modelling."},{"key":"e_1_3_3_2_62_2","unstructured":"Penghao Zhao Hailin Zhang Qinhan Yu Zhengren Wang Yunteng Geng Fangcheng Fu Ling Yang Wentao Zhang and Bin Cui. 2024. Retrieval-Augmented Generation for AI-Generated Content: A Survey. ArXiv abs\/2402.19473 (2024)."},{"key":"e_1_3_3_2_63_2","unstructured":"Ruochen Zhao Hailin Chen Weishi Wang Fangkai Jiao Xuan\u00a0Long Do Chengwei Qin Bosheng Ding Xiaobao Guo Minzhi Li Xingxuan Li and Shafiq Joty. 2023. Retrieving Multimodal Information for Augmented Generation: A Survey. arxiv:https:\/\/arXiv.org\/abs\/2303.10868\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2303.10868"},{"key":"e_1_3_3_2_64_2","doi-asserted-by":"crossref","unstructured":"Ruochen Zhao Hailin Chen Weishi Wang Fangkai Jiao Do\u00a0Xuan Long Chengwei Qin Bosheng Ding Xiaobao Guo Minzhi Li Xingxuan Li and Shafiq\u00a0R. Joty. 2023. Retrieving Multimodal Information for Augmented Generation: A Survey. ArXiv abs\/2303.10868 (2023).","DOI":"10.18653\/v1\/2023.findings-emnlp.314"},{"key":"e_1_3_3_2_65_2","unstructured":"Wayne\u00a0Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong Yifan Du Chen Yang Yushuo Chen Zhipeng Chen Jinhao Jiang Ruiyang Ren Yifan Li Xinyu Tang Zikang Liu Peiyu Liu Jian-Yun Nie and Ji-Rong Wen. 2025. A Survey of Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2303.18223\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2303.18223"},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.1145\/3652583.3658032"},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"crossref","unstructured":"Hongyi Zhu Jia-Hong Huang Yixian Shen Stevan Rudinac and Evangelos Kanoulas. 2025. Interactive Image Retrieval Meets Query Rewriting with Large Language and Vision Language Models. ACM Trans. Multimedia Comput. Commun. Appl. Article 286 (Oct. 2025) 23\u00a0pages.","DOI":"10.1145\/3744910"}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:42:32Z","timestamp":1781538152000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810730"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":66,"alternative-id":["10.1145\/3805622.3810730","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810730","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}