{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,14]],"date-time":"2026-05-14T19:22:10Z","timestamp":1778786530441,"version":"3.51.4"},"reference-count":88,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00845","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"9036-9047","source":"Crossref","is-referenced-by-count":0,"title":["INS-MMBench: A Comprehensive Benchmark for Evaluating LVLMs' Performance in Insurance"],"prefix":"10.1109","author":[{"given":"Chenwei","family":"Lin","sequence":"first","affiliation":[{"name":"Institute of Big Data, Fudan University,Shanghai,China,200433"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hanjia","family":"Lyu","sequence":"additional","affiliation":[{"name":"University of Rochester,Department of Computer Science,Rochester,NY,USA,14627"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xian","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Economics, Fudan University,Shanghai,China,200433"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiebo","family":"Luo","sequence":"additional","affiliation":[{"name":"University of Rochester,Department of Computer Science,Rochester,NY,USA,14627"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"issue":"6","key":"ref1","article-title":"Vqa-med: Overview of the medical visual question answering task at imageclef 2019","volume":"2","author":"Abacha","year":"2019","journal-title":"CLEF (working notes)"},{"key":"ref2","article-title":"Gpt-4 technical report","volume-title":"arXiv preprint","author":"Achiam","year":"2023"},{"key":"ref3","volume-title":"Damage level dataset","author":"Agyemang","year":"2021"},{"key":"ref4","volume-title":"Damage type dataset","author":"Agyemang","year":"2022"},{"key":"ref5","volume-title":"Agriculture crop images","year":"2021"},{"key":"ref6","volume-title":"Car crash severity detection dataset","year":"2022"},{"key":"ref7","article-title":"Qwen-vl: A frontier large vision-language model with versatile abilities","volume-title":"arXiv preprint","author":"Bai","year":"2023"},{"key":"ref8","volume-title":"Damages dataset","year":"2022"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3641289"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.328"},{"key":"ref11","doi-asserted-by":"crossref","DOI":"10.52202\/079017-0850","article-title":"Are we on the right way for evaluating large vision-language models?","volume-title":"arXiv preprint","author":"Chen","year":"2024"},{"key":"ref12","article-title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","volume-title":"arXiv preprint","author":"Chen","year":"2023"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.3390\/drones4010007"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00290"},{"key":"ref15","volume-title":"fire detection dataset","year":"2023"},{"key":"ref16","volume-title":"Worker-safety dataset","year":"2022"},{"key":"ref17","volume-title":"dataset dashboard dataset","year":"2024"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1093\/jamia\/ocv080"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0795"},{"key":"ref20","article-title":"Talk2bev: Language-enhanced bird\u2019s-eye view maps for autonomous driving","volume-title":"arXiv preprint","author":"Dewangan","year":"2023"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.2478\/fprj-2018-0003"},{"key":"ref22","volume-title":"Vlmevalkit: An open-source toolkit for evaluating large multi-modality models","author":"Duan","year":"2024"},{"key":"ref23","volume-title":"Wheat growth stage challenge","year":"2023"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1057\/s41288-017-0073-0"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1057\/s41288-020-00201-7"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3133797"},{"key":"ref27","volume-title":"Tuning car detection dataset","author":"Nagiyev","year":"2023"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2020.3044678"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICTer58063.2022.10024089"},{"key":"ref30","article-title":"A challenger to gpt-4v? early explorations of gemini in visual expertise","volume-title":"arXiv preprint","author":"Fu","year":"2023"},{"key":"ref31","volume-title":"Gemini pro","year":"2024"},{"key":"ref32","first-page":"10","article-title":"Creating xbd: A dataset for assessing building damage from satellite imagery","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops","author":"Gupta","year":"2019"},{"key":"ref33","volume-title":"Hackerearth machine learning challenge: Vehicle insurance claim","year":"2020"},{"key":"ref34","doi-asserted-by":"crossref","DOI":"10.1109\/CVPR52733.2024.02093","article-title":"Omnimedvqa: A new large-scale comprehensive evaluation benchmark for medical lvlm","volume-title":"arXiv preprint","author":"Hu","year":"2024"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.67"},{"key":"ref36","volume-title":"Fall detection dataset","year":"2024"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/j.lindif.2023.102274"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.77"},{"key":"ref39","article-title":"Seed-bench-2: Benchmarking multimodal large language models","volume-title":"arXiv preprint","author":"Li","year":"2023"},{"key":"ref40","article-title":"Seed-bench: Benchmarking multimodal llms with generative comprehension","volume-title":"arXiv preprint","author":"Li","year":"2023"},{"key":"ref41","article-title":"Seed-bench-2-plus: Benchmarking multimodal large language models with text-rich visual comprehension","volume-title":"arXiv preprint","author":"Li","year":"2024"},{"key":"ref42","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"International conference on machine learning","author":"Li","year":"2023"},{"key":"ref43","article-title":"Mvbench: A comprehensive multi-modal video understanding benchmark","volume-title":"arXiv preprint","author":"Li","year":"2023"},{"key":"ref44","article-title":"An anti-fraud system for car insurance claim based on visual evidence","author":"Li","year":"2018","journal-title":"arXiv preprint"},{"key":"ref45","article-title":"A comprehensive evaluation of gpt-4v on knowledge-intensive visual question answering","volume-title":"arXiv preprint","author":"Li","year":"2023"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1145\/3604237.3626869"},{"key":"ref47","article-title":"Automated evaluation of large vision-language models on self-driving corner cases","volume-title":"arXiv preprint","author":"Li","year":"2024"},{"key":"ref48","article-title":"Harnessing gpt-4v (ision) for insurance: A preliminary exploration","volume-title":"arXiv preprint","author":"Lin","year":"2024"},{"key":"ref49","article-title":"Mitigating hallucination in large multi-modal models via robust instruction tuning","volume-title":"The Twelfth International Conference on Learning Representations","author":"Liu","year":"2023"},{"key":"ref50","article-title":"Mmc: Advancing multimodal chart understanding with large-scale instruction tuning","volume-title":"arXiv preprint","author":"Liu","year":"2023"},{"key":"ref51","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2024","journal-title":"Advances in neural information processing systems"},{"key":"ref52","article-title":"Mmbench: Is your multi-modal model an all-around player?","volume-title":"arXiv preprint","author":"Liu","year":"2023"},{"key":"ref53","article-title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","volume-title":"arXiv preprint","author":"Lu","year":"2023"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/3709005"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3299223"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1016\/j.dib.2021.107321"},{"key":"ref57","volume-title":"Hello gpt-4o","year":"2024"},{"key":"ref58","volume-title":"blood-pressure-monitor-display dataset","year":"2024"},{"key":"ref59","article-title":"Charting new territories: Exploring the geographic and geospatial capabilities of multimodal llms","volume-title":"arXiv preprint","author":"Roberts","year":"2023"},{"key":"ref60","article-title":"Scibench: Benchmarking large multimodal models for scientific figure interpretation","volume-title":"arXiv preprint","author":"Roberts","year":"2024"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2020.06.008"},{"key":"ref62","doi-asserted-by":"crossref","DOI":"10.1148\/radiol.230163","volume-title":"Chatgpt and other large language models are double-edged swords","author":"Shen","year":"2023"},{"key":"ref63","volume-title":"Car dent scratch detection(1) dataset","year":"2022"},{"key":"ref64","volume-title":"Introducing qwen-vl","year":"2024"},{"key":"ref65","volume-title":"Precision viticulture dataset for detailed vineyard mapping composed of geotagged smartphone ground images, phytosanitary status, uav orthomosaics, 3d point clouds, and rtk gnss data - northern spain, july 2022","author":"V\u00e9lez","year":"2023"},{"key":"ref66","article-title":"Surgical-lvlm: Learning to adapt large vision-language model for grounded visual question answering in robotic surgery","volume-title":"arXiv preprint","author":"Wang","year":"2024"},{"key":"ref67","article-title":"Measuring multimodal mathematical reasoning with math-vision dataset","volume-title":"arXiv preprint","author":"Wang","year":"2024"},{"key":"ref68","article-title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","volume":"36","author":"Wang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2023.3258480"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.3390\/su11236795"},{"key":"ref71","article-title":"Emergent abilities of large language models","volume-title":"arXiv preprint","author":"Wei","year":"2022"},{"key":"ref72","volume-title":"mjdfodf-qmbuf dataset","year":"2023"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/BigData59044.2023.10386743"},{"key":"ref74","article-title":"Lvlm-ehub: A comprehensive evaluation benchmark for large vision-language models","volume-title":"arXiv preprint","author":"Xu","year":"2023"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1007\/s11831-020-09504-3"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_16"},{"issue":"1","key":"ref77","article-title":"The dawn of lmms: Preliminary explorations with gpt-4v (ision)","volume-title":"arXiv preprint","volume":"9","author":"Yang","year":"2023"},{"key":"ref78","article-title":"Mplug-owl: Modularization empowers large language models with multimodality","volume-title":"arXiv preprint","author":"Ye","year":"2023"},{"key":"ref79","article-title":"A survey on multimodal large language models","volume-title":"arXiv preprint","author":"Yin","year":"2023"},{"key":"ref80","article-title":"Mmt-bench: A comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi","volume-title":"arXiv preprint","author":"Ying","year":"2024"},{"key":"ref81","article-title":"Llama-adapter: Efficient fine-tuning of language models with zero-init attention","volume-title":"arXiv preprint","author":"Zhang","year":"2023"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i09.7110"},{"key":"ref83","article-title":"M3exam: A multilingual, multimodal, multilevel benchmark for examining large language models","volume":"36","author":"Zhang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.140"},{"key":"ref85","article-title":"Automatic chain of thought prompting in large language models","volume-title":"arXiv preprint","author":"Zhang","year":"2022"},{"key":"ref86","article-title":"A survey of large language models","volume-title":"arXiv preprint","author":"Zhao","year":"2023"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"ref88","article-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","volume-title":"arXiv preprint","author":"Zhu","year":"2023"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11445980.pdf?arnumber=11445980","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T06:16:17Z","timestamp":1777529777000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11445980\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":88,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00845","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}