{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T07:12:34Z","timestamp":1778051554618,"version":"3.51.4"},"reference-count":93,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006228","name":"Oak Ridge National Laboratory","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006228","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00582","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"6016-6026","source":"Crossref","is-referenced-by-count":0,"title":["RoadBench: A Vision-Language Foundation Model and Benchmark for Road Damage Understanding"],"prefix":"10.1109","author":[{"given":"Xi","family":"Xiao","sequence":"first","affiliation":[{"name":"University of Alabama at Birmingham,Birmingham,AL,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunbei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tulane University,New Orleans,LA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Janet","family":"Wang","sequence":"additional","affiliation":[{"name":"Tulane University,New Orleans,LA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Zhao","sequence":"additional","affiliation":[{"name":"Northeastern University,Boston,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxiang","family":"Wei","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology,Atlanta,GA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hengjia","family":"Li","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University,Pittsburgh,PA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanshu","family":"Li","sequence":"additional","affiliation":[{"name":"Brown University,Providence,RI,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiao","family":"Wang","sequence":"additional","affiliation":[{"name":"Oak Ridge National Laboratory,Oak Ridge,TN,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Swalpa Kumar","family":"Roy","sequence":"additional","affiliation":[{"name":"Tezpur University,Assam,India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Xu","sequence":"additional","affiliation":[{"name":"Harvard University,Cambridge,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianyang","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Alabama at Birmingham,Birmingham,AL,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1723"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.dib.2021.107133"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/BigData55660.2022.10021040"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1002\/gdj3.260"},{"key":"ref5","article-title":"Visit-bench: A dynamic benchmark for evaluating instruction-following vision-and-language models","author":"Bitton","year":"2023"},{"key":"ref6","article-title":"Yolov4: Optimal speed and accuracy of object detection","author":"Bochkovskiy","year":"2020"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.autcon.2024.105328"},{"key":"ref9","article-title":"Attentive heatmap: Visual explanations for vision transformers","author":"Chefer","year":"2022"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/tvcg.2024.3456320"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2255"},{"key":"ref12","article-title":"Sok: Can synthetic images replace real data? a survey of utility and privacy of synthetic image generation","author":"Chung","year":"2025"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2142"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.isprsjprs.2024.01.004"},{"key":"ref15","first-page":"1","article-title":"Carla: An open urban driving simulator","volume-title":"Conference on robot learning","author":"Dosovitskiy"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.123940"},{"key":"ref17","article-title":"You only look at one sequence: Rethinking transformer in vision through object detection","author":"Fang","year":"2021"},{"key":"ref18","article-title":"You only look at one sequence: Rethinking transformer in vision through object detection","author":"Fang","year":"2021"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i16.29777"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/tgrs.2020.3016820"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01354"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01081"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/IV55156.2024.10588373"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00128"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.02629"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2025.3571946"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01263"},{"key":"ref29","article-title":"Elevater: A benchmark and toolkit for evaluating language-augmented visual models","author":"Li","year":"2022"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01780"},{"key":"ref31","article-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation","author":"Li","year":"2022"},{"key":"ref32","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"ref33","article-title":"A survey on benchmarks of multimodal large language models","author":"Li","year":"2024"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2024.3389945"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/PRAI55851.2022.9904187"},{"key":"ref36","article-title":"Cycle-yolo: A efficient and robust framework for pavement damage detection","author":"Li","year":"2024"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.32388\/ob1z2a"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02520"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"ref42","article-title":"Ii-bench: An image implication understanding benchmark for multimodal large language models","author":"Liu","year":"2024"},{"key":"ref43","article-title":"Deepseek-vl: Towards real-world vision-language understanding","author":"Lu","year":"2024"},{"key":"ref44","article-title":"Learn to explain: Multimodal reasoning via thought chains for science question answering","author":"Lu","year":"2022"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00331"},{"key":"ref46","first-page":"116513","article-title":"Understanding the transferability of representations via task-relatedness","volume-title":"Advances in Neural Information Processing Systems","author":"Mehra","year":"2024"},{"key":"ref47","article-title":"Representation learning with contrastive predictive coding","author":"den Oord","year":"2018"},{"key":"ref48","article-title":"Gpt-4 technical report","year":"2023"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.303"},{"key":"ref50","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2016.2577031"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_7"},{"key":"ref54","article-title":"Image2struct: Benchmarking structure extraction for vision-language models","author":"Roberts","year":"2024"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2016.2552248"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.191"},{"key":"ref58","article-title":"Wikido: A new benchmark evaluating cross-modal retrieval for vision-language models","author":"Tankala","year":"2024"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2018.00143"},{"key":"ref60","article-title":"Yolov10: Real-time end-to-end object detection","author":"Wang","year":"2024"},{"key":"ref61","article-title":"Doctor approved: Generating medically accurate skin disease images through AI\u2013expert feedback","volume-title":"2nd Workshop on Models of Human Feedback for AI Alignment","author":"Wang"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2024.3391751"},{"key":"ref63","article-title":"Cogvlm: Visual expert for pretrained language models","author":"Wang","year":"2024"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73195-2_4"},{"key":"ref65","article-title":"Journey-bench: A challenging one-stop vision-language understanding benchmark of generated images","author":"Wang","year":"2024"},{"key":"ref66","article-title":"Synscapes: A photore-alistic synthetic dataset for street scene parsing","author":"Wrenninge","year":"2018"},{"key":"ref67","article-title":"One-step effective diffusion network for real-world image super-resolution","author":"Wu","year":"2024"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02405"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888616"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888616"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754858"},{"key":"ref72","article-title":"Visual variational autoencoder prompt tuning","author":"Xiao","year":"2025"},{"key":"ref73","article-title":"Describe anything in medical images","author":"Xiao","year":"2025"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25428"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02212"},{"key":"ref76","article-title":"mplug-owl: Modularization empowers large language models with multimodality","author":"Ye","year":"2023"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1158"},{"key":"ref78","article-title":"Pp-picodet: A better real-time object detector on mobile devices","author":"Yu","year":"2021"},{"key":"ref79","article-title":"Pp-picodet: A better real-time object detector on mobile devices","author":"Yu","year":"2021"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2024.3382837"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"ref82","article-title":"Florence: A new foundation model for computer vision","author":"Yuan","year":"2021"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.3390\/app12157594"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2024.3416508"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3767"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/WACV61041.2025.00117"},{"key":"ref87","article-title":"DPCore: Dynamic prompt coreset for continual test-time adaptation","volume-title":"Forty-second International Conference on Machine Learning","author":"Zhang"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00314"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01605"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0048"},{"key":"ref92","article-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","year":"2023"},{"key":"ref93","article-title":"Deformable detr: Deformable transformers for end-to-end object detection","author":"Zhu","year":"2021"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492204.pdf?arnumber=11492204","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:17:32Z","timestamp":1778048252000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492204\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":93,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00582","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}