{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T23:16:10Z","timestamp":1761174970715,"version":"build-2065373602"},"reference-count":16,"publisher":"Polish Information Processing Society","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"DOI":"10.15439\/2025f2608","type":"proceedings-article","created":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T07:44:23Z","timestamp":1761119063000},"page":"665-673","source":"Crossref","is-referenced-by-count":0,"title":["reVISION: A Polish Benchmark for Evaluating Vision-Language Models on Multimodal National Exam Data"],"prefix":"10.15439","volume":"43","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8283-1118","authenticated-orcid":true,"given":"Micha\u0142","family":"Ciesi\u00f3\u0142ka","sequence":"first","affiliation":[{"name":"Adam Mickiewicz University, Center for Artificial Intelligence AMU, Uniwersytetu Pozna\u0144skiego 4, 61-614 Pozna\u0144, Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8066-4533","authenticated-orcid":true,"given":"Filip","family":"Grali\u0144ski","sequence":"additional","affiliation":[{"name":"Adam Mickiewicz University, Center for Artificial Intelligence AMU, Uniwersytetu Pozna\u0144skiego 4, 61-614 Pozna\u0144, Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"6175","published-online":{"date-parts":[[2025,10,15]]},"reference":[{"key":"ref1","unstructured":"M. Abdin, J. Aneja, H. Behl, S. Bubeck, R. Eldan, S. Gunasekar,\nM. Harrison, R. J. Hewett, M. Javaheripi, P. Kauffmann, J. R. Lee, Y. T.\nLee, Y. Li, W. Liu, C. C. T. Mendes, A. Nguyen, E. Price, G. de Rosa,\nO. Saarikivi, A. Salim, S. Shah, X. Wang, R. Ward, Y. Wu, D. Yu,\nC. Zhang, and Y. Zhang. Phi-4 technical report, 2024."},{"key":"ref2","unstructured":"S. Bai, K. Chen, X. Liu, J. Wang, W. Ge, S. Song, K. Dang, P. Wang,\nS. Wang, J. Tang, H. Zhong, Y. Zhu, M. Yang, Z. Li, J. Wan, P. Wang,\nW. Ding, Z. Fu, Y. Xu, J. Ye, X. Zhang, T. Xie, Z. Cheng, H. Zhang,\nZ. Yang, H. Xu, and J. Lin. Qwen2.5-vl technical report. arXiv preprint\nhttps:\/\/arxiv.org\/abs\/2502.13923, 2025."},{"key":"ref3","unstructured":"Z. Chen, W. Wang, Y. Cao, Y. Liu, Z. Gao, E. Cui, J. Zhu, S. Ye,\nH. Tian, Z. Liu, et al. Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling. arXiv\npreprint https:\/\/arxiv.org\/abs\/2412.05271, 2024."},{"key":"ref4","doi-asserted-by":"crossref","unstructured":"R. J. Das, S. E. Hristov, H. Li, D. I. Dimitrov, I. Koychev, and P. Nakov.\nExams-v: A multi-discipline multilingual multimodal exam benchmark\nfor evaluating vision language models, 2024.","DOI":"10.18653\/v1\/2024.acl-long.420"},{"key":"ref5","unstructured":"L. Gao, J. Tow, B. Abbasi, S. Biderman, S. Black, A. DiPofi, C. Foster,\nL. Golding, J. Hsu, A. Le Noac\u2019h, H. Li, K. McDonell, N. Muennighoff, C. Ociepa, J. Phang, L. Reynolds, H. Schoelkopf, A. Skowron,\nL. Sutawika, E. Tang, A. Thite, B. Wang, K. Wang, and A. Zou. A\nframework for few-shot language model evaluation, 07 2024."},{"key":"ref6","unstructured":"A. P. Gema, J. O. J. Leang, G. Hong, A. Devoto, A. C. M. Mancino,\nR. Saxena, X. He, Y. Zhao, X. Du, M. R. G. Madani, et al. Are we\ndone with MMLU? arXiv preprint https:\/\/arxiv.org\/abs\/2406.04127, 2024."},{"key":"ref7","unstructured":"D. Hendrycks, C. Burns, S. Basart, A. Zou, M. Mazeika, D. Song, and\nJ. Steinhardt. Measuring massive multitask language understanding.\narXiv preprint https:\/\/arxiv.org\/abs\/2009.03300, 2020."},{"key":"ref8","unstructured":"K. Jassem, M. Ciesi\u00f3\u0142ka, F. Grali\u0144ski, P. Jab\u0142o\u0144ski, J. Pokrywka, M. Kubis,\nM. Jab\u0142o\u0144ska, and R. Staruch. LLMzSz\u0141: a comprehensive LLM\nbenchmark for Polish, 2025."},{"key":"ref9","unstructured":"H. Lauren\u00e7on, L. Tronchon, M. Cord, and V. Sanh. What matters when\nbuilding vision-language models?, 2024."},{"key":"ref10","doi-asserted-by":"crossref","unstructured":"B. Li, P. Zhang, K. Zhang, F. Pu, X. Du, Y. Dong, H. Liu, Y. Zhang,\nG. Zhang, C. Li, and Z. Liu. Lmms-eval: Accelerating the development\nof large multimoal models, March 2024.","DOI":"10.18653\/v1\/2025.findings-naacl.51"},{"key":"ref11","doi-asserted-by":"crossref","unstructured":"H. Liu, C. Li, Y. Li, and Y. J. Lee. Improved baselines with visual\ninstruction tuning, 2023.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref12","doi-asserted-by":"crossref","unstructured":"O. Vinyals, A. Toshev, S. Bengio, and D. Erhan. Show and tell: A neural\nimage caption generator. In Computer Vision and Pattern Recognition,\n2015.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"ref13","unstructured":"P. Wang, S. Bai, S. Tan, S. Wang, Z. Fan, J. Bai, K. Chen, X. Liu, J. Wang,\nW. Ge, Y. Fan, K. Dang, M. Du, X. Ren, R. Men, D. Liu, C. Zhou, J. Zhou,\nand J. Lin. Qwen2-vl: Enhancing vision-language model\u2019s perception of\nthe world at any resolution. arXiv preprint https:\/\/arxiv.org\/abs\/2409.12191, 2024."},{"key":"ref14","doi-asserted-by":"crossref","unstructured":"X. Yue, Y. Ni, K. Zhang, T. Zheng, R. Liu, G. Zhang, S. Stevens, D. Jiang,\nW. Ren, Y. Sun, C. Wei, B. Yu, R. Yuan, R. Sun, M. Yin, B. Zheng,\nZ. Yang, Y. Liu, W. Huang, H. Sun, Y. Su, and W. Chen. MMMU:\nA massive multi-discipline multimodal understanding and reasoning\nbenchmark for expert AGI, 2024.","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"ref15","unstructured":"W. Zhang, S. M. Aljunied, C. Gao, Y. K. Chia, and L. Bing. M3exam:\nA multilingual, multimodal, multilevel benchmark for examining large\nlanguage models, 2023."},{"key":"ref16","doi-asserted-by":"crossref","unstructured":"T. Zhao, T. Zhang, M. Zhu, H. Shen, K. Lee, X. Lu, and J. Yin. Vl-checklist: Evaluating pre-trained vision-language models with objects,\nattributes and relations, 2023.","DOI":"10.18653\/v1\/2022.emnlp-demos.4"}],"event":{"name":"20th Conference on Computer Science and Intelligence Systems (FedCSIS)","theme":"Computer Science and Intelligence Systems","location":"Krak\u00f3w, Poland","acronym":"FedCSIS","number":"20","start":{"date-parts":[[2025,9,14]]},"end":{"date-parts":[[2025,9,17]]}},"container-title":["Annals of Computer Science and Information Systems","Proceedings of the 20th Conference on Computer Science and Intelligence Systems (FedCSIS)"],"original-title":[],"deposited":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T07:50:42Z","timestamp":1761119442000},"score":1,"resource":{"primary":{"URL":"https:\/\/annals-csis.org\/Volume_43\/drp\/2608.html"}},"subtitle":[],"proceedings-subject":"Computer Science and Information Systems","short-title":[],"issued":{"date-parts":[[2025,10,15]]},"references-count":16,"URL":"https:\/\/doi.org\/10.15439\/2025f2608","relation":{},"ISSN":["2300-5963"],"issn-type":[{"value":"2300-5963","type":"print"}],"subject":[],"published":{"date-parts":[[2025,10,15]]}}}