{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:52:56Z","timestamp":1781931176872,"version":"3.54.5"},"reference-count":53,"publisher":"Association for Natural Language Processing","issue":"2","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Journal of Natural Language Processing"],"published-print":{"date-parts":[[2026]]},"DOI":"10.5715\/jnlp.33.509","type":"journal-article","created":{"date-parts":[[2026,6,14]],"date-time":"2026-06-14T22:11:45Z","timestamp":1781475105000},"page":"509-536","source":"Crossref","is-referenced-by-count":0,"title":["Cross-Task Evaluation and Empirical Analysis of Japanese Visual Language Models","\u65e5\u672c\u8a9e\u8996\u899a\u8a00\u8a9e\u30e2\u30c7\u30eb\u306e\u30bf\u30b9\u30af\u6a2a\u65ad\u8a55\u4fa1\u3068\u5b9f\u8a3c\u7684\u5206\u6790"],"prefix":"10.5715","volume":"33","author":[{"given":"Koki","family":"Maeda","sequence":"first","affiliation":[{"name":"Institute of Science Tokyo"},{"name":"National Institute of Informatics Research and Development Center for Large Language Models"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Issa","family":"Sugiura","sequence":"additional","affiliation":[{"name":"Kyoto University"},{"name":"National Institute of Informatics Research and Development Center for Large Language Models"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yusuke","family":"Oda","sequence":"additional","affiliation":[{"name":"National Institute of Informatics Research and Development Center for Large Language Models"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuhei","family":"Kurita","sequence":"additional","affiliation":[{"name":"National Institute of Informatics"},{"name":"National Institute of Informatics Research and Development Center for Large Language Models"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Naoaki","family":"Okazaki","sequence":"additional","affiliation":[{"name":"Institute of Science Tokyo"},{"name":"National Institute of Informatics Research and Development Center for Large Language Models"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"3685","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"Akiba, T., Shing, M., Tang, Y., Sun, Q., and Ha, D. (2024). \u201cEvolutionary Optimization of Model Merging Recipes.\u201d <i>arXiv preprint arXiv:2403.13187<\/i>.","DOI":"10.1038\/s42256-024-00975-8"},{"key":"2","unstructured":"Bai, S., Chen, K., Liu, X., Wang, J., Ge, W., Song, S., Dang, K., Wang, P., Wang, S., Tang, J., Zhong, H., Zhu, Y., Yang, M., Li, Z., Wan, J., Wang, P., Ding, W., Fu, Z., Xu, Y., Ye, J., Zhang, X., Xie, T., Cheng, Z., Zhang, H., Yang, Z., Xu, H., and Lin, J. (2025). \u201cQwen2.5-VL Technical Report.\u201d <i>arXiv preprint arXiv:2502.13923<\/i>."},{"key":"3","unstructured":"Dash, S., Nan, Y., Dang, J., Ahmadian, A., Singh, S., Smith, M., Venkitesh, B., Shmyhlo, V., Aryabumi, V., Beller-Morales, W., Pekmez, J., Ozuzu, J., Richemond, P., Locatelli, A., Frosst, N., Blunsom, P., Gomez, A., Zhang, I., Fadaee, M., Govindassamy, M., Roy, S., Gall\u00e9, M., Ermis, B., \u00dcst\u00fcn, A., and Hooker, S. (2025). \u201cAya Vision: Advancing the Frontier of Multilingual Multimodality.\u201d <i>arXiv preprint arXiv:2505.08751<\/i>."},{"key":"4","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., and Fei-Fei, L. (2009). \u201cImageNet: A Large-scale Hierarchical Image Database.\u201d In <i>2009 IEEE Conference on Computer Vision and Pattern Recognition<\/i>, pp. 248\u2013255.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"5","doi-asserted-by":"crossref","unstructured":"Duan, H., Yang, J., Qiao, Y., Fang, X., Chen, L., Liu, Y., Dong, X., Zang, Y., Zhang, P., Wang, J., Lin, D., and Chen, K. (2024). \u201cVLMEvalKit: An Open-Source Toolkit for Evaluating Large Multi-modality Models.\u201d In <i>ACM International Conference on Multimedia<\/i>, pp. 11198\u201311201.","DOI":"10.1145\/3664647.3685520"},{"key":"6","doi-asserted-by":"crossref","unstructured":"Fu, X., Hu, Y., Li, B., Feng, Y., Wang, H., Lin, X., Roth, D., Smith, N. A., Ma, W.-C., and Krishna, R. (2024). \u201cBLINK: Multimodal Large Language Models Can See but Not Perceive.\u201d In <i>European Conference on Computer Vision<\/i>, pp. 148\u2013166.","DOI":"10.1007\/978-3-031-73337-6_9"},{"key":"7","unstructured":"Gao, L., Tow, J., Abbasi, B., Biderman, S., Black, S., DiPofi, A., Foster, C., Golding, L., Hsu, J., Le Noac\u2019h, A., Li, H., McDonell, K., Muennighoff, N., Ociepa, C., Phang, J., Reynolds, L., Schoelkopf, H., Skowron, A., Sutawika, L., Tang, E., Thite, A., Wang, B., Wang, K., and Zou, A. (2024). \u201cThe Language Model Evaluation Harness.\u201d https:\/\/zenodo.org\/records\/12608602."},{"key":"8","unstructured":"Gemma-Team et al. (2025). \u201cGemma 3 Technical Report.\u201d <i>arXiv preprint arXiv:2503.19786<\/i>."},{"key":"9","unstructured":"Habib, N., Fourrier, C., Kydl\u00ed\u010dek, H., Wolf, T., and Tunstall, L. (2023). \u201cLightEval: A Lightweight Framework for LLM Evaluation.\u201d https:\/\/github.com\/huggingface\/lighteval."},{"key":"10","doi-asserted-by":"crossref","unstructured":"Han, N.\uff0c\u690d\u7530\u66a2\u5927\uff0c\u5927\u5dbd\u5321\u4fca\uff0c\u52dd\u53c8\u667a\uff0c\u938c\u7530\u5553\u8f14\uff0c\u6e05\u4e38\u5bdb\u4e00\uff0c\u5150\u7389\u8cb4\u5fd7\uff0c\u83c5\u539f\u6714\uff0cChen, B.\uff0c\u677e\u7530\u5bdb\uff0c\u5bae\u5c3e\u7950\u4ecb\uff0c\u6751\u8107\u6709\u543e\uff0c\u5289\u5f18\u6bc5 (2024). llm-jp-eval: \u65e5\u672c\u8a9e\u5927\u898f\u6a21\u8a00\u8a9e\u30e2\u30c7\u30eb\u306e\u81ea\u52d5\u8a55\u4fa1\u30c4\u30fc\u30eb. \u8a00\u8a9e\u51e6\u7406\u5b66\u4f1a\u7b2c30\u56de\u5e74\u6b21\u5927\u4f1a (NLP2024), pp. 2085\u20132089. [N. Han et al. (2024). llm-jp-eval: Nihongo Daikibo Gengo Moderu No Jido Hyoka Tsuru. In Proceedings of the 30th Annual Meeting of the Association for Natural Language Processing, pp. 2085\u20132089.].","DOI":"10.5715\/jnlp.30.1128"},{"key":"11","unstructured":"Inagaki, A. (2024). \u201cLLaVA-CALM2-SigLIP.\u201d. https:\/\/huggingface.co\/cyberagent\/llava-calm2-siglip."},{"key":"12","unstructured":"Inoue, Y., Akiba, T., and Makoto, S. (2024a). \u201cJA-Multi-Image-VQA.\u201d https:\/\/huggingface.co\/datasets\/SakanaAI\/JA-Multi-Image-VQA."},{"key":"13","unstructured":"Inoue, Y., Sasaki, K., Ochi, Y., Fujii, K., Tanahashi, K., and Yamaguchi, Y. (2024b). \u201cHeron-Bench: A Benchmark for Evaluating Vision Language Models in Japanese.\u201d <i>arXiv preprint arXiv:2404.07824<\/i>."},{"key":"14","doi-asserted-by":"crossref","unstructured":"Kembhavi, A., Salvato, M., Kolve, E., Seo, M., Hajishirzi, H., and Farhadi, A. (2016). \u201cA Diagram is Worth a Dozen Images.\u201d In <i>European Conference on Computer Vision<\/i>, pp. 235\u2013251.","DOI":"10.1007\/978-3-319-46493-0_15"},{"key":"15","unstructured":"Liang, P., Bommasani, R., Lee, T., Tsipras, D., Soylu, D., Yasunaga, M., Zhang, Y., Narayanan, D., Wu, Y., Kumar, A., Newman, B., Yuan, B., Yan, B., Zhang, C., Cosgrove, C., Manning, C. D., Re, C., Acosta-Navas, D., Hudson, D. A., Zelikman, E., Durmus, E., Ladhak, F., Rong, F., Ren, H., Yao, H., WANG, J., Santhanam, K., Orr, L., Zheng, L., Yuksekgonul, M., Suzgun, M., Kim, N., Guha, N., Chatterji, N. S., Khattab, O., Henderson, P., Huang, Q., Chi, R. A., Xie, S. M., Santurkar, S., Ganguli, S., Hashimoto, T., Icard, T., Zhang, T., Chaudhary, V., Wang, W., Li, X., Mai, Y., Zhang, Y., and Koreeda, Y. (2023). \u201cHolistic Evaluation of Language Models.\u201d <i>Transactions of Machine Learning Research<\/i>."},{"key":"16","unstructured":"Lin, C.-Y. (2004). \u201cROUGE: A Package for Automatic Evaluation of Summaries.\u201d In <i>Text Summarization Branches Out<\/i>, pp. 74\u201381, Barcelona, Spain. Association for Computational Linguistics."},{"key":"17","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., and Lee, Y. J. (2024a). \u201cImproved Baselines with Visual Instruction Tuning.\u201d In <i>IEEE\/CVF Conference on Computer Vision and Pattern Recognition<\/i>, pp. 26296\u201326306.","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"18","unstructured":"Liu, H., Li, C., Li, Y., Li, B., Zhang, Y., Shen, S., and Lee, Y. J. (2024b). \u201cLLaVA-NeXT: Improved Reasoning, OCR, and World Knowledge.\u201d https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/."},{"key":"19","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Wu, Q., and Lee, Y. J. (2023). \u201cVisual Instruction Tuning.\u201d In <i>Advances in Neural Information Processing Systems<\/i>, pp. 34892\u201334916.","DOI":"10.52202\/075280-1516"},{"key":"20","unstructured":"Lu, P., Bansal, H., Xia, T., Liu, J., Li, C., Hajishirzi, H., Cheng, H., Chang, K.-W., Galley, M., and Gao, J. (2024). \u201cMathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts.\u201d In <i>International Conference on Learning Representations (ICLR)<\/i>."},{"key":"21","unstructured":"\u524d\u7530\u822a\u5e0c\uff0c\u6749\u6d66\u4e00\u7473\uff0c\u5c0f\u7530\u60a0\u4ecb\uff0c\u6817\u7530\u4fee\u5e73\uff0c\u5ca1\u5d0e\u76f4\u89b3 (2025a). llm-jp-eval-mm: \u65e5\u672c\u8a9e\u8996\u899a\u8a00\u8a9e\u30e2\u30c7\u30eb\u306e\u81ea\u52d5\u8a55\u4fa1\u57fa\u76e4. \u8a00\u8a9e\u51e6\u7406\u5b66\u4f1a\u7b2c31\u56de\u5e74\u6b21\u5927\u4f1a (NLP), pp. 1303\u20131308. [K. Maeda et al. (2025a). llm-jp-eval-mm: Nihongo Shikaku Gengo Moderu No Jido Hyoka Kiban. Proceedings of the 31st Annual Meeting of the Association for Natural Language Processing, pp. 1303\u20131308.]."},{"key":"22","unstructured":"\u524d\u7530\u822a\u5e0c\uff0c\u9577\u8c37\u5ddd\u9a0e\u5e73\uff0c\u6817\u7530\u4fee\u5e73\uff0c\u5c0f\u7530\u60a0\u4ecb\uff0c\u5fb3\u4e45\u826f\u5b50\uff0c\u5ca1\u5d0e\u76f4\u89b3 (2025b). \u65e5\u672c\u306e\u6587\u5316\u5e38\u8b58\u30fb\u65e5\u5e38\u751f\u6d3b\u77e5\u8b58\u7406\u89e3\u306e\u305f\u3081\u306e\u8996\u899a\u8a00\u8a9e\u30d9\u30f3\u30c1\u30de\u30fc\u30afMECHA-Ja\u306e\u69cb\u7bc9. \u7814\u7a76\u5831\u544a\u81ea\u7136\u8a00\u8a9e\u51e6\u7406\uff08NL\uff09. [K. Maeda et al. (2025b). Nihon no Bunka Joshiki Nichijo Seikatsu Chishiki Rikai no Tame no Shikaku Gengo Benchimaku MECHA-Ja no Kochiku. Kenkyu Hokoku Shizen Gengo Shori (NL).]."},{"key":"23","doi-asserted-by":"crossref","unstructured":"Marino, K., Rastegari, M., Farhadi, A., and Mottaghi, R. (2019). \u201cOK-VQA: A Visual Question Answering Benchmark Requiring External Knowledge.\u201d In <i>IEEE\/CVF Conference on Computer Vision and Pattern Recognition<\/i>.","DOI":"10.1109\/CVPR.2019.00331"},{"key":"24","doi-asserted-by":"crossref","unstructured":"Masry, A., Long, D. X., Tan, J. Q., Joty, S., and Hoque, E. (2022). \u201cChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning.\u201d In Muresan, S., Nakov, P., and Villavicencio, A. (Eds.), <i>Findings of the Association for Computational Linguistics: ACL 2022<\/i>, pp. 2263\u20132279, Dublin, Ireland. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2022.findings-acl.177"},{"key":"25","doi-asserted-by":"crossref","unstructured":"Mathew, M., Bagal, V., Tito, R. P., Karatzas, D., Valveny, E., and Jawahar, C. V. (2021a). \u201cInfographicVQA.\u201d <i>arXiv preprint arXiv:2104.12756<\/i>.","DOI":"10.1109\/WACV51458.2022.00264"},{"key":"26","doi-asserted-by":"crossref","unstructured":"Mathew, M., Karatzas, D., and Jawahar, C. V. (2021b). \u201cDocVQA: A Dataset for VQA on Document Images.\u201d In <i>IEEE\/CVF Winter Conference on Applications of Computer Vision<\/i>, pp. 2199\u20132208.","DOI":"10.1109\/WACV48630.2021.00225"},{"key":"27","unstructured":"Meta (2024). \u201cLlama-3.2-11B-Vision.\u201d https:\/\/huggingface.co\/meta-llama\/Llama-3.2-11B-Vision."},{"key":"28","doi-asserted-by":"crossref","unstructured":"Nguyen, D., Prasad, A., Stengel-Eskin, E., and Bansal, M. (2025). \u201cMulti-Attribute Steering of Language Models via Targeted Intervention.\u201d In Che, W., Nabende, J., Shutova, E., and Pilehvar, M. T. (Eds.), <i>Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)<\/i>, pp. 20619\u201320634, Vienna, Austria. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2025.acl-long.1007"},{"key":"29","doi-asserted-by":"crossref","unstructured":"Onami, E., Kurita, S., Miyanishi, T., and Watanabe, T. (2024). \u201cJDocQA: Japanese Document Question Answering Dataset for Generative Language Models.\u201d In Calzolari, N., Kan, M.-Y., Hoste, V., Lenci, A., Sakti, S., and Xue, N. (Eds.), <i>Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)<\/i>, pp. 9503\u20139514, Torino, Italia. ELRA and ICCL.","DOI":"10.63317\/5k3radukw3sc"},{"key":"30","doi-asserted-by":"crossref","unstructured":"Onohara, S., Miyai, A., Imajuku, Y., Egashira, K., Baek, J., Yue, X., Neubig, G., and Aizawa, K. (2025). \u201cJMMMU: A Japanese Massive Multi-discipline Multimodal Understanding Benchmark for Culture-aware Evaluation.\u201d In Chiruzzo, L., Ritter, A., and Wang, L. (Eds.), <i>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)<\/i>, pp. 932\u2013950, Albuquerque, New Mexico. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2025.naacl-long.43"},{"key":"31","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., and Zhu, W.-J. (2002). \u201cBleu: A Method for Automatic Evaluation of Machine Translation.\u201d In Isabelle, P., Charniak, E., and Lin, D. (Eds.), <i>Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics<\/i>, pp. 311\u2013318, Philadelphia, Pennsylvania, USA. Association for Computational Linguistics.","DOI":"10.3115\/1073083.1073135"},{"key":"32","doi-asserted-by":"crossref","unstructured":"Rasheed, H., Maaz, M., Shaker, A., Khan, S., Cholakal, H., Anwer, R. M., Baldwin, T., Felsberg, M., and Khan, F. S. (2025). \u201cPalo: A Large Multilingual Multimodal Language Model.\u201d In <i>IEEE\/CVF Winter Conference on Applications of Computer Vision<\/i>, pp. 1745\u20131754.","DOI":"10.1109\/WACV61041.2025.00177"},{"key":"33","unstructured":"Recruit Co., L. (2024). \u201crecruit-jp\/japanese-image-classification-evaluation-dataset.\u201d https:\/\/huggingface.co\/datasets\/recruit-jp\/japanese-image-classification-evaluation-dataset."},{"key":"34","doi-asserted-by":"crossref","unstructured":"Romero, D. et al. (2024). \u201cCVQA: Culturally-diverse Multilingual Visual Question Answering Benchmark.\u201d In <i>Advances in Neural Information Processing Systems<\/i>, pp. 11479\u201311505.","DOI":"10.52202\/079017-0366"},{"key":"35","doi-asserted-by":"crossref","unstructured":"Sasagawa, K., Maeda, K., Sugiura, I., Kurita, S., Okazaki, N., and Kawahara, D. (2025). \u201cConstructing Multimodal Datasets from Scratch for Rapid Development of a Japanese Visual Language Model.\u201d In Dziri, N., Ren, S. X., and Diao, S. (Eds.), <i>Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (System Demonstrations)<\/i>, pp. 470\u2013484, Albuquerque, New Mexico. Association for Computational Linguistics.","DOI":"10.18653\/v1\/2025.naacl-demo.38"},{"key":"36","unstructured":"SB-Intuitions (2025). Sarashina2-Vision: \u65e5\u672c\u8a9e\u7279\u5316\u306e\u5927\u898f\u6a21\u8996\u899a\u8a00\u8a9e\u30e2\u30c7\u30eb\u306e\u516c\u958b. https:\/\/www.sbintuitions.co.jp\/blog\/entry\/2025\/03\/17\/111659."},{"key":"37","unstructured":"Shimizu, N., Rong, N., and Miyazaki, T. (2018). \u201cVisual Question Answering Dataset for Bilingual Image Understanding: A Study of Cross-Lingual Transfer Using Attention Maps.\u201d In Bender, E. M., Derczynski, L., and Isabelle, P. (Eds.), <i>Proceedings of the 27th International Conference on Computational Linguistics<\/i>, pp. 1918\u20131928, Santa Fe, New Mexico, USA. Association for Computational Linguistics."},{"key":"38","unstructured":"Shing, M., and Akiba, T. (2023). \u201cJapanese InstructBLIP Alpha.\u201d https:\/\/huggingface.co\/stabilityai\/japanese-instructblip-alpha."},{"key":"39","unstructured":"Shing, M., and Akiba, T. (2024). \u201cJapanese Stable VLM.\u201d https:\/\/huggingface.co\/stabilityai\/japanese-stable-vlm."},{"key":"40","doi-asserted-by":"crossref","unstructured":"Singh, A., Natarjan, V., Shah, M., Jiang, Y., Chen, X., Parikh, D., and Rohrbach, M. (2019). \u201cTowards VQA Models That Can Read.\u201d In <i>IEEE\/CVF Conference on Computer Vision and Pattern Recognition<\/i>, pp. 8317\u20138326.","DOI":"10.1109\/CVPR.2019.00851"},{"key":"41","unstructured":"Tanaka, M., Zhu, P., and Yokoo, S. (2025). \u201cJapanese Image Classification Visual Question Answering (JIC-VQA) Dataset.\u201d https:\/\/huggingface.co\/line-corporation\/JIC-VQA."},{"key":"42","unstructured":"Turing-Inc. (2025). \u65e5\u672c\u8a9eVLM\u300cHeron-NVILA\u300d\u516c\u958b\u2014Qwen2.5-VL-7B\u30fbGemma3-12B\u306b\u5339\u6575\u3059\u308b\u6027\u80fd. https:\/\/zenn.dev\/turing_motors\/articles\/7ac8ebe8756a3e. [Turing-Inc. (2025). Nihongo VLM \u201cHeron-NVILA\u201d Kokai\u2014Qwen2.5-VL-7B Gemma3-12B ni Hitteki suru Seno. https:\/\/zenn.dev\/turing_motors\/articles\/7ac8ebe8756a3e.]."},{"key":"43","unstructured":"\u4e0a\u539f\u5eb7\u5e73\uff0c\u9ed2\u702c\u512a\u4ecb\uff0c\u5b89\u9053\u5065\u4e00\u90ce\uff0cChenJiali\uff0cGaoFan\uff0c\u91d1\u6fa4\u723d\u592a\u90ce\uff0c\u5742\u672c\u62d3\u5f4c\uff0c\u7af9\u7530\u60a0\u54c9\uff0cYangBoming\uff0cZhaoXinjie\uff0c\u6751\u5c3e\u6643\u5e73\uff0c\u5409\u7530\u6d69\uff0c\u7530\u6751\u5b5d\u4e4b\uff0c\u5408\u7530\u61b2\u4eba\uff0c\u559c\u9023\u5ddd\u512a\uff0c\u539f\u7530\u9054\u4e5f (2025). Asagi: \u5408\u6210\u30c7\u30fc\u30bf\u30bb\u30c3\u30c8\u3092\u6d3b\u7528\u3057\u305f\u5927\u898f\u6a21\u65e5\u672c\u8a9eVLM. \u8a00\u8a9e\u51e6\u7406\u5b66\u4f1a\u7b2c31\u56de\u5e74\u6b21\u5927\u4f1a (NLP), pp. 1202\u20131207. [K. Uehara et al. (2025). Asagi: Gosei Detasetto wo Katsuyo shita Daikibo Nihongo VLM. Proceedings of the 31st Annual Meeting of the Association for Natural Language Processing, pp. 1202\u20131207.]."},{"key":"44","doi-asserted-by":"crossref","unstructured":"Wang, K., Pan, J., Shi, W., Lu, Z., Zhan, M., and Li, H. (2024a). \u201cMeasuring Multimodal Mathematical Reasoning with the MATH-Vision Dataset.\u201d In <i>Advances in Neural Information Processing Systems (NeurIPS), Datasets and Benchmarks Track<\/i>, pp. 95095\u201395169.","DOI":"10.52202\/079017-3014"},{"key":"45","unstructured":"Wang, P., Bai, S., Tan, S., Wang, S., Fan, Z., Bai, J., Chen, K., Liu, X., Wang, J., Ge, W., Fan, Y., Dang, K., Du, M., Ren, X., Men, R., Liu, D., Zhou, C., Zhou, J., and Lin, J. (2024b). \u201cQwen2-VL: Enhancing Vision-Language Model\u2019s Perception of the World at Any Resolution.\u201d <i>arXiv preprint arXiv:2409.12191<\/i>."},{"key":"46","unstructured":"Xu, P., Shao, W., Zhang, K., Gao, P., Liu, S., Lei, M., Meng, F., Huang, S., Qiao, Y., and Luo, P. (2023). \u201cLVLM-eHub: A Comprehensive Evaluation Benchmark for Large Vision-Language Models.\u201d <i>arXiv preprint arXiv:2306.09265<\/i>."},{"key":"47","doi-asserted-by":"crossref","unstructured":"Yang, Z., Tang, J., Li, Z., Wang, P., Wan, J., Zhong, H., Liu, X., Yang, M., Wang, P., Bai, S., Jin, L., and Lin, J. (2024). \u201cCC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy.\u201d <i>arXiv preprint arXiv:2412.02210<\/i>.","DOI":"10.1109\/ICCV51701.2025.02019"},{"key":"48","doi-asserted-by":"crossref","unstructured":"Yue, X., Ni, Y., Zhang, K., Zheng, T., Liu, R., Zhang, G., Stevens, S., Jiang, D., Ren, W., Sun, Y., Wei, C., Yu, B., Yuan, R., Sun, R., Yin, M., Zheng, B., Yang, Z., Liu, Y., Huang, W., Sun, H., Su, Y., and Chen, W. (2024a). \u201cMMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI.\u201d In <i>IEEE\/CVF Conference on Computer Vision and Pattern Recognition<\/i>, pp. 9556\u20139567.","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"49","unstructured":"Yue, X., Song, Y., Asai, A., Kim, S., de Dieu Nyandwi, J., Khanuja, S., Kantharuban, A., Sutawika, L., Ramamoorthy, S., and Neubig, G. (2024b). \u201cPangea: A Fully Open Multilingual Multimodal LLM for 39 Languages.\u201d <i>arXiv preprint arXiv:2410.16153<\/i>."},{"key":"50","doi-asserted-by":"crossref","unstructured":"Zhang, K., Li, B., Zhang, P., Pu, F., Cahyono, J. A., Hu, K., Liu, S., Zhang, Y., Yang, J., Li, C., and Liu, Z. (2024a). \u201cLMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models.\u201d <i>arXiv preprint arXiv:2407.12772<\/i>.","DOI":"10.18653\/v1\/2025.findings-naacl.51"},{"key":"51","doi-asserted-by":"crossref","unstructured":"Zhang, R., Jiang, D., Zhang, Y., Lin, H., Guo, Z., Qiu, P., Zhou, A., Lu, P., Chang, K.-W., Gao, P., and Li, H. (2024b). \u201cMathVerse: Does Your Multi-modal LLM Truly See the Diagrams in Visual Math Problems?\u201d In <i>European Conference on Computer Vision (ECCV)<\/i>.","DOI":"10.1007\/978-3-031-73242-3_10"},{"key":"52","doi-asserted-by":"crossref","unstructured":"Zheng, L., Chiang, W.-L., Sheng, Y., Zhuang, S., Wu, Z., Zhuang, Y., Lin, Z., Li, Z., Li, D., Xing, E. P., Zhang, H., Gonzalez, J. E., and Stoica, I. (2024). \u201cJudging LLM-as-a-judge with MT-bench and Chatbot Arena.\u201d In <i>Advances in Neural Information Processing Systems<\/i>, pp. 46595\u201346623.","DOI":"10.52202\/075280-2020"},{"key":"53","unstructured":"Zhu, J., Wang, W., Chen, Z., Liu, Z., Ye, S., Gu, L., Tian, H., Duan, Y., Su, W., Shao, J., Gao, Z., Cui, E., Wang, X., Cao, Y., Liu, Y., Wei, X., Zhang, H., Wang, H., Xu, W., Li, H., Wang, J., Deng, N., Li, S., He, Y., Jiang, T., Luo, J., Wang, Y., He, C., Shi, B., Zhang, X., Shao, W., He, J., Xiong, Y., Qu, W., Sun, P., Jiao, P., Lv, H., Wu, L., Zhang, K., Deng, H., Ge, J., Chen, K., Wang, L., Dou, M., Lu, L., Zhu, X., Lu, T., Lin, D., Qiao, Y., Dai, J., and Wang, W. (2025). \u201cInternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models.\u201d <i>arXiv preprint arXiv:2503.19786<\/i>."}],"container-title":["Journal of Natural Language Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_509\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T04:45:08Z","timestamp":1781930708000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/jnlp\/33\/2\/33_509\/_article\/-char\/ja\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":53,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/doi.org\/10.5715\/jnlp.33.509","relation":{},"ISSN":["1340-7619","2185-8314"],"issn-type":[{"value":"1340-7619","type":"print"},{"value":"2185-8314","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}