{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T00:50:54Z","timestamp":1767315054681,"version":"3.48.0"},"publisher-location":"Cham","reference-count":48,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032101846","type":"print"},{"value":"9783032101853","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-10185-3_46","type":"book-chapter","created":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T00:48:36Z","timestamp":1767314916000},"page":"584-596","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Automatic Benchmarking of\u00a0Large Multimodal Models via\u00a0Iterative Experiment Programming"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3044-1320","authenticated-orcid":false,"given":"Alessandro","family":"Conti","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8595-9955","authenticated-orcid":false,"given":"Massimiliano","family":"Mancini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1671-875X","authenticated-orcid":false,"given":"Enrico","family":"Fini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5932-4371","authenticated-orcid":false,"given":"Yiming","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0663-5659","authenticated-orcid":false,"given":"Paolo","family":"Rota","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0228-1147","authenticated-orcid":false,"given":"Elisa","family":"Ricci","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"46_CR1","doi-asserted-by":"crossref","unstructured":"Andreas, J., Rohrbach, M., Darrell, T., Klein, D.: Neural module networks. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.12"},{"key":"46_CR2","doi-asserted-by":"crossref","unstructured":"Antol, S., Agrawal, A., Lu, J., Mitchell, M., Batra, D., Zitnick, C.L., Parikh, D.: VQA: visual question answering. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"46_CR3","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et\u00a0al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"key":"46_CR4","unstructured":"Chen, L., Zhang, Y., Ren, S., Zhao, H., Cai, Z., Wang, Y., Liu, T., Chang, B.: Towards end-to-end embodied decision making with multi-modal large language model: explorations with GPT4-vision and beyond. In: NeurIPS-WS (2023)"},{"key":"46_CR5","unstructured":"Chen, S., Gu, J., Han, Z., Ma, Y., Torr, P., Tresp, V.: Benchmarking robustness of adaptation methods on pre-trained vision-language models. In: NeurIPS (2024)"},{"key":"46_CR6","doi-asserted-by":"crossref","unstructured":"Chen, Y., Hu, H., Luan, Y., Sun, H., Changpinyo, S., Ritter, A., Chang, M.W.: Can pre-trained vision and language models answer visual information-seeking questions? In: EMNLP (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.925"},{"key":"46_CR7","unstructured":"Cheng, S., Guo, Z., Wu, J., Fang, K., Li, P., Liu, H., Liu, Y.: Can vision-language models think from a first-person perspective? (2023) arXiv preprint arXiv:2311.15596"},{"key":"46_CR8","unstructured":"Cho, J., Zala, A., Bansal, M.: Visual programming for step-by-step text-to-image generation and evaluation. In: NeurIPS (2024)"},{"key":"46_CR9","unstructured":"Dai, W., Li, J., Li, D., Tiong, A.M.H., Zhao, J., Wang, W., Li, B., Fung, P.N., Hoi, S.: Instructblip: towards general-purpose vision-language models with instruction tuning. In: NeurIPS (2024)"},{"key":"46_CR10","unstructured":"Fang, A., Ilharco, G., Wortsman, M., Wan, Y., Shankar, V., Dave, A., Schmidt, L.: Data determines distributional robustness in contrastive language image pre-training (clip). In: ICML. PMLR (2022)"},{"key":"46_CR11","unstructured":"Gavrikov, P., Lukasik, J., Jung, S., Geirhos, R., Lamm, B., Mirza, M.J., Keuper, M., Keuper, J.: Are vision language models texture or shape biased and can we steer them? (2024). arXiv preprint arXiv:2403.09193"},{"key":"46_CR12","doi-asserted-by":"crossref","unstructured":"Gupta, T., Kembhavi, A.: Visual programming: compositional visual reasoning without training. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01436"},{"key":"46_CR13","unstructured":"Hsieh, C.Y., Zhang, J., Ma, Z., Kembhavi, A., Krishna, R.: Sugarcrepe: fixing hackable benchmarks for vision-language compositionality. In: NeurIPS (2024)"},{"key":"46_CR14","unstructured":"Idrissi, B.Y., Bouchacourt, D., Balestriero, R., Evtimov, I., Hazirbas, C., Ballas, N., Vincent, P., Drozdzal, M., Lopez-Paz, D., Ibrahim, M.: Imagenet-X: understanding model mistakes with factor of variation annotations (2022). arXiv preprint arXiv:2211.01866"},{"key":"46_CR15","doi-asserted-by":"crossref","unstructured":"Kamath, A., Hessel, J., Chang, K.W.: What\u2019s \u201cup\u2019\u2019 with vision-language models? Investigating their struggle with spatial reasoning. In: The 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.568"},{"key":"46_CR16","doi-asserted-by":"crossref","unstructured":"Khan, Z., BG, V.K., Schulter, S., Fu, Y., Chandraker, M.: Self-training large language models for improved visual program synthesis with visual reinforcement. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01360"},{"key":"46_CR17","unstructured":"Lauren\u00e7on, H., Saulnier, L., Tronchon, L., Bekman, S., Singh, A., Lozhkov, A., Wang, T., Karamcheti, S., Rush, A., Kiela, D., et\u00a0al.: Obelics: an open web-scale filtered dataset of interleaved image-text documents. In: NeurIPS (2023)"},{"key":"46_CR18","unstructured":"Li, C., Wong, C., Zhang, S., Usuyama, N., Liu, H., Yang, J., Naumann, T., Poon, H., Gao, J.: LLaVA-Med: training a large language-and-vision assistant for biomedicine in one day. In: NeurIPS (2024)"},{"key":"46_CR19","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: ICML (2023)"},{"key":"46_CR20","doi-asserted-by":"crossref","unstructured":"Lin, S., Hilton, J., Evans, O.: TruthfulQA: measuring how models mimic human falsehoods. In: ACL (2022)","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"46_CR21","unstructured":"Lu, P., Peng, B., Cheng, H., Galley, M., Chang, K.W., Wu, Y.N., Zhu, S.C., Gao, J.: Chameleon: plug-and-play compositional reasoning with large language models. In: NeurIPS (2024)"},{"key":"46_CR22","doi-asserted-by":"crossref","unstructured":"Ma, M., Ren, J., Zhao, L., Testuggine, D., Peng, X.: Are multimodal transformers robust to missing modality? In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01764"},{"key":"46_CR23","doi-asserted-by":"crossref","unstructured":"Ma, Z., Hong, J., Gul, M.O., Gandhi, M., Gao, I., Krishna, R.: Crepe: can vision-language foundation models reason compositionally? In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01050"},{"key":"46_CR24","unstructured":"Mayilvahanan, P., Wiedemer, T., Rusak, E., Bethge, M., Brendel, W.: Does clip\u2019s generalization performance mainly stem from high train-test similarity? In: NeurIPS-WS (2023)"},{"key":"46_CR25","unstructured":"OpenAI: Introducing chatGPT (2022). https:\/\/openai.com\/blog\/chatgpt"},{"key":"46_CR26","unstructured":"Parisi, A., Zhao, Y., Fiedel, N.: TALM: tool augmented language models (2022). arXiv preprint arXiv:2205.12255"},{"key":"46_CR27","unstructured":"Podell, D., English, Z., Lacey, K., Blattmann, A., Dockhorn, T., M\u00fcller, J., Penna, J., Rombach, R.: SDXL: improving latent diffusion models for high-resolution image synthesis (2023). arXiv preprint arXiv:2307.01952"},{"key":"46_CR28","unstructured":"Qin, Y., Liang, S., Ye, Y., Zhu, K., Yan, L., Lu, Y., Lin, Y., Cong, X., Tang, X., Qian, B., et\u00a0al.: ToolLLM: facilitating large language models to master 16000+ real-world APIs. In: ICLR (2024)"},{"key":"46_CR29","unstructured":"Qiu, J., Zhu, Y., Shi, X., Wenzel, F., Tang, Z., Zhao, D., Li, B., Li, M.: Benchmarking robustness of multimodal image-text models under distribution shift. J. Data-Centric Mach. Learn. Res. (2023)"},{"key":"46_CR30","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"46_CR31","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"46_CR32","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., Huang, Z., Karpathy, A., Khosla, A., Bernstein, M., et al.: Imagenet large scale visual recognition challenge. IIJCV 115, 211\u2013252 (2015)","journal-title":"IIJCV"},{"key":"46_CR33","unstructured":"Schick, T., Dwivedi-Yu, J., Dess\u00ec, R., Raileanu, R., Lomeli, M., Hambro, E., Zettlemoyer, L., Cancedda, N., Scialom, T.: Toolformer: language models can teach themselves to use tools. In: NeurIPS (2024)"},{"key":"46_CR34","unstructured":"Shaham, T.R., Schwettmann, S., Wang, F., Rajaram, A., Hernandez, E., Andreas, J., Torralba, A.: A multimodal automated interpretability agent (2024). arXiv preprint arXiv:2404.14394"},{"key":"46_CR35","unstructured":"Shen, Y., Song, K., Tan, X., Li, D., Lu, W., Zhuang, Y.: HuggingGPT: solving AI tasks with chatGPT and its friends in hugging face. In: NeurIPS (2024)"},{"key":"46_CR36","unstructured":"Shi, Z., Wang, Z., Fan, H., Yin, Z., Sheng, L., Qiao, Y., Shao, J.: Chef: a comprehensive evaluation framework for standardized assessment of multimodal large language models (2023). arXiv preprint arXiv:2311.02692"},{"key":"46_CR37","doi-asserted-by":"crossref","unstructured":"Subramanian, S., Narasimhan, M., Khangaonkar, K., Yang, K., Nagrani, A., Schmid, C., Zeng, A., Darrell, T., Klein, D.: Modular visual question answering via code generation. In: ACL (2023)","DOI":"10.18653\/v1\/2023.acl-short.65"},{"key":"46_CR38","doi-asserted-by":"crossref","unstructured":"Sur\u00eds, D., Menon, S., Vondrick, C.: ViperGPT: visual inference via python execution for reasoning. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01092"},{"key":"46_CR39","doi-asserted-by":"crossref","unstructured":"Tang, Y., Yamada, Y., Zhang, Y., Yildirim, I.: When are lemons purple? The concept association bias of vision-language models. In: EMNLP (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.886"},{"key":"46_CR40","doi-asserted-by":"crossref","unstructured":"Thrush, T., Jiang, R., Bartolo, M., Singh, A., Williams, A., Kiela, D., Ross, C.: Winoground: probing vision and language models for visio-linguistic compositionality. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00517"},{"key":"46_CR41","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F., Rodriguez, A., Joulin, A., Grave, E., Lample, G.: Llama: open and efficient foundation language models (2023). arXiv preprint arXiv:2302.13971"},{"key":"46_CR42","unstructured":"Udandarao, V., Burg, M.F., Albanie, S., Bethge, M.: Visual data-type understanding does not emerge from scaling vision-language models. In: ICLR (2024)"},{"key":"46_CR43","unstructured":"Udandarao, V., Prabhu, A., Ghosh, A., Sharma, Y., Torr, P.H., Bibi, A., Albanie, S., Bethge, M.: No \u201czero-shot\u2019\u2019 without exponential data: pretraining concept frequency determines multimodal model performance (2024). arXiv preprint arXiv:2404.04125"},{"key":"46_CR44","unstructured":"Ye, S., Lauer, J., Zhou, M., Mathis, A., Mathis, M.: AmadeusGPT: a natural language interface for interactive animal behavioral analysis. In: NeurIPS (2024)"},{"key":"46_CR45","doi-asserted-by":"crossref","unstructured":"Yuan, Z., Ren, J., Feng, C.M., Zhao, H., Cui, S., Li, Z.: Visual programming for zero-shot open-vocabulary 3D visual grounding. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01949"},{"key":"46_CR46","unstructured":"Yuksekgonul, M., Bianchi, F., Kalluri, P., Jurafsky, D., Zou, J.: When and why vision-language models behave like bags-of-words, and what to do about it? In: ICLR (2022)"},{"key":"46_CR47","doi-asserted-by":"crossref","unstructured":"Zhang, G., Zhang, Y., Zhang, K., Tresp, V.: Can vision-language models be a good guesser? Exploring VLMS for times and location reasoning. In: WACV (2024)","DOI":"10.1109\/WACV57701.2024.00069"},{"key":"46_CR48","unstructured":"Zhao, Y., Pang, T., Du, C., Yang, X., Li, C., Cheung, N.M.M., Lin, M.: On evaluating adversarial robustness of large vision-language models. In: NeurIPS (2024)"}],"container-title":["Lecture Notes in Computer Science","Image Analysis and Processing \u2013 ICIAP 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-10185-3_46","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T00:48:40Z","timestamp":1767314920000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-10185-3_46"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032101846","9783032101853"],"references-count":48,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-10185-3_46","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ICIAP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Image Analysis and Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Rome","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iciap2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.iciap.org\/home","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}