{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T16:18:00Z","timestamp":1785255480840,"version":"3.55.0"},"publisher-location":"Cham","reference-count":61,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031730320","type":"print"},{"value":"9783031730337","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73033-7_4","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:03:55Z","timestamp":1730333035000},"page":"56-73","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["PathMMU: A Massive Multimodal Expert-Level Benchmark for\u00a0Understanding and\u00a0Reasoning in\u00a0Pathology"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1277-4316","authenticated-orcid":false,"given":"Yuxuan","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8642-4569","authenticated-orcid":false,"given":"Hao","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5705-3718","authenticated-orcid":false,"given":"Chenglu","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sunyi","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qizi","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kai","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunlong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dan","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoxiao","family":"Lan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mengyue","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingxiong","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinheng","family":"Lyu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3246-6935","authenticated-orcid":false,"given":"Tao","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"4_CR1","unstructured":"Alayrac, J.B., et\u00a0al.: Flamingo: a visual language model for few-shot learning. In: NeurIPS, pp. 23716\u201323736 (2022)"},{"key":"4_CR2","doi-asserted-by":"publisher","first-page":"122","DOI":"10.1016\/j.media.2019.05.010","volume":"56","author":"G Aresta","year":"2019","unstructured":"Aresta, G., et al.: Bach: grand challenge on breast cancer histology images. Med. Image Anal. 56, 122\u2013139 (2019)","journal-title":"Med. Image Anal."},{"issue":"4","key":"4_CR3","doi-asserted-by":"publisher","first-page":"e0210706","DOI":"10.1371\/journal.pone.0210706","volume":"14","author":"HB Arunachalam","year":"2019","unstructured":"Arunachalam, H.B., et al.: Viable and necrotic tumor assessment from whole slide images of osteosarcoma using machine-learning and deep-learning models. PLoS ONE 14(4), e0210706 (2019)","journal-title":"PLoS ONE"},{"key":"4_CR4","unstructured":"Bai, J., et al.: Qwen-VL: a frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)"},{"key":"4_CR5","unstructured":"Bavishi, R., et al.: Introducing our multimodal models (2023). https:\/\/www.adept.ai\/blog\/fuyu-8b"},{"key":"4_CR6","unstructured":"Ben\u00a0Abacha, A., Sarrouti, M., Demner-Fushman, D., Hasan, S.A., M\u00fcller, H.: Overview of the VQA-med task at ImageCLEF 2021: visual question answering and generation in the medical domain. In: Proceedings of the CLEF 2021 Conference and Labs of the Evaluation Forum-working notes (2021)"},{"key":"4_CR7","unstructured":"Borkowski, A.A., Bui, M.M., Thomas, L.B., Wilson, C.P., DeLand, L.A., Mastorides, S.M.: Lung and colon cancer histopathological image dataset (lc25000). arXiv preprint arXiv:1912.12142 (2019)"},{"key":"4_CR8","unstructured":"Brown, T., et\u00a0al.: Language models are few-shot learners. In: NeurIPS, pp. 1877\u20131901 (2020)"},{"key":"4_CR9","doi-asserted-by":"crossref","unstructured":"Cai, R., et al.: BenchLMM: benchmarking cross-style visual capability of large multimodal models. arXiv preprint arXiv:2312.02896 (2023)","DOI":"10.1007\/978-3-031-72973-7_20"},{"key":"4_CR10","unstructured":"Chiang, W.L., et al.: Vicuna: an open-source chatbot impressing GPT-4 with 90%* ChatGPT quality (2023). https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"4_CR11","unstructured":"Dai, W., et al.: InstructBLIP: towards general-purpose vision-language models with instruction tuning. arXiv preprint arXiv:2305.06500 (2023)"},{"key":"4_CR12","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: NAACL, pp. 4171\u20134186 (2019)"},{"key":"4_CR13","unstructured":"Driess, D., et\u00a0al.: PaLM-E: an embodied multimodal language model. In: ICML, pp. 8469\u20138488 (2023)"},{"key":"4_CR14","doi-asserted-by":"crossref","unstructured":"Gamper, J., Rajpoot, N.: Multiple instance captioning: learning representations from histopathology textbooks and articles. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 16549\u201316559 (2021)","DOI":"10.1109\/CVPR46437.2021.01628"},{"key":"4_CR15","unstructured":"Gao, P., et\u00a0al.: LLaMA-Adapter V2: parameter-efficient visual instruction model. arXiv preprint arXiv:2304.15010 (2023)"},{"issue":"1","key":"4_CR16","first-page":"1","volume":"3","author":"Y Gu","year":"2021","unstructured":"Gu, Y., et al.: Domain-specific language model pretraining for biomedical natural language processing. ACM Trans. Comput. Healthcare (HEALTH) 3(1), 1\u201323 (2021)","journal-title":"ACM Trans. Comput. Healthcare (HEALTH)"},{"key":"4_CR17","doi-asserted-by":"crossref","unstructured":"Guan, T., et\u00a0al.: HallusionBench: an advanced diagnostic suite for entangled language hallucination and visual illusion in large vision-language models. In: CVPR, pp. 14375\u201314385 (2024)","DOI":"10.1109\/CVPR52733.2024.01363"},{"key":"4_CR18","unstructured":"Han, C., et\u00a0al.: Wsss4luad: grand challenge on weakly-supervised tissue semantic segmentation for lung adenocarcinoma. arXiv preprint arXiv:2204.06455 (2022)"},{"key":"4_CR19","doi-asserted-by":"crossref","unstructured":"He, X., Zhang, Y., Mou, L., Xing, E., Xie, P.: Pathvqa: 30000+ questions for medical visual question answering. arXiv preprint arXiv:2003.10286 (2020)","DOI":"10.36227\/techrxiv.13127537.v1"},{"issue":"9","key":"4_CR20","doi-asserted-by":"publisher","first-page":"2307","DOI":"10.1038\/s41591-023-02504-3","volume":"29","author":"Z Huang","year":"2023","unstructured":"Huang, Z., Bianchi, F., Yuksekgonul, M., Montine, T.J., Zou, J.: A visual-language foundation model for pathology image analysis using medical twitter. Nat. Med. 29(9), 2307\u20132316 (2023)","journal-title":"Nat. Med."},{"key":"4_CR21","unstructured":"Ikezogwo, W., et al.: Quilt-1M: one million image-text pairs for histopathology. In: NeurIPS, pp. 37995\u201338017 (2023)"},{"key":"4_CR22","unstructured":"Kather, J.N., Halama, N., Marx, A.: 100,000 histological images of human colorectal cancer and healthy tissue. Zenodo10 5281 (2018)"},{"key":"4_CR23","doi-asserted-by":"publisher","first-page":"1022967","DOI":"10.3389\/fonc.2022.1022967","volume":"12","author":"K Kriegsmann","year":"2022","unstructured":"Kriegsmann, K., et al.: Deep learning for the detection of anatomical tissue structures and neoplasms of the skin on scanned histopathological tissue sections. Front. Oncol. 12, 1022967 (2022)","journal-title":"Front. Oncol."},{"key":"4_CR24","unstructured":"Kumar, V., Abbas, A.K., Fausto, N., Aster, J.C.: Robbins and Cotran Pathologic Basis of Disease, Professional Edition E-book. Elsevier Health Sciences (2014)"},{"issue":"1","key":"4_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/sdata.2018.251","volume":"5","author":"JJ Lau","year":"2018","unstructured":"Lau, J.J., Gayen, S., Ben Abacha, A., Demner-Fushman, D.: A dataset of clinically generated visual questions and answers about radiology images. Sci. Data 5(1), 1\u201310 (2018)","journal-title":"Sci. Data"},{"key":"4_CR26","unstructured":"Li, B., Zhang, Y., Chen, L., Wang, J., Yang, J., Liu, Z.: Otter: a multi-modal model with in-context instruction tuning. arXiv preprint arXiv:2305.03726 (2023)"},{"key":"4_CR27","doi-asserted-by":"crossref","unstructured":"Li, B., Wang, R., Wang, G., Ge, Y., Ge, Y., Shan, Y.: SEED-Bench: benchmarking multimodal LLMs with generative comprehension. arXiv preprint arXiv:2307.16125 (2023)","DOI":"10.1109\/CVPR52733.2024.01263"},{"key":"4_CR28","unstructured":"Li, C., et al.: LLaVA-Med: training a large language-and-vision assistant for biomedicine in one day. In: NeurIPS, pp. 28541\u201328564 (2023)"},{"key":"4_CR29","unstructured":"Li, C., et\u00a0al.: YOLOv6: a single-stage object detection framework for industrial applications. arXiv preprint arXiv:2209.02976 (2022)"},{"key":"4_CR30","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: ICML, pp. 19730\u201319742 (2023)"},{"key":"4_CR31","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: NeurIPS, pp. 34892\u201334916 (2023)"},{"key":"4_CR32","doi-asserted-by":"crossref","unstructured":"Liu, Y., et\u00a0al.: MMBench: is your multi-modal model an all-around player? arXiv preprint arXiv:2307.06281 (2023)","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"4_CR33","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.Y., Feichtenhofer, C., Darrell, T., Xie, S.: A convnet for the 2020s. In: CVPR, pp. 11976\u201311986 (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"4_CR34","unstructured":"OpenAI: Introducing ChatGPT (2022). https:\/\/openai.com\/blog\/chatgpt"},{"key":"4_CR35","unstructured":"OpenAI: Gpt-4 technical report (2023)"},{"key":"4_CR36","unstructured":"OpenAI: Gpt-4v(ision) system card (2023). https:\/\/cdn.openai.com\/papers\/GPTV_System_Card.pdf"},{"key":"4_CR37","unstructured":"Peng, Z., Wang, W., Dong, L., Hao, Y., Huang, S., Ma, S., Wei, F.: Kosmos-2: grounding multimodal large language models to the world. arXiv preprint arXiv:2306.14824 (2023)"},{"key":"4_CR38","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML, pp. 8748\u20138763 (2021)"},{"issue":"1","key":"4_CR39","first-page":"5485","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(1), 5485\u20135551 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"4_CR40","doi-asserted-by":"crossref","unstructured":"Seyfioglu, M.S., Ikezogwo, W.O., Ghezloo, F., Krishna, R., Shapiro, L.: Quilt-llava: visual instruction tuning by extracting localized narratives from open-source histopathology videos. In: CVPR, pp. 13183\u201313192 (2024)","DOI":"10.1109\/CVPR52733.2024.01252"},{"key":"4_CR41","doi-asserted-by":"publisher","first-page":"105637","DOI":"10.1016\/j.cmpb.2020.105637","volume":"195","author":"J Silva-Rodr\u00edguez","year":"2020","unstructured":"Silva-Rodr\u00edguez, J., Colomer, A., Sales, M.A., Molina, R., Naranjo, V.: Going deeper through the gleason scoring scale: an automatic end-to-end system for histology prostate grading and cribriform pattern detection. Comput. Methods Programs Biomed. 195, 105637 (2020)","journal-title":"Comput. Methods Programs Biomed."},{"key":"4_CR42","unstructured":"Sun, Y., et\u00a0al.: Ernie 3.0: large-scale knowledge enhanced pre-training for language understanding and generation. arXiv preprint arXiv:2107.02137 (2021)"},{"key":"4_CR43","unstructured":"Sun, Y., et al.: PathGen-1.6M: 1.6 million pathology image-text pairs generation through multi-agent collaboration (2024). https:\/\/arxiv.org\/abs\/2407.00203"},{"key":"4_CR44","doi-asserted-by":"crossref","unstructured":"Sun, Y., Zhu, C., Zhang, Y., Li, H., Chen, P., Yang, L.: Assessing the robustness of deep learning-assisted pathological image analysis under practical variables of imaging system. In: ICASSP, pp.\u00a01\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10095887","DOI":"10.1109\/ICASSP49357.2023.10095887"},{"key":"4_CR45","doi-asserted-by":"crossref","unstructured":"Sun, Y., et al.: PathAsst: a generative foundation AI assistant towards artificial general intelligence of pathology. In: AAAI, pp. 5034\u20135042 (2024)","DOI":"10.1609\/aaai.v38i5.28308"},{"key":"4_CR46","unstructured":"Team, G., et\u00a0al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"},{"key":"4_CR47","unstructured":"Touvron, H., et\u00a0al.: LLaMA: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"4_CR48","doi-asserted-by":"publisher","unstructured":"Veeling, B.S., Linmans, J., Winkens, J., Cohen, T., Welling, M.: Rotation equivariant CNNs for digital pathology. In: Frangi, A.F., Schnabel, J.A., Davatzikos, C., Alberola-L\u00f3pez, C., Fichtinger, G. (eds.) MICCAI 2018. LNCS, vol. 11071, pp. 210\u2013218. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-00934-2_24","DOI":"10.1007\/978-3-030-00934-2_24"},{"key":"4_CR49","unstructured":"Wang, J., et\u00a0al.: Evaluation and analysis of hallucination in large vision-language models. arXiv preprint arXiv:2308.15126 (2023)"},{"key":"4_CR50","unstructured":"Wang, W., et\u00a0al.: CogVLM: visual expert for pretrained language models. arXiv preprint arXiv:2311.03079 (2023)"},{"key":"4_CR51","doi-asserted-by":"publisher","unstructured":"Wei, J., et al.: A Petri dish for histopathology image analysis. In: Tucker, A., Henriques Abreu, P., Cardoso, J., Pereira Rodrigues, P., Ria\u00f1o, D. (eds.) AIME 2021. LNCS (LNAI), vol. 12721, pp. 11\u201324. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-77211-6_2","DOI":"10.1007\/978-3-030-77211-6_2"},{"key":"4_CR52","doi-asserted-by":"crossref","unstructured":"Xu, P., et al.: LVLM-eHub: a comprehensive evaluation benchmark for large vision-language models. arXiv preprint arXiv:2306.09265 (2023)","DOI":"10.1109\/TPAMI.2024.3507000"},{"key":"4_CR53","unstructured":"Yin, Z., et al.: LAMM: language-assisted multi-modal instruction-tuning dataset, framework, and benchmark. In: NeurIPS, pp. 26650\u201326685 (2023)"},{"key":"4_CR54","unstructured":"Yu, W., et al.: MM-Vet: evaluating large multimodal models for integrated capabilities. arXiv preprint arXiv:2308.02490 (2023)"},{"key":"4_CR55","doi-asserted-by":"crossref","unstructured":"Yue, X., et al.: MMMU: a massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI. In: CVPR, pp. 9556\u20139567 (2024)","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"4_CR56","unstructured":"Zhang, X., et al.: PMC-VQA: visual instruction tuning for medical visual question answering. arXiv preprint arXiv:2305.10415 (2023)"},{"key":"4_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Sun, Y., Li, H., Zheng, S., Zhu, C., Yang, L.: Benchmarking the robustness of deep neural networks to common corruptions in digital pathology. In: MICCAI, pp. 242\u2013252 (2022)","DOI":"10.1007\/978-3-031-16434-7_24"},{"issue":"5","key":"4_CR58","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1038\/s42256-019-0052-1","volume":"1","author":"Z Zhang","year":"2019","unstructured":"Zhang, Z., et al.: Pathologist-level interpretable whole-slide cancer diagnosis with deep learning. Nat. Mach. Intell. 1(5), 236\u2013245 (2019)","journal-title":"Nat. Mach. Intell."},{"key":"4_CR59","doi-asserted-by":"publisher","unstructured":"Zheng, S., et al.: Benchmarking pathCLIP for pathology image analysis. J. Imaging Inform. Med. 1\u201317 (2024). https:\/\/doi.org\/10.1007\/s10278-024-01128-4","DOI":"10.1007\/s10278-024-01128-4"},{"key":"4_CR60","doi-asserted-by":"crossref","unstructured":"Zhu, C., et al.: Weakly supervised classification using multi-level instance-aware optimization on cervical cytologic image. In: 2022 IEEE 19th International Symposium on Biomedical Imaging (ISBI), pp.\u00a01\u20135. IEEE (2022)","DOI":"10.1109\/ISBI52829.2022.9761702"},{"key":"4_CR61","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: MiniGPT-4: enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592 (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73033-7_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T15:05:20Z","timestamp":1732979120000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73033-7_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031730320","9783031730337"],"references-count":61,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73033-7_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}