{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T12:29:43Z","timestamp":1771244983892,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":42,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819569496","type":"print"},{"value":"9789819569502","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-6950-2_6","type":"book-chapter","created":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T11:59:19Z","timestamp":1771243159000},"page":"74-88","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MMDocBench: Benchmarking Large Vision-Language Models for\u00a0Fine-Grained Visual Document Understanding and\u00a0Grounding"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6776-2040","authenticated-orcid":false,"given":"Fengbin","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-5086-3886","authenticated-orcid":false,"given":"Ziyang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1704-4650","authenticated-orcid":false,"given":"NG Xiang","family":"Yao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4259-1121","authenticated-orcid":false,"given":"Haohui","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5199-1428","authenticated-orcid":false,"given":"Wenjie","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5828-9842","authenticated-orcid":false,"given":"Fuli","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7427-793X","authenticated-orcid":false,"given":"Chao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3938-119X","authenticated-orcid":false,"given":"Huanbo","family":"Luan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6097-7807","authenticated-orcid":false,"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,17]]},"reference":[{"key":"6_CR1","unstructured":"Chen, W., et\u00a0al.: Guicourse: From general vision language models to versatile gui agents. arXiv preprint arXiv:2406.11317 (2024)"},{"key":"6_CR2","doi-asserted-by":"crossref","unstructured":"Deng, C., Wu, Q., Wu, Q., Hu, F., Lyu, F., Tan, M.: Visual grounding via accumulated attention. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7746\u20137755 (2018)","DOI":"10.1109\/CVPR.2018.00808"},{"key":"6_CR3","unstructured":"Fu, C., et al.: Mme: a comprehensive evaluation benchmark for multimodal large language models (2024). https:\/\/arxiv.org\/abs\/2306.13394"},{"key":"6_CR4","doi-asserted-by":"crossref","unstructured":"Harley, A.W., Ufkes, A., Derpanis, K.G.: Evaluation of deep convolutional nets for document image classification and retrieval. In: 2015 13th International Conference on Document Analysis and Recognition (ICDAR), pp. 991\u2013995. IEEE (2015)","DOI":"10.1109\/ICDAR.2015.7333910"},{"key":"6_CR5","doi-asserted-by":"crossref","unstructured":"Huang, Z., Chen, K., He, J., Bai, X., Karatzas, D., Lu, S., Jawahar, C.V.: Icdar2019 competition on scanned receipt ocr and information extraction. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1516\u20131520 (2019)","DOI":"10.1109\/ICDAR.2019.00244"},{"key":"6_CR6","doi-asserted-by":"crossref","unstructured":"Kim, G., et al.: Ocr-free document understanding transformer. In: European Conference on Computer Vision, pp. 498\u2013517. Springer (2022)","DOI":"10.1007\/978-3-031-19815-1_29"},{"key":"6_CR7","doi-asserted-by":"crossref","unstructured":"Li, M., Xu, Y., Cui, L., Huang, S., Wei, F., Li, Z., Zhou, M.: Docbank: a benchmark dataset for document layout analysis. arXiv preprint arXiv:2006.01038 (2020)","DOI":"10.18653\/v1\/2020.coling-main.82"},{"key":"6_CR8","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pp. 740\u2013755. Springer (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"6_CR9","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Advances in Neural Information Processing Systems, pp. 34892\u201334916. Curran Associates, Inc. (2023)"},{"key":"6_CR10","doi-asserted-by":"crossref","unstructured":"Liu, Y., et\u00a0al.: Mmbench: Is your multi-modal model an all-around player? arXiv preprint arXiv:2307.06281 (2023)","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"6_CR11","unstructured":"Ma, Y., et\u00a0al.: Mmlongbench-doc: Benchmarking long-context document understanding with visualizations. arXiv preprint arXiv:2407.01523 (2024)"},{"key":"6_CR12","doi-asserted-by":"crossref","unstructured":"Masry, A., Do, X.L., Tan, J.Q., Joty, S., Hoque, E.: ChartQA: A benchmark for question answering about charts with visual and logical reasoning. In: Muresan, S., Nakov, P., Villavicencio, A. (eds.) Findings of the Association for Computational Linguistics, pp. 2263\u20132279. Association for Computational Linguistics (2022)","DOI":"10.18653\/v1\/2022.findings-acl.177"},{"key":"6_CR13","doi-asserted-by":"crossref","unstructured":"Mathew, M., Bagal, V., Tito, R., Karatzas, D., Valveny, E., Jawahar, C.: Infographicvqa. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1697\u20131706 (2022)","DOI":"10.1109\/WACV51458.2022.00264"},{"key":"6_CR14","doi-asserted-by":"crossref","unstructured":"Mathew, M., Karatzas, D., Jawahar, C.: Docvqa: A dataset for vqa on document images. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2200\u20132209 (2021)","DOI":"10.1109\/WACV48630.2021.00225"},{"issue":"7","key":"6_CR15","first-page":"3523","volume":"44","author":"S Minaee","year":"2022","unstructured":"Minaee, S., Boykov, Y., Porikli, F., Plaza, A., Kehtarnavaz, N., Terzopoulos, D.: Image segmentation using deep learning: a survey. IEEE Trans. Pattern Anal. Mach. Intell. 44(7), 3523\u20133542 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"6_CR16","doi-asserted-by":"crossref","unstructured":"Mishra, A., Shekhar, S., Singh, A.K., Chakraborty, A.: Ocr-vqa: Visual question answering by reading text in images. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 947\u2013952. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00156"},{"key":"6_CR17","unstructured":"OpenAI: Gpt-4v(ision) system card (2023)"},{"key":"6_CR18","unstructured":"Park, S., Shin, S., Lee, B., Lee, J., Surh, J., Seo, M., Lee, H.: Cord: a consolidated receipt dataset for post-ocr parsing. In: Workshop on Document Intelligence at NeurIPS 2019 (2019)"},{"key":"6_CR19","doi-asserted-by":"crossref","unstructured":"Pasupat, P., Liang, P.: Compositional semantic parsing on semi-structured tables. In: Zong, C., Strube, M. (eds.) Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 1470\u20131480. Association for Computational Linguistics (2015)","DOI":"10.3115\/v1\/P15-1142"},{"key":"6_CR20","unstructured":"Peng, Z., et al.: Grounding multimodal large language models to the world. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"6_CR21","doi-asserted-by":"crossref","unstructured":"Qu, C., Liu, C., Liu, Y., Chen, X., Peng, D., Guo, F., Jin, L.: Towards robust tampered text detection in document image: New dataset and new solution. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5937\u20135946 (2023)","DOI":"10.1109\/CVPR52729.2023.00575"},{"key":"6_CR22","doi-asserted-by":"crossref","unstructured":"\u0160imsa, \u0160., et\u00a0al.: Docile benchmark for document information localization and extraction. In: International Conference on Document Analysis and Recognition, pp. 147\u2013166. Springer (2023)","DOI":"10.1007\/978-3-031-41679-8_9"},{"key":"6_CR23","doi-asserted-by":"crossref","unstructured":"Singh, A., et al.: Towards vqa models that can read. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8317\u20138326 (2019)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"6_CR24","doi-asserted-by":"crossref","unstructured":"Singh, A., Pang, G., Toh, M., Huang, J., Galuba, W., Hassner, T.: Textocr: towards large-scale end-to-end reasoning for arbitrary-shaped scene text. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8802\u20138812 (2021)","DOI":"10.1109\/CVPR46437.2021.00869"},{"key":"6_CR25","doi-asserted-by":"crossref","unstructured":"Smock, B., Pesala, R., Abraham, R.: Pubtables-1m: Towards comprehensive table extraction from unstructured documents. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4634\u20134642 (2022)","DOI":"10.1109\/CVPR52688.2022.00459"},{"key":"6_CR26","unstructured":"Sun, H., Kuang, Z., Yue, X., Lin, C., Zhang, W.: Spatial dual-modality graph reasoning for key information extraction. arXiv preprint arXiv:2103.14470 (2021)"},{"key":"6_CR27","doi-asserted-by":"crossref","unstructured":"Tong, S., Liu, Z., Zhai, Y., Ma, Y., LeCun, Y., Xie, S.: Eyes wide shut? exploring the visual shortcomings of multimodal llms. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9568\u20139578 (2024)","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"6_CR28","doi-asserted-by":"crossref","unstructured":"Van\u00a0Landeghem, J., et\u00a0al.: Document understanding dataset and evaluation (dude). In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19528\u201319540 (2023)","DOI":"10.1109\/ICCV51070.2023.01789"},{"key":"6_CR29","unstructured":"Wang, G., Ge, Y., Ding, X., Kankanhalli, M., Shan, Y.: What makes for good visual tokenizers for large language models? arXiv preprint arXiv:2305.12223 (2023)"},{"key":"6_CR30","unstructured":"Wang, Z., et al.: Charxiv: charting gaps in realistic chart understanding in multimodal llms. arXiv preprint arXiv:2406.18521 (2024)"},{"key":"6_CR31","doi-asserted-by":"crossref","unstructured":"Xuan, S., Guo, Q., Yang, M., Zhang, S.: Pink: unveiling the power of referential comprehension for multi-modal llms. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13838\u201313848 (2024)","DOI":"10.1109\/CVPR52733.2024.01313"},{"key":"6_CR32","doi-asserted-by":"crossref","unstructured":"Yin, S., Fu, C., Zhao, S., Li, K., Sun, X., Xu, T., Chen, E.: A survey on multimodal large language models (2024). https:\/\/arxiv.org\/abs\/2306.13549","DOI":"10.1093\/nsr\/nwae403"},{"key":"6_CR33","unstructured":"Ying, K., et\u00a0al.: Mmt-bench: a comprehensive multimodal benchmark for evaluating large vision-language models towards multitask agi. arXiv preprint arXiv:2404.16006 (2024)"},{"key":"6_CR34","unstructured":"You, H., et al.: Ferret: refer and ground anything anywhere at any granularity. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"6_CR35","doi-asserted-by":"crossref","unstructured":"Yu, L., Poirson, P., Yang, S., Berg, A.C., Berg, T.L.: Modeling context in referring expressions. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Proceedings, Part II 14, pp. 69\u201385. Springer (2016)","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"6_CR36","unstructured":"Yu, W., Yang, Z., Li, L., Wang, J., Lin, K., Liu, Z., Wang, X., Wang, L.: Mm-vet: Evaluating large multimodal models for integrated capabilities. arXiv preprint arXiv:2308.02490 (2023)"},{"key":"6_CR37","unstructured":"Yuxin, W., Boqiang, Z., Hongtao, X., Yongdong, Z.: Tampered text detection via rgb and frequency relationship modeling. Chinese Journal of Network and Information Security, p.\u00a029 (2022)"},{"key":"6_CR38","doi-asserted-by":"crossref","unstructured":"Zheng, X., Burdick, D., Popa, L., Zhong, P., Wang, N.X.R.: Global table extractor (gte): A framework for joint table identification and cell structure recognition using visual context. Winter Conference for Applications in Computer Vision (WACV) (2021)","DOI":"10.1109\/WACV48630.2021.00074"},{"key":"6_CR39","doi-asserted-by":"crossref","unstructured":"Zhong, X., ShafieiBavani, E., Jimeno\u00a0Yepes, A.: Image-based table recognition: data, model, and evaluation. In: European Conference on Computer Vision, pp. 564\u2013580. Springer (2020)","DOI":"10.1007\/978-3-030-58589-1_34"},{"key":"6_CR40","doi-asserted-by":"crossref","unstructured":"Zhong, X., Tang, J., Yepes, A.J.: Publaynet: largest dataset ever for document layout analysis. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1015\u20131022. IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00166"},{"key":"6_CR41","doi-asserted-by":"crossref","unstructured":"Zhu, F., Lei, W., Feng, F., Wang, C., Zhang, H., Chua, T.S.: Towards complex document understanding by discrete reasoning. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4857\u20134866 (2022)","DOI":"10.1145\/3503161.3548422"},{"key":"6_CR42","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Groth, O., Bernstein, M., Fei-Fei, L.: Visual7w: Grounded question answering in images. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4995\u20135004 (2016)","DOI":"10.1109\/CVPR.2016.540"}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-6950-2_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T11:59:29Z","timestamp":1771243169000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-6950-2_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819569496","9789819569502"],"references-count":42,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-6950-2_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"17 February 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Prague","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Czech Republic","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 January 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31 January 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"32","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/mmm2026.cz\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}