{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T08:02:33Z","timestamp":1784361753943,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":23,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819234288","type":"print"},{"value":"9789819234295","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3429-5_25","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:06:42Z","timestamp":1784358402000},"page":"300-311","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["HAUR: Human Annotation Understanding and Recognition Through Text-Heavy Images"],"prefix":"10.1007","author":[{"given":"Wenqing","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junjie","family":"Jiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuchen","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenyan","family":"Hao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingqiang","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"25_CR1","doi-asserted-by":"publisher","first-page":"742","DOI":"10.1007\/978-3-030-58536-5_44","volume-title":"Computer Vision \u2013 ECCV 2020. Lecture Notes in Computer Science","author":"O Sidorov","year":"2020","unstructured":"Sidorov, O., Hu, R., Rohrbach, M., Singh, A.: TextCaps: a dataset for image captioning with reading comprehension. In: Computer Vision \u2013 ECCV 2020. Lecture Notes in Computer Science, vol. 12347, pp. 742\u2013758. Springer, Cham (2020)"},{"key":"25_CR2","doi-asserted-by":"crossref","unstructured":"Singh, A. et al.: Towards VQA models that can read. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8317\u20138326 (2019)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"25_CR3","doi-asserted-by":"crossref","unstructured":"Biten, A.F., Tito, R., Mafla, A., Gomez, L., Rusinol, M., Karatzas, D.: Scene text visual question answering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 4291\u20134301 (2019)","DOI":"10.1109\/ICCV.2019.00439"},{"key":"25_CR4","doi-asserted-by":"crossref","unstructured":"Yagcioglu, S., Erdem, A., Erdem, E., Ikizler-Cinbis, N.: RecipeQA: a challenge dataset for multimodal comprehension of cooking recipes. arXiv preprint https:\/\/arxiv.org\/abs\/1809.00812 (2018)","DOI":"10.18653\/v1\/D18-1166"},{"issue":"15","key":"25_CR5","first-page":"13878","volume":"35","author":"R Tanaka","year":"2021","unstructured":"Tanaka, R., Nishida, K., Yoshida, S.: VisualMRC: machine reading comprehension on document images. Proc. AAAI Conf. Artif. Intell. 35(15), 13878\u201313888 (2021)","journal-title":"Proc. AAAI Conf. Artif. Intell"},{"key":"25_CR6","doi-asserted-by":"crossref","unstructured":"Li, J., Su, H., Zhu, J., Wang, S., Zhang, B.: Textbook question answering under instructor guidance with memory networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3655\u20133663 (2018)","DOI":"10.1109\/CVPR.2018.00385"},{"key":"25_CR7","doi-asserted-by":"crossref","unstructured":"Mishra, A., Shekhar, S., Singh, A.K., Chakraborty, A.: OCR-VQA: visual question answering by reading text in images. In: Proceedings of the International Conference on Document Analysis and Recognition (ICDAR), pp. 947\u2013952 (2019)","DOI":"10.1109\/ICDAR.2019.00156"},{"key":"25_CR8","doi-asserted-by":"crossref","unstructured":"Mathew, M., Karatzas, D., Jawahar, C.V.: DocVQA: a dataset for VQA on document images. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 2200\u20132209 (2021)","DOI":"10.1109\/WACV48630.2021.00225"},{"key":"25_CR9","doi-asserted-by":"crossref","unstructured":"Mathew, M., Karatzas, D., Jawahar, C.V.: InfographicVQA. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 1697\u20131706 (2022)","DOI":"10.1109\/WACV51458.2022.00264"},{"key":"25_CR10","unstructured":"Kim, W., Son, B., Kim, I.: ViLT: vision-and-language transformer without convolution or region supervision. In: Proceedings of the 38th International Conference on Machine Learning (ICML). Proceedings of Machine Learning Research (PMLR), pp. 5583\u20135594 (2021)"},{"key":"25_CR11","unstructured":"Chen, X. et al.: PaLI: a jointly-scaled multilingual language-image model. arXiv preprint https:\/\/arxiv.org\/abs\/2209.06794 (2022)"},{"key":"25_CR12","unstructured":"Chen, X., et al.: PaLI-3 vision language models: Smaller, faster, stronger. arXiv preprint https:\/\/arxiv.org\/abs\/2310.09199 (2023)"},{"key":"25_CR13","unstructured":"Baechler, G., et al.: ScreenAI: a vision-language model for UI and infographics understanding. arXiv preprint https:\/\/arxiv.org\/abs\/2402.04615 (2024)"},{"key":"25_CR14","unstructured":"Lee, K., Chen, X., Hua, G., Hu, H., Gao, J.: Pix2Struct: screenshot parsing as pretraining for visual language understanding. In: Proceedings of the 40th International Conference on Machine Learning (ICML). Proceedings of Machine Learning Research (PMLR), vol. 202, pp. 18893\u201318912 (2023)"},{"key":"25_CR15","unstructured":"Liu, H., et al.: LLaVA-NeXT: improved reasoning, OCR, and world knowledge. LLaVA Blog. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/ (2024). Accessed 23 May 2026"},{"key":"25_CR16","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. arXiv preprint https:\/\/arxiv.org\/abs\/2310.03744 (2023)"},{"key":"25_CR17","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Advances in Neural Information Processing Systems (NeurIPS), vol. 36, pp. 34892\u201334916 (2023)","DOI":"10.52202\/075280-1516"},{"key":"25_CR18","unstructured":"Wang, P., et al.: Qwen2-VL: enhancing vision-language model\u2019s perception of the world at any resolution. arXiv preprint https:\/\/arxiv.org\/abs\/2409.12191 (2024)"},{"key":"25_CR19","unstructured":"Bai, J., et al.: Qwen-VL: a versatile vision-language model for understanding, localization, text reading, and beyond. arXiv preprint https:\/\/arxiv.org\/abs\/2308.12966 (2023)"},{"key":"25_CR20","doi-asserted-by":"crossref","unstructured":"Wang, A., et al.: YOLOv10: real-time end-to-end object detection. arXiv preprint https:\/\/arxiv.org\/abs\/2405.14458 (2024)","DOI":"10.52202\/079017-3429"},{"key":"25_CR21","doi-asserted-by":"crossref","unstructured":"Pfitzmann, B. Auer, C., Dolfi, M., Nassar, A.S., Staar, P.W.J.: DocLayNet: a large human-annotated dataset for document-layout analysis. In: Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining (KDD), pp. 3743\u20133751. ACM, New York (2022)","DOI":"10.1145\/3534678.3539043"},{"key":"25_CR22","unstructured":"Hu, S., et al.: MiniCPM: unveiling the potential of small language models with scalable training strategies. arXiv preprint https:\/\/arxiv.org\/abs\/2404.06395 (2024)"},{"key":"25_CR23","unstructured":"Beyer, L., et al.: PaliGemma: a versatile 3B VLM for transfer. arXiv preprint https:\/\/arxiv.org\/abs\/2407.07726 (2024)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3429-5_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T07:06:44Z","timestamp":1784358404000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3429-5_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819234288","9789819234295"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3429-5_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}