{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:53:07Z","timestamp":1782762787591,"version":"3.54.5"},"reference-count":21,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100003052","name":"Ministry of Trade, Industry and Energy","doi-asserted-by":"crossref","award":["SG20240201"],"award-info":[{"award-number":["SG20240201"]}],"id":[{"id":"10.13039\/501100003052","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Institute of Information & Communications Technology Planning & Evaluation","award":["IITP-2026-RS-2024-00436773"],"award-info":[{"award-number":["IITP-2026-RS-2024-00436773"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/access.2026.3705442","type":"journal-article","created":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T19:57:22Z","timestamp":1782158242000},"page":"94148-94166","source":"Crossref","is-referenced-by-count":0,"title":["Beyond Simple Character Recognition: A Comparative Study of Vision-Language Models and Dedicated OCR Systems in Edge Cases"],"prefix":"10.1109","volume":"14","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-0865-1162","authenticated-orcid":false,"given":"Juran","family":"Kim","sequence":"first","affiliation":[{"name":"Google Infra Team, Sortech, Busan, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chang-Young","family":"Lee","sequence":"additional","affiliation":[{"name":"Google AI Team, Sortech, Busan, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chang-Yup","family":"Han","sequence":"additional","affiliation":[{"name":"CEO, Sortech, Busan, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2161-9251","authenticated-orcid":false,"given":"Nam-Hyun","family":"Yoo","sequence":"additional","affiliation":[{"name":"Department of Computer Engineering, Kyungnam University, Changwon, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7756-0263","authenticated-orcid":false,"given":"Jinhong","family":"Yang","sequence":"additional","affiliation":[{"name":"Department of Medical Information Technology, Inje University, Gimhae, Republic of Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2646371"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i11.26538"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2007.4376991"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6812"},{"key":"ref5","article-title":"PP-OCRv3: More attempts for the improvement of ultra lightweight OCR system","author":"Li","year":"2022","journal-title":"arXiv:2206.03001"},{"key":"ref6","volume-title":"EasyOCR: Ready-to-Use OCR With 80+ Supported Languages","year":"2020"},{"key":"ref7","article-title":"General OCR theory: Towards OCR-2.0 via a unified end-to-end model","author":"Wei","year":"2024","journal-title":"arXiv:2409.01704"},{"key":"ref8","volume-title":"Qwen3-VL: Open-Source Vision-Language Models","year":"2025"},{"key":"ref9","article-title":"InternVL3.5: Advancing open-source multimodal models in versatility, reasoning, and efficiency","author":"Wang","year":"2025","journal-title":"arXiv:2508.18265"},{"key":"ref10","volume-title":"DeepSeek-OCR-2","year":"2026"},{"key":"ref11","volume-title":"Chandra","year":"2025"},{"key":"ref12","first-page":"1","article-title":"DocRobust: Enhancing robustness of multi-modal LLMs in low-quality document image scenarios","volume-title":"ICLR 2026 Conf. Withdrawn Submission, OpenReview","author":"Zhou"},{"key":"ref13","article-title":"OCRVerse: Towards holistic OCR in end-to-end vision-language models","author":"Zhong","year":"2026","journal-title":"arXiv:2601.21639"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4235-6"},{"key":"ref15","article-title":"Hallucination of multimodal large language models: A survey","author":"Bai","year":"2024","journal-title":"arXiv:2404.18930"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01363"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2013.221"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/icdar.2019.00244"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.5244\/C.26.127"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2015.7333942"},{"key":"ref21","article-title":"Appropriate statistics for ordinal level data: Should we really be using t-test and Cohen\u2019s d for evaluating group differences on the NSSE and other surveys","author":"Romano"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/11323511\/11571805.pdf?arnumber=11571805","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T19:41:37Z","timestamp":1782762097000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11571805\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":21,"URL":"https:\/\/doi.org\/10.1109\/access.2026.3705442","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}