{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T07:07:44Z","timestamp":1783840064240,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":28,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819228669","type":"print"},{"value":"9789819228645","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T00:00:00Z","timestamp":1783900800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T00:00:00Z","timestamp":1783900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-2864-5_12","type":"book-chapter","created":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T06:20:53Z","timestamp":1783837253000},"page":"131-141","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MedStruct-S: A Benchmark for\u00a0Key Discovery, Key-Conditioned QA and\u00a0Semi-structured Extraction from\u00a0OCR Clinical Reports"],"prefix":"10.1007","author":[{"given":"Yingyun","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haiyang","family":"Qian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,13]]},"reference":[{"key":"12_CR1","unstructured":"Baidu AI Cloud: Baidu ocr technical documentation (2025). https:\/\/ai.baidu.com\/tech\/ocr. Accessed 2025"},{"key":"12_CR2","doi-asserted-by":"publisher","unstructured":"Bhattacharyya, A., Tripathi, A., Das, U., Karmakar, A., Pathak, A., Gupta, M.: Information extraction from visually rich documents using LLM-based organization of documents into independent textual segments, pp. 17241\u201317256. Association for Computational Linguistics (2025). https:\/\/doi.org\/10.18653\/v1\/2025.acl-long.844","DOI":"10.18653\/v1\/2025.acl-long.844"},{"key":"12_CR3","doi-asserted-by":"publisher","unstructured":"Chen, W., et al.: A benchmark for automatic medical consultation system: frameworks, tasks and datasets. Bioinformatics 39(1), btac817 (2022). https:\/\/doi.org\/10.1093\/bioinformatics\/btac817","DOI":"10.1093\/bioinformatics\/btac817"},{"key":"12_CR4","doi-asserted-by":"crossref","unstructured":"Cui, Y., Che, W., Liu, T., Qin, B., Wang, S., Hu, G.: Revisiting pre-trained models for Chinese natural language processing. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 657\u2013668 (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.58"},{"key":"12_CR5","doi-asserted-by":"publisher","first-page":"3504","DOI":"10.1109\/TASLP.2021.3124365","volume":"29","author":"Y Cui","year":"2021","unstructured":"Cui, Y., Che, W., Liu, T., Qin, B., Yang, Z.: Pre-training with whole word masking for Chinese bert. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3504\u20133514 (2021)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"12_CR6","doi-asserted-by":"publisher","unstructured":"Dai, Z., Wang, X., Ni, P., Li, Y., Li, G., Bai, X.: Named entity recognition using bert bilstm crf for Chinese electronic health records, pp. 1\u20135 (2019). https:\/\/doi.org\/10.1109\/CISP-BMEI48845.2019.8965823","DOI":"10.1109\/CISP-BMEI48845.2019.8965823"},{"key":"12_CR7","unstructured":"Devlin, J., Chang, M., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. CoRR abs\/1810.04805 (2018). http:\/\/arxiv.org\/abs\/1810.04805"},{"key":"12_CR8","doi-asserted-by":"publisher","unstructured":"Duan, Y., et al.: Docopilot: improving multimodal models for document-level understanding. In: 2025 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4026\u20134037 (2025). https:\/\/doi.org\/10.1109\/CVPR52734.2025.00381","DOI":"10.1109\/CVPR52734.2025.00381"},{"key":"12_CR9","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbi.2020.103526","volume":"109","author":"S Fu","year":"2020","unstructured":"Fu, S., et al.: Clinical concept extraction: a methodology review. J. Biomed. Inform. 109, 103526 (2020). https:\/\/doi.org\/10.1016\/j.jbi.2020.103526","journal-title":"J. Biomed. Inform."},{"key":"12_CR10","unstructured":"Group, A.: Antangelmed: a large-scale medical moe model. Hugging Face Repository (2025)"},{"key":"12_CR11","doi-asserted-by":"publisher","unstructured":"Guan, T., Zan, H., Zhou, X., Xu, H., Zhang, K.: CMeIE: construction and evaluation of Chinese medical information extraction dataset. In: Zhu, X., Zhang, M., Hong, Y., He, R. (eds.), Natural Language Processing and Chinese Computing. NLPCC 2020. LNCS, vol. 12430, pp. 270\u2013282. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-60450-9_22","DOI":"10.1007\/978-3-030-60450-9_22"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Jaume, G., Ekenel, H.K., Thiran, J.P.: Funsd: a dataset for form understanding in noisy scanned documents. In: Accepted to ICDAR-OST (2019)","DOI":"10.1109\/ICDARW.2019.10029"},{"key":"12_CR13","unstructured":"Jain, S., et al.: Radgraph: extracting clinical entities and relations from radiology reports. arXiv preprint arXiv:2106.14463 (2021)"},{"key":"12_CR14","unstructured":"Liu, Y., et al.: Roberta: a robustly optimized bert pretraining approach (2019). https:\/\/arxiv.org\/abs\/1907.11692"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Lu, Y., et al.: Unified structure generation for universal information extraction. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 5755\u20135772 (2022)","DOI":"10.18653\/v1\/2022.acl-long.395"},{"key":"12_CR16","unstructured":"Ouyang, L., et al.: Omnidocbench: benchmarking diverse pdf document parsing with comprehensive annotations (2024). https:\/\/arxiv.org\/abs\/2412.07626"},{"key":"12_CR17","doi-asserted-by":"publisher","unstructured":"Tanwar, E., Dutta, S., Borthakur, M., Chakraborty, T.: Multilingual LLMs are better cross-lingual in-context learners with alignment. In: Rogers, A., Boyd-Graber, J., Okazaki, N. (eds.) Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 6292\u20136307. Association for Computational Linguistics, Toronto, Canada (2023). https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.346, https:\/\/aclanthology.org\/2023.acl-long.346\/","DOI":"10.18653\/v1\/2023.acl-long.346"},{"key":"12_CR18","unstructured":"Team, B.M.: Baichuan-m2: scaling medical capability with large verifier system. arXiv preprint arXiv:2501.00000 (2025)"},{"key":"12_CR19","unstructured":"Team, Q.: Qwen2.5 technical report. arXiv preprint arXiv:2412.15115 (2024)"},{"key":"12_CR20","unstructured":"Tkachenko, M., Malyuk, M., Holmanyuk, A., Liubimov, N.: Label Studio: data labeling software (2020\u20132025). https:\/\/github.com\/HumanSignal\/label-studio, open source software available from https:\/\/github.com\/HumanSignal\/label-studio"},{"key":"12_CR21","unstructured":"Wang, X., et al.: Instructuie: multi-task instruction tuning for unified information extraction. arxiv 2023. arXiv preprint arXiv:2304.08085 (2023)"},{"key":"12_CR22","doi-asserted-by":"publisher","unstructured":"Wolf, T., et al.: Transformers: state-of-the-art natural language processing. In: Liu, Q., Schlangen, D. (eds.) Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, pp. 38\u201345. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.emnlp-demos.6, https:\/\/aclanthology.org\/2020.emnlp-demos.6\/","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"12_CR23","unstructured":"Xu, Z., et al.: Mc-bert: efficient language pre-training via a meta controller (2020). https:\/\/arxiv.org\/abs\/2006.05744"},{"key":"12_CR24","unstructured":"Yang, A., et al.: Qwen3 technical report. ArXiv abs\/2505.09388 (2025), https:\/\/api.semanticscholar.org\/CorpusID:278602855"},{"key":"12_CR25","unstructured":"Yang, X., Zhao, X., Shen, Z.: Ehrstruct: a comprehensive benchmark framework for evaluating large language models on structured electronic health record tasks. ArXiv abs\/2511.08206 (2025). https:\/\/api.semanticscholar.org\/CorpusID:282922202"},{"key":"12_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, N., et al.: CBLUE: a Chinese biomedical language understanding evaluation benchmark. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 7888\u20137915 (2022)","DOI":"10.18653\/v1\/2022.acl-long.544"},{"key":"12_CR27","unstructured":"Zhang, N., et al.: Cblue benchmark: technical report. arXiv preprint arXiv:2106.08087 (2021)"},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Zhu, W., Hou, G., Chen, M., Zhang, N.: Promptcblue: a Chinese prompt tuning benchmark for the medical domain. arXiv preprint arXiv:2310.14151 (2023)","DOI":"10.2139\/ssrn.4685921"}],"container-title":["Lecture Notes in Computer Science","Knowledge Science, Engineering and Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-2864-5_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T06:20:55Z","timestamp":1783837255000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-2864-5_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,13]]},"ISBN":["9789819228669","9789819228645"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-2864-5_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,13]]},"assertion":[{"value":"13 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"KSEM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Knowledge Science, Engineering and Management","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Beijing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ksem2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ksem2026.rosc.org.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}