{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,14]],"date-time":"2026-03-14T08:28:17Z","timestamp":1773476897111,"version":"3.50.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2016,1,11]],"date-time":"2016-01-11T00:00:00Z","timestamp":1452470400000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"DOI":"10.13039\/100004440","name":"Wellcome Trust (GB)","doi-asserted-by":"publisher","award":["086105"],"award-info":[{"award-number":["086105"]}],"id":[{"id":"10.13039\/100004440","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Lang Resources &amp; Evaluation"],"published-print":{"date-parts":[[2016,9]]},"DOI":"10.1007\/s10579-015-9330-7","type":"journal-article","created":{"date-parts":[[2016,1,11]],"date-time":"2016-01-11T05:51:24Z","timestamp":1452491484000},"page":"523-548","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":21,"title":["Annotating patient clinical records with syntactic chunks and named entities: the Harvey Corpus"],"prefix":"10.1007","volume":"50","author":[{"given":"Aleksandar","family":"Savkov","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John","family":"Carroll","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rob","family":"Koeling","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jackie","family":"Cassell","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,1,11]]},"reference":[{"key":"9330_CR1","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1007\/978-94-011-3474-3_10","volume-title":"Principle-based parsing: Computation and psycholinguistics","author":"S Abney","year":"1991","unstructured":"Abney, S. (1991). Parsing by chunks. In R. C. Berwick, S. P. Abney, & C. Tenny (Eds.), Principle-based parsing: Computation and psycholinguistics (pp. 257\u2013278). Dordrecht: Kluwer."},{"key":"9330_CR2","doi-asserted-by":"crossref","unstructured":"Alnazzawi, N., Thompson, P., & Ananiadou, S. (2014). Building a semantically annotated corpus for congestive heart and renal failure from clinical records and the literature. In Proceedings of the 5th international workshop on health text mining and information analysis (Louhi), (pp. 69\u201374). Association for Computational Linguistics.","DOI":"10.3115\/v1\/W14-1110"},{"key":"9330_CR3","doi-asserted-by":"crossref","first-page":"161","DOI":"10.1186\/1471-2105-13-161","volume":"13","author":"M Bada","year":"2012","unstructured":"Bada, M., Eckert, M., Evans, D., & Garcia, K., et al. (2012). Concept annotation in the craft corpus. BMC Bioinformatics, 13, 161.","journal-title":"BMC Bioinformatics"},{"key":"9330_CR4","unstructured":"Bentley, T., Price, C., & Brown, P. (1996). Structural and lexical features of successive versions of the read codes. In Proceedings of the annual conference of the primary health care specialist group of the British computer society (pp. 91\u2013103)."},{"key":"9330_CR5","unstructured":"Bharati, A., Sangal, R., Sharma, D. M., & Bai, L. (2006). Anncorra: Annotating corpora guidelines for POS and chunk annotation for Indian languages. Technical report TR-LTRC-31, LTRC, IIIT-Hyderabad."},{"key":"9330_CR6","unstructured":"Bies, A., Ferguson, M., Katz, K., MacIntyre, R., et al. (1995). Bracketing guidelines for Treebank II style Penn Treebank project. Technical report, University of Pennsylvania."},{"key":"9330_CR7","unstructured":"Boisen, S., Crystal, M., Schwartz, R. M., Stone, R., & Weischedel, R. M. (2000). Annotating resources for information extraction. In LREC European language resources association"},{"key":"9330_CR8","unstructured":"Chinchor, N. (1998). MUC-7 test scores introduction. In Proceedings of the seventh message understanding conference."},{"issue":"1","key":"9330_CR9","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1177\/001316446002000104","volume":"20","author":"J Cohen","year":"1960","unstructured":"Cohen, J. (1960). A coefficient of agreement for nominal scales. Educational and Psychological Measurement, 20(1), 37\u201346.","journal-title":"Educational and Psychological Measurement"},{"key":"9330_CR10","unstructured":"Cohen, K. B., Lanfranchi, A., Corvey, W., Baumgartner, W. A. Jr., Roeder, C., Ogren, P. V., & Palmer, M., et al. (2010). Annotation of all coreference in biomedical text: Guideline selection and adaptation. In BioTxtM 2010: 2nd Workshop on building and evaluating resources for biomedical text mining, (pp. 37\u201341)."},{"issue":"438","key":"9330_CR11","first-page":"548","volume":"92","author":"B Efron","year":"1997","unstructured":"Efron, B., & Tibshirani, R. (1997). Improvements on cross-validation: The 632+ bootstrap method. Journal of the American Statistical Association, 92(438), 548\u2013560.","journal-title":"Journal of the American Statistical Association"},{"key":"9330_CR12","unstructured":"Fan, J.-W., Prasad, R., Yabut, R. M., Loomis, R. M., Zisook, D. S., Mattison, J. E., & Huang, Y. (2011). Part-of-speech tagging for clinical text: Wall or bridge between institutions? In AMIA Annual symposium (Vol. 1, pp. 382\u2013391). AMIA."},{"issue":"6","key":"9330_CR13","first-page":"1168","volume":"20","author":"J-W Fan","year":"2013","unstructured":"Fan, J.-W., Yang, E., Jiang, M., Prasad, R., Loomis, R., & Zisook, D., et al. (2013). Research and applications: Syntactic parsing of clinical text: guideline and corpus development with handling ill-formed sentences. JAMIA, 20(6), 1168\u20131177.","journal-title":"JAMIA"},{"issue":"3","key":"9330_CR14","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1007\/s10032-007-0059-8","volume":"10","author":"J Foster","year":"2007","unstructured":"Foster, J. (2007). Treebanks gone bad: Parser evaluation and retraining using a treebank of ungrammatical sentences. International Journal on Document Analysis and Recognition, 10(3), 129\u2013145.","journal-title":"International Journal on Document Analysis and Recognition"},{"key":"9330_CR15","unstructured":"Hovy, E., Marcus, M., Palmer, M., Ramshaw, L., & Weischedel, R. (2006). Ontonotes: The 90 In Proceedings of the human language technology conference of the NAACL, companion volume: Short papers, NAACL-Short \u201906 (pp. 57\u201360). Stroudsburg, PA: Association for Computational Linguistics."},{"issue":"3","key":"9330_CR16","first-page":"296","volume":"12","author":"G Hripcsak","year":"2005","unstructured":"Hripcsak, G., & Rothschild, A. S. (2005). Technical brief: Agreement, the f-measure, and reliability in information retrieval. JAMIA, 12(3), 296\u2013298.","journal-title":"JAMIA"},{"key":"9330_CR17","unstructured":"ISO (2008). Iso dis 24617\u20131: 2008 language resource management\u2014semantic annotation framework\u2014part 1: Time and events. Technical report."},{"key":"9330_CR18","doi-asserted-by":"crossref","unstructured":"Koeling, R., Tate, A. R., & Carroll, J. A. (2011). Automatically estimating the incidence of symptoms recorded in GP free text notes. In Proceedings MIXHS 2011 (pp. 43\u201350). New York, NY: ACM.","DOI":"10.1145\/2064747.2064757"},{"key":"9330_CR19","volume-title":"Content analysis: An introduction to its methodology","author":"KH Krippendorff","year":"2003","unstructured":"Krippendorff, K. H. (2003). Content analysis: An introduction to its methodology (2nd ed.). Thousand Oaks: Sage Publications Inc.","edition":"2"},{"key":"9330_CR20","doi-asserted-by":"crossref","unstructured":"Kudo, T., & Matsumoto, Y. (2001). Chunking with support vector machines. In Proceedings of the second meeting of NACL 2001 (pp. 1\u20138). Stroudsburg, PA: ACL.","DOI":"10.3115\/1073336.1073361"},{"key":"9330_CR21","doi-asserted-by":"crossref","unstructured":"Kudo, T., & Matsumoto, Y. (2003). Fast methods for kernel-based text analysis. In Proceedings of ACL 2003 (pp. 24\u201331). Morristown, NJ: ACL.","DOI":"10.3115\/1075096.1075100"},{"issue":"2","key":"9330_CR22","first-page":"313","volume":"19","author":"MP Marcus","year":"1993","unstructured":"Marcus, M. P., Santorini, B., & Marcinkiewicz, M. A. (1993). Building a large annotated corpus of English: The Penn Treebank. Computational Linguistics, 19(2), 313\u2013330.","journal-title":"Computational Linguistics"},{"key":"9330_CR23","unstructured":"National Information Board (2014). Personalised health and care 2020: Using data and technology to transform outcomes for patients and citizens."},{"key":"9330_CR24","unstructured":"Ogren, P. V., Savova, G. K., & Chute, C. G. (2008). Constructing evaluation corpora for automated clinical named entity recognition. In LREC European Language Resources Association"},{"key":"9330_CR25","doi-asserted-by":"crossref","unstructured":"Ohta, T., Tateisi, Y., & Kim, J.-D. (2002). The GENIA corpus: an annotated research abstract corpus in molecular biology domain. In Proceedings of the second international conference on Human Language Technology Research, HLT \u201902, (pp. 82\u201386). San Francisco, CA: Morgan Kaufmann Publishers Inc.","DOI":"10.3115\/1289189.1289260"},{"key":"9330_CR26","doi-asserted-by":"crossref","unstructured":"Pakhomov, S., Coden, A., & Chute, C. (2004). Creating a test corpus of clinical notes manually tagged for part-of-speech information. In Proceedings of JNLPBA 2004 (pp. 62\u201365). Stroudsburg, PA: Association for Computational Linguistics.","DOI":"10.3115\/1567594.1567607"},{"key":"9330_CR27","doi-asserted-by":"crossref","unstructured":"Pestian, J. P., Brew, C., Matykiewicz, P., Hovermale, D. J., Johnson, N., Cohen, K. B., & Duch, W. (2007). A shared task involving multi-label classification of clinical free text. In BioNLP 2007 Proceedings, BioNLP \u201907 (pp. 97\u2013104). Stroudsburg, PA: ACL.","DOI":"10.3115\/1572392.1572411"},{"key":"9330_CR28","unstructured":"Roberts, A., Gaizauskas, R., Hepple, M., Demetriou, G., Guo, Y., & Setzer, A. (2008). Semantic Annotation of Clinical Text: The CLEF Corpus. In Proceedings of the LREC 2008 workshop on building and evaluating resources for biomedical text mining (pp. 19\u201326). Marrakech."},{"issue":"5","key":"9330_CR29","doi-asserted-by":"crossref","first-page":"950","DOI":"10.1016\/j.jbi.2008.12.013","volume":"42","author":"A Roberts","year":"2009","unstructured":"Roberts, A., Gaizauskas, R. J., Hepple, M., et al. (2009). Building a semantically annotated corpus of clinical texts. Journal of Biomedical Informatics, 42(5), 950\u2013966.","journal-title":"Journal of Biomedical Informatics"},{"key":"9330_CR30","unstructured":"Santorini, B. (1990). Part-of-speech tagging guidelines for the Penn Treebank project (3rd revision, 2nd printing). Technical report, Department of Linguistics, University of Pennsylvania, Philadelphia, PA."},{"key":"9330_CR31","doi-asserted-by":"crossref","unstructured":"Savkov, A., Carroll, J., & Cassell, J. (2014). Chunking clinical text containing non-canonical language. In BioNLP Workshop proceedings, Baltimore, USA","DOI":"10.3115\/v1\/W14-3411"},{"issue":"5","key":"9330_CR32","first-page":"507","volume":"17","author":"G Savova","year":"2010","unstructured":"Savova, G., Masanz, J., Ogren, P., Zheng, J., Sohn, S., Kipper-Schuler, K., et al. (2010). Mayo clinical text analysis and knowledge extraction system (cTAKES): Architecture, component evaluation and applications. JAMIA, 17(5), 507\u2013513.","journal-title":"JAMIA"},{"key":"9330_CR33","doi-asserted-by":"crossref","first-page":"88","DOI":"10.1186\/1472-6947-12-88","volume":"12","author":"A Shah","year":"2012","unstructured":"Shah, A., Martinez, C., & Hemingway, H. (2012). The freetext matching algorithm: A computer program to extract diagnoses and causes of death from unstructured text in electronic health records. BMC Medical Informatics and Decision Making, 12, 88.","journal-title":"BMC Medical Informatics and Decision Making"},{"key":"9330_CR34","unstructured":"Stenetorp, P., Pyysalo, S., Topi\u0107, G., Ohta, T., Ananiadou, S., & Tsujii, J. (2012). Brat: A Web-based Tool for NLP-Assisted Text Annotation. In Proceedings of the demonstrations at EACL (pp. 102\u2013107). ACL."},{"key":"9330_CR35","doi-asserted-by":"crossref","first-page":"5","DOI":"10.1016\/j.jbi.2013.07.004","volume":"46","author":"W Sun","year":"2013","unstructured":"Sun, W., Rumshisky, A., & Uzuner, \u00d6. (2013). Annotating temporal information in clinical narratives. Journal of Biomedical Informatics, 46, 5\u201312.","journal-title":"Journal of Biomedical Informatics"},{"key":"9330_CR36","doi-asserted-by":"crossref","unstructured":"Tanabe, L., Xie, N., Thom, L., Matten, W., & Wilbur, W.J. (2005). GENETAG: a tagged corpus for gene\/protein named entity recognition. BMC Bioinformatics, 6(S-1).","DOI":"10.1186\/1471-2105-6-S1-S3"},{"issue":"8","key":"9330_CR37","doi-asserted-by":"crossref","first-page":"1124","DOI":"10.1093\/bioinformatics\/18.8.1124","volume":"18","author":"LK Tanabe","year":"2002","unstructured":"Tanabe, L. K., & Wilbur, W. J. (2002). Tagging gene and protein names in biomedical text. Bioinformatics, 18(8), 1124\u20131132.","journal-title":"Bioinformatics"},{"key":"9330_CR38","doi-asserted-by":"crossref","unstructured":"Tjong Kim Sang, E.F., & Buchholz, S. (2000). Introduction to the conll-2000 shared task: Chunking. ConLL \u201900 (pp. 127\u2013132). Stroudsburg, PA: Association for Computational Linguistics.","DOI":"10.3115\/1117601.1117631"},{"issue":"4","key":"9330_CR39","first-page":"561","volume":"16","author":"\u00d6 Uzuner","year":"2009","unstructured":"Uzuner, \u00d6. (2009). Recognising obesity and comorbidities in sparse data. JAMIA, 16(4), 561\u2013570.","journal-title":"JAMIA"},{"key":"9330_CR40","doi-asserted-by":"crossref","unstructured":"Uzuner, \u00d6., Goldstein, I., Luo, Y., & Kohane, I. (2007a). Identifying patient smoking status from medical discharge records. JAMIA.","DOI":"10.1197\/jamia.M2408"},{"issue":"5","key":"9330_CR41","first-page":"550","volume":"14","author":"\u00d6 Uzuner","year":"2007","unstructured":"Uzuner, \u00d6., Luo, Y., & Szolovits, P. (2007b). Evaluating the state-of-the-art in automatic de-identification. JAMIA, 14(5), 550\u2013563.","journal-title":"JAMIA"},{"issue":"5","key":"9330_CR42","first-page":"514","volume":"17","author":"\u00d6 Uzuner","year":"2010","unstructured":"Uzuner, \u00d6., Solti, I., & Cadag, E. (2010a). Extracting medication information from clinical text. JAMIA, 17(5), 514\u2013518.","journal-title":"JAMIA"},{"issue":"5","key":"9330_CR43","first-page":"519","volume":"17","author":"\u00d6 Uzuner","year":"2010","unstructured":"Uzuner, \u00d6., Solti, I., Xia, F., & Cadag, E. (2010b). Community annotation experiment for ground truth generation for the i2b2 medication challenge. JAMIA, 17(5), 519\u2013523.","journal-title":"JAMIA"},{"issue":"5","key":"9330_CR44","first-page":"552","volume":"18","author":"\u00d6 Uzuner","year":"2011","unstructured":"Uzuner, \u00d6., South, B. R., Shen, S., & DuVall, S. L. (2011). 2010 i2b2\/va challenge on concepts, assertions, and relations in clinical text. JAMIA, 18(5), 552\u2013556.","journal-title":"JAMIA"},{"key":"9330_CR45","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1186\/1471-2105-13-207","volume":"13","author":"K Verspoor","year":"2012","unstructured":"Verspoor, K., Cohen, K. B., & Lanfranchi, A., et al. (2012). A corpus of full-text journal articles is a robust evaluation tool for revealing differences in performance of biomedical natural language processing tools. BMC Bioinformatics, 13, 207.","journal-title":"BMC Bioinformatics"},{"key":"9330_CR46","unstructured":"Voorhees, E. M., & Hersh, W. (2012). Overview of the TREC 2012 medical records track. In TREC 2012 Proceedings."},{"key":"9330_CR47","unstructured":"Warner, C., Bies, A., Brisson, C., & Mott, J. (2004). Addendum to the penn treebank ii style bracketing guidelines: Biomedical treebank annotation. Technical report."}],"container-title":["Language Resources and Evaluation"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-015-9330-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10579-015-9330-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-015-9330-7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,3]],"date-time":"2019-09-03T07:05:48Z","timestamp":1567494348000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10579-015-9330-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,1,11]]},"references-count":47,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2016,9]]}},"alternative-id":["9330"],"URL":"https:\/\/doi.org\/10.1007\/s10579-015-9330-7","relation":{},"ISSN":["1574-020X","1574-0218"],"issn-type":[{"value":"1574-020X","type":"print"},{"value":"1574-0218","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,1,11]]}}}