{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T11:18:37Z","timestamp":1784719117780,"version":"3.55.0"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:00:00Z","timestamp":1727913600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:00:00Z","timestamp":1727913600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["BMC Med Inform Decis Mak"],"DOI":"10.1186\/s12911-024-02677-y","type":"journal-article","created":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:01:47Z","timestamp":1727913707000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":20,"title":["Validation of large language models for detecting pathologic complete response in breast cancer using population-based pathology reports"],"prefix":"10.1186","volume":"24","author":[{"given":"Ken","family":"Cheligeer","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guosong","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alison","family":"Laws","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"May Lynn","family":"Quan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andrea","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Anne-Marie","family":"Brisson","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jason","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuan","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,3]]},"reference":[{"key":"2677_CR1","doi-asserted-by":"publisher","first-page":"1441","DOI":"10.1245\/s10434-015-4404-8","volume":"22","author":"P Cortazar","year":"2015","unstructured":"Cortazar P, Geyer CE. Pathological complete response in neoadjuvant treatment of breast cancer. Ann Surg Oncol. 2015;22:1441\u20136.","journal-title":"Ann Surg Oncol"},{"key":"2677_CR2","doi-asserted-by":"publisher","first-page":"1425","DOI":"10.1245\/s10434-015-4406-6","volume":"22","author":"EP Mamounas","year":"2015","unstructured":"Mamounas EP. Impact of neoadjuvant chemotherapy on locoregional surgical treatment of breast cancer. Ann Surg Oncol. 2015;22:1425\u201333.","journal-title":"Ann Surg Oncol"},{"key":"2677_CR3","doi-asserted-by":"publisher","first-page":"164","DOI":"10.1016\/S0140-6736(13)62422-8","volume":"384","author":"P Cortazar","year":"2014","unstructured":"Cortazar P, et al. Pathological complete response and long-term clinical benefit in breast cancer: the CTNeoBC pooled analysis. Lancet. 2014;384:164\u201372.","journal-title":"The Lancet"},{"key":"2677_CR4","doi-asserted-by":"publisher","first-page":"10","DOI":"10.1093\/annonc\/mdv507","volume":"27","author":"E Korn","year":"2016","unstructured":"Korn E, Sachs M, McShane L. Statistical controversies in clinical research: assessing pathologic complete response as a trial-level surrogate endpoint for early-stage breast cancer. Ann Oncol. 2016;27:10\u20135.","journal-title":"Ann Oncol"},{"key":"2677_CR5","doi-asserted-by":"publisher","first-page":"E206","DOI":"10.1136\/amiajnl-2013-002428","volume":"20","author":"J Pathak","year":"2013","unstructured":"Pathak J, Kho AN, Denny JC. Electronic health records-driven phenotyping: challenges, recent advances, and perspectives. J Am Med Inform Assn. 2013;20:E206\u201311. https:\/\/doi.org\/10.1136\/amiajnl-2013-002428.","journal-title":"J Am Med Inform Assn"},{"key":"2677_CR6","doi-asserted-by":"publisher","unstructured":"Wu G, Cheligeer C, Brisson AM, Quan ML, Cheung WY, Brenner D, et al. A new method of identifying pathologic complete response after neoadjuvant chemotherapy for breast cancer patients using a population-based electronic medical record system. Ann Surg Oncol. 2023;30(4):2095\u2013103. https:\/\/doi.org\/10.1245\/s10434-022-12955-6.","DOI":"10.1245\/s10434-022-12955-6"},{"key":"2677_CR7","doi-asserted-by":"publisher","first-page":"160","DOI":"10.1007\/s42979-021-00592-x","volume":"2","author":"IH Sarker","year":"2021","unstructured":"Sarker IH. Machine learning: algorithms, real-world applications and research directions. SN Comput Sci. 2021;2:160. https:\/\/doi.org\/10.1007\/s42979-021-00592-x.","journal-title":"SN Comput Sci"},{"key":"2677_CR8","doi-asserted-by":"publisher","first-page":"607","DOI":"10.1093\/jamia\/ocw144","volume":"24","author":"N Garcelon","year":"2017","unstructured":"Garcelon N, Neuraz A, Benoit V, Salomon R, Burgun A. Improving a full-text search engine: the importance of negation detection and family history context to identify cases in a biomedical data warehouse. J Am Med Inform Assoc. 2017;24:607\u201313. https:\/\/doi.org\/10.1093\/jamia\/ocw144.","journal-title":"J Am Med Inform Assoc"},{"key":"2677_CR9","doi-asserted-by":"publisher","first-page":"e12239","DOI":"10.2196\/12239","volume":"7","author":"S Sheikhalishahi","year":"2019","unstructured":"Sheikhalishahi S, et al. Natural language processing of clinical notes on chronic diseases: systematic review. JMIR Med Inform. 2019;7:e12239. https:\/\/doi.org\/10.2196\/12239.","journal-title":"JMIR Med Inform"},{"issue":"5","key":"2677_CR10","doi-asserted-by":"publisher","first-page":"986","DOI":"10.1093\/jamia\/ocx039","volume":"24","author":"DS Carrell","year":"2017","unstructured":"Carrell DS, Schoen RE, Leffler DA, Morris M, Rose S, Baer A, et al. Challenges in adapting existing clinical natural language processing systems to multiple, diverse health care settings. J Am Med Inform Assoc. 2017;24(5):986\u201391.","journal-title":"J Am Med Inform Assoc"},{"key":"2677_CR11","doi-asserted-by":"crossref","unstructured":"Perera S, Sheth A, Thirunarayan K, Nair S, Shah N. Challenges in understanding clinical notes: Why NLP engines fall short and where background knowledge can help. In Proceedings of the 2013 international workshop on Data management & analytics for healthcare; 2013. p. 21\u20136.","DOI":"10.1145\/2512410.2512427"},{"key":"2677_CR12","doi-asserted-by":"publisher","first-page":"520","DOI":"10.1111\/jep.13541","volume":"27","author":"S van Baalen","year":"2021","unstructured":"van Baalen S, Boon M, Verhoef P. From clinical decision support to clinical reasoning support systems. J Eval Clin Pract. 2021;27:520\u20138. https:\/\/doi.org\/10.1111\/jep.13541.","journal-title":"J Eval Clin Pract"},{"key":"2677_CR13","doi-asserted-by":"publisher","first-page":"1036","DOI":"10.1093\/jamia\/ocae005","volume":"31","author":"WQ Wei","year":"2024","unstructured":"Wei WQ, et al. Improving reporting standards for phenotyping algorithm in biomedical research: 5 fundamental dimensions. J Am Med Inform Assn. 2024;31:1036\u201341. https:\/\/doi.org\/10.1093\/jamia\/ocae005.","journal-title":"J Am Med Inform Assn"},{"key":"2677_CR14","doi-asserted-by":"publisher","first-page":"1930","DOI":"10.1038\/s41591-023-02448-8","volume":"29","author":"AJ Thirunavukarasu","year":"2023","unstructured":"Thirunavukarasu AJ, et al. Large language models in medicine. Nat Med. 2023;29:1930\u201340.","journal-title":"Nat Med"},{"key":"2677_CR15","doi-asserted-by":"publisher","first-page":"100338","DOI":"10.1016\/j.jpi.2023.100338","volume":"14","author":"SN Hart","year":"2023","unstructured":"Hart SN, et al. Organizational preparedness for the use of large language models in pathology informatics. J Pathol Inform. 2023;14:100338.","journal-title":"J Pathol Inform"},{"key":"2677_CR16","first-page":"4171","volume-title":"2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Naacl Hlt 2019)","author":"J Devlin","year":"2019","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K. Bert: pre-training of deep bidirectional transformers for language understanding. In: 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Naacl Hlt 2019), vol. 1. 2019. p. 4171\u201386."},{"key":"2677_CR17","volume-title":"Improving language understanding by generative pre-training","author":"A Radford","year":"2018","unstructured":"Radford A, Narasimhan K, Salimans T, Sutskever I. Improving language understanding by generative pre-training. 2018."},{"key":"2677_CR18","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1258\/000456303766476986","volume":"40","author":"PM Bossuyt","year":"2003","unstructured":"Bossuyt PM, et al. Towards complete and accurate reporting of studies of diagnostic accuracy: the STARD initiative. Ann Clin Biochem. 2003;40:357\u201363. https:\/\/doi.org\/10.1258\/000456303766476986.","journal-title":"Ann Clin Biochem"},{"key":"2677_CR19","doi-asserted-by":"crossref","unstructured":"Lewis M, et al. Bart: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461. 2019.","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"2677_CR20","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel C, et al. Exploring the limits of transfer learning with a unified text-to-text transformer. J Mach Learn Res. 2020;21:1\u201367.","journal-title":"J Mach Learn Res"},{"key":"2677_CR21","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford A, et al. Language models are unsupervised multitask learners. OpenAI blog. 2019;1:9.","journal-title":"OpenAI blog"},{"key":"2677_CR22","doi-asserted-by":"publisher","first-page":"e48995","DOI":"10.2196\/48995","volume":"12","author":"C Cheligeer","year":"2024","unstructured":"Cheligeer C, et al. BERT-based neural network for inpatient fall detection from electronic medical records: retrospective cohort study. JMIR Med Inform. 2024;12:e48995. https:\/\/doi.org\/10.2196\/48995.","journal-title":"JMIR Med Inform"},{"key":"2677_CR23","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1186\/s12874-022-01665-y","volume":"22","author":"HX Lu","year":"2022","unstructured":"Lu HX, Ehwerhemuepha L, Rakovski C. A comparative study on deep learning models for text classification of unstructured medical notes with various levels of class imbalance. Bmc Med Res Methodol. 2022;22:181. https:\/\/doi.org\/10.1186\/s12874-022-01665-y.","journal-title":"Bmc Med Res Methodol"},{"key":"2677_CR24","unstructured":"Snoek J, Larochelle H, Adams RP. Practical Bayesian optimization of machine learning algorithms. Adv Neural Inf Process Syst. 2012;25. https:\/\/proceedings.neurips.cc\/paper\/2012\/file\/05311655a15b75fab86956663e1819cd-Paper.pdf."},{"key":"2677_CR25","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1613\/jair.953","volume":"16","author":"NV Chawla","year":"2002","unstructured":"Chawla NV, Bowyer KW, Hall LO, Kegelmeyer WP. SMOTE: synthetic minority over-sampling technique. J Artif Intell Res. 2002;16:321\u201357. https:\/\/doi.org\/10.1613\/jair.953.","journal-title":"J Artif Intell Res"},{"key":"2677_CR26","unstructured":"Glorot X, Bengio Y. Understanding the difficulty of training deep feedforward neural networks. In Proceedings of the thirteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings; 2010. p. 249\u201356."},{"key":"2677_CR27","unstructured":"Hu EJ, Shen Y, Wallis P, Allen-Zhu Z, Li Y, Wang S, et al. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685. 2021."},{"key":"2677_CR28","unstructured":"Kingma DP. Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980. 2014."},{"key":"2677_CR29","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1038\/s41586-020-2649-2","volume":"585","author":"CR Harris","year":"2020","unstructured":"Harris CR, et al. Array programming with NumPy. Nature. 2020;585:357\u201362. https:\/\/doi.org\/10.1038\/s41586-020-2649-2.","journal-title":"Nature"},{"key":"2677_CR30","doi-asserted-by":"publisher","first-page":"352","DOI":"10.1038\/s41592-020-0772-5","volume":"17","author":"P Virtanen","year":"2020","unstructured":"Virtanen P, et al. SciPy 1.0: fundamental algorithms for scientific computing in Python (vol 33, pg 219, 2020). Nat Methods. 2020;17:352\u2013352. https:\/\/doi.org\/10.1038\/s41592-020-0772-5.","journal-title":"Nat Methods"},{"key":"2677_CR31","unstructured":"Paszke A, et al. PyTorch: an imperative style, high-performance deep learning library. Adv Neur In. 2019;32."},{"key":"2677_CR32","unstructured":"Sanh V. DistilBERT, A distilled version of BERT: smaller, faster, cheaper and lighter. arXiv preprint arXiv:1910.01108. 2019."},{"key":"2677_CR33","doi-asserted-by":"crossref","unstructured":"Alsentzer E, Murphy JR, Boag W, Weng WH, Jin D, Naumann T, et al. Publicly available clinical BERT embeddings. arXiv preprint arXiv:1904.03323. 2019.","DOI":"10.18653\/v1\/W19-1909"},{"key":"2677_CR34","doi-asserted-by":"crossref","unstructured":"Jiao X, Yin Y, Shang L, Jiang X, Chen X, Li L, et al. Tinybert: distilling BERT for natural language understanding. arXiv preprint arXiv:1909.10351. 2019.","DOI":"10.18653\/v1\/2020.findings-emnlp.372"},{"issue":"1","key":"2677_CR35","doi-asserted-by":"publisher","first-page":"194","DOI":"10.1038\/s41746-022-00742-2","volume":"5","author":"X Yang","year":"2022","unstructured":"Yang X, Chen A, PourNejatian N, Shin HC, Smith KE, Parisien C, et al. A large language model for electronic health records. NPJ Digit Med. 2022;5(1):194.","journal-title":"NPJ Digit Med"},{"issue":"70","key":"2677_CR36","first-page":"1","volume":"25","author":"HW Chung","year":"2024","unstructured":"Chung HW, Hou L, Longpre S, Zoph B, Tay Y, Fedus W, et al. Scaling instruction-finetuned language models. J Mach Learn Res. 2024;25(70):1\u201353.","journal-title":"J Mach Learn Res"},{"key":"2677_CR37","doi-asserted-by":"publisher","first-page":"1233","DOI":"10.1056\/NEJMsr2214184","volume":"388","author":"P Lee","year":"2023","unstructured":"Lee P, Bubeck S, Petro J. Benefits, limits, and risks of GPT-4 as an AI chatbot for medicine. N Engl J Med. 2023;388:1233\u20139.","journal-title":"N Engl J Med"},{"key":"2677_CR38","doi-asserted-by":"publisher","first-page":"12176","DOI":"10.1038\/ncomms12176","volume":"7","author":"P Ramkumar","year":"2016","unstructured":"Ramkumar P, et al. Chunking as the result of an efficiency computation trade-off. Nat Commun. 2016;7:12176.","journal-title":"Nat Commun"}],"container-title":["BMC Medical Informatics and Decision Making"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s12911-024-02677-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s12911-024-02677-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s12911-024-02677-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:02:24Z","timestamp":1727913744000},"score":1,"resource":{"primary":{"URL":"https:\/\/bmcmedinformdecismak.biomedcentral.com\/articles\/10.1186\/s12911-024-02677-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,3]]},"references-count":38,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2024,12]]}},"alternative-id":["2677"],"URL":"https:\/\/doi.org\/10.1186\/s12911-024-02677-y","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-4004164\/v1","asserted-by":"object"}]},"ISSN":["1472-6947"],"issn-type":[{"value":"1472-6947","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,3]]},"assertion":[{"value":"1 March 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This study received approval from the Health Research Ethics Board of Alberta \u2013 Cancer Committee, with a waiver of informed consent granted.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable as this manuscript does not contain any individual person\u2019s data in any form that could be used to identify them.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The authors declare no competing interests.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"283"}}