{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,11]],"date-time":"2026-03-11T07:44:57Z","timestamp":1773215097289,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,9,16]]},"DOI":"10.1145\/3774816.3774838","type":"proceedings-article","created":{"date-parts":[[2026,2,23]],"date-time":"2026-02-23T10:48:11Z","timestamp":1771843691000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Semantic Encoding in Medical LLMs for Vocabulary Standardisation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-6919-1239","authenticated-orcid":false,"given":"Samuel Thomas","family":"Mainwood","sequence":"first","affiliation":[{"name":"Data Science and AI(DSAI), School of Computing Technologies, RMIT University, Melbourne, Victoria, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1519-2736","authenticated-orcid":false,"given":"Aashish","family":"Bhandari","sequence":"additional","affiliation":[{"name":"Data Science and AI(DSAI), School of Computing Technologies, RMIT University, Melbourne, Victoria, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0181-6258","authenticated-orcid":false,"given":"Sonika","family":"Tyagi","sequence":"additional","affiliation":[{"name":"Data Science and AI(DSAI), School of Computing Technologies, RMIT University, Melbourne, Vic, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,2,23]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Emily Alsentzer John\u00a0R. Murphy Willie Boag Wei-Hung Weng Di Jin Tristan Naumann and Matthew B.\u00a0A. McDermott. 2019. Publicly Available Clinical BERT Embeddings. http:\/\/arxiv.org\/abs\/1904.03323 arXiv:https:\/\/arXiv.org\/abs\/1904.03323 [cs]."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-47240-422"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-78952-63"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","unstructured":"O. Bodenreider. 2004. The Unified Medical Language System (UMLS): integrating biomedical terminology. Nucleic Acids Research 32 90001 (Jan. 2004) 267D\u2013270. 10.1093\/nar\/gkh061","DOI":"10.1093\/nar\/gkh061"},{"key":"e_1_3_3_2_6_2","unstructured":"Elliot Bolton Abhinav Venigalla Michihiro Yasunaga David Hall Betty Xiong Tony Lee Roxana Daneshjou Jonathan Frankle Percy Liang Michael Carbin and Christopher\u00a0D Manning. 2024. BioMedLM: A 2.7B Parameter Language Model Trained On Biomedical Text. (March 2024). https:\/\/doi.org\/arXiv:2403.18421v1"},{"key":"e_1_3_3_2_7_2","unstructured":"Aakanksha Chowdhery Sharan Narang Jacob Devlin Maarten Bosma Gaurav Mishra Adam Roberts Paul Barham Hyung\u00a0Won Chung Charles Sutton Sebastian Gehrmann et\u00a0al. 2023. Palm: Scaling language modeling with pathways. Journal of Machine Learning Research 24 240 (2023) 1\u2013113."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","unstructured":"Cl\u00e9ment Christophe Praveen\u00a0K. Kanithi Tathagata Raha Shadab Khan and Marco\u00a0AF Pimentel. 2024. Med42-v2: A Suite of Clinical LLMs. 10.48550\/arXiv.2408.06142arXiv:https:\/\/arXiv.org\/abs\/2408.06142 [cs].","DOI":"10.48550\/arXiv.2408.06142"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","unstructured":"Nick Craswell. 2009. Mean Reciprical Rank. Encyclopedia of Database Systems (2009). 10.1007\/978-0-387-39940-9488","DOI":"10.1007\/978-0-387-39940-9488"},{"key":"e_1_3_3_2_10_2","volume-title":"Prompt Engineering-The Ultimate Guide for Success in Artificial Intelligence","author":"Deshmukh Jayant","year":"2024","unstructured":"Jayant Deshmukh. 2024. Prompt Engineering-The Ultimate Guide for Success in Artificial Intelligence. Jayant Deshmukh."},{"key":"e_1_3_3_2_11_2","unstructured":"Jacob Devlin Ming-Wei Chang Kenton Lee and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. http:\/\/arxiv.org\/abs\/1810.04805 arXiv:https:\/\/arXiv.org\/abs\/1810.04805 [cs]."},{"key":"e_1_3_3_2_12_2","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et\u00a0al. 2024. The llama 3 herd of models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024)."},{"key":"e_1_3_3_2_13_2","unstructured":"Yuan He Jiaoyan Chen Denvar Antonyrajah and Ian Horrocks. 2022. BERTMap: A BERT-based Ontology Alignment System. http:\/\/arxiv.org\/abs\/2112.02682 arXiv:https:\/\/arXiv.org\/abs\/2112.02682 [cs]."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Yuan He Jiaoyan Chen Hang Dong Ian Horrocks Carlo Allocca Taehun Kim and Brahmananda Sapkota. 2024. DeepOnto: A Python Package for Ontology Engineering with Deep Learning. 10.48550\/arXiv.2307.03067arXiv:https:\/\/arXiv.org\/abs\/2307.03067 [cs].","DOI":"10.48550\/arXiv.2307.03067"},{"key":"e_1_3_3_2_15_2","unstructured":"Yuan He Jiaoyan Chen Hang Dong Ernesto Jim\u00e9nez-Ruiz Ali Hadian and Ian Horrocks. 2023. Machine Learning-Friendly Biomedical Datasets for Equivalence and Subsumption Ontology Matching. http:\/\/arxiv.org\/abs\/2205.03447 arXiv:https:\/\/arXiv.org\/abs\/2205.03447 [cs q-bio]."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3587259.3627571"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","unstructured":"Andreas Holzinger Chris Biemann Constantinos\u00a0S. Pattichis and Douglas\u00a0B. Kell. 2017. What do we need to build explainable AI systems for the medical domain?10.48550\/arXiv.1712.09923arXiv:https:\/\/arXiv.org\/abs\/1712.09923 [cs].","DOI":"10.48550\/arXiv.1712.09923"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-25073-618"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","unstructured":"Renren Jin Jiangcun Du Wuwei Huang Wei Liu Jian Luan Bin Wang and Deyi Xiong. 2024. A Comprehensive Evaluation of Quantization Strategies for Large Language Models. 10.48550\/arXiv.2402.16775arXiv:https:\/\/arXiv.org\/abs\/2402.16775 [cs].","DOI":"10.48550\/arXiv.2402.16775"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Jeff Johnson Matthijs Douze and Herv\u00e9 J\u00e9gou. 2019. Billion-scale similarity search with GPUs. IEEE Transactions on Big Data 7 3 (2019) 535\u2013547.","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","unstructured":"Bogumil\u00a0M. Konopka. 2015. Biomedical ontologies\u2014A review. Biocybernetics and Biomedical Engineering 35 2 (2015) 75\u201386. 10.1016\/j.bbe.2014.06.002","DOI":"10.1016\/j.bbe.2014.06.002"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","unstructured":"Zeljko Kraljevic Thomas Searle Anthony Shek Lukasz Roguski Kawsar Noor Daniel Bean Aurelie Mascio Leilei Zhu Amos\u00a0A. Folarin Angus Roberts Rebecca Bendayan Mark\u00a0P. Richardson Robert Stewart Anoop\u00a0D. Shah Wai\u00a0Keong Wong Zina Ibrahim James\u00a0T. Teo and Richard\u00a0J.B. Dobson. 2021. Multi-domain clinical natural language processing with MedCAT: The Medical Concept Annotation Toolkit. Artificial Intelligence in Medicine 117 (July 2021) 102083. 10.1016\/j.artmed.2021.102083","DOI":"10.1016\/j.artmed.2021.102083"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","unstructured":"Jinhyuk Lee Wonjin Yoon Sungdong Kim Donghyeon Kim Sunkyu Kim Chan\u00a0Ho So and Jaewoo Kang. 2020. BioBERT: a pre-trained biomedical language representation model for biomedical text mining. Bioinformatics 36 4 (Feb. 2020) 1234\u20131240. 10.1093\/bioinformatics\/btz682","DOI":"10.1093\/bioinformatics\/btz682"},{"key":"e_1_3_3_2_24_2","unstructured":"Eric Lehman Evan Hernandez Diwakar Mahajan Jonas Wulff Micah\u00a0J Smith Zachary Ziegler Daniel Nadler Peter Szolovits Alistair Johnson and Emily Alsentzer. 2023. Do We Still Need Clinical Language Models? PhysioNet (Feb. 2023). https:\/\/doi.org\/arXiv:2302.08091v1"},{"key":"e_1_3_3_2_25_2","unstructured":"Eric Lehman and Alistair Johnson. 2023. Clinical-t5: Large language models built using mimic clinical text. PhysioNet 101 (2023) 215\u2013220."},{"key":"e_1_3_3_2_26_2","unstructured":"Fangyu Liu Ehsan Shareghi Zaiqiao Meng Marco Basaldella and Nigel Collier. 2021. Self-Alignment Pretraining for Biomedical Entity Representations. http:\/\/arxiv.org\/abs\/2010.11784 arXiv:https:\/\/arXiv.org\/abs\/2010.11784 [cs]."},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-emnlp.398"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","unstructured":"Alejandro Metke-Jimenez Jim Steel David Hansen and Michael Lawley. 2018. Ontoserver: a syndicated terminology server. J Biomed Semant 9 1 (Dec. 2018) 24. 10.1186\/s13326-018-0191-z","DOI":"10.1186\/s13326-018-0191-z"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.139"},{"key":"e_1_3_3_2_30_2","unstructured":"OAEI. 2024. Ontology Alignment Evaluation Intiative. http:\/\/oaei.ontologymatching.org\/2024\/"},{"key":"e_1_3_3_2_31_2","first-page":"104","volume-title":"Proceedings of the 19th International Workshop on Ontology Matching co-located with the 23rd International Semantic Web Conference (ISWC 2024), Baltimore, USA","volume":"3897","author":"Oulefki Samira","year":"2024","unstructured":"Samira Oulefki, Lamia Berkani, Ladjel Bellatreche, Nassim Boudjenah, and Aicha Mokhtari. 2024. Results for BioGITOM in OAEI 2024. In Proceedings of the 19th International Workshop on Ontology Matching co-located with the 23rd International Semantic Web Conference (ISWC 2024), Baltimore, USA , Vol.\u00a03897. 104\u2013109."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","unstructured":"Vaclav Papez Maxim Moinat Erica\u00a0A Voss Sofia Bazakou Anne Van\u00a0Winzum Alessia Peviani Stefan Payralbe Elena\u00a0Garcia Lara Michael Kallfelz Folkert\u00a0W Asselbergs Daniel Prieto-Alhambra Richard J\u00a0B Dobson and Spiros Denaxas. 2022. Transforming and evaluating the UK Biobank to the OMOP Common Data Model for COVID-19 research and beyond. Journal of the American Medical Informatics Association 30 1 (Dec. 2022) 103\u2013111. 10.1093\/jamia\/ocac203","DOI":"10.1093\/jamia\/ocac203"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","unstructured":"Yashpal Ramakrishnaiah Nenad Macesic Geoffrey\u00a0I. Webb Anton\u00a0Y. Peleg and Sonika Tyagi. 2023. EHR-QC: A streamlined pipeline for automated electronic health records standardisation and preprocessing to predict clinical outcomes. Journal of Biomedical Informatics 147 (Nov. 2023) 104509. 10.1016\/j.jbi.2023.104509","DOI":"10.1016\/j.jbi.2023.104509"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","unstructured":"Tabinda Sarwar Sattar Seifollahi Jeffrey Chan Xiuzhen Zhang Vural Aksakalli Irene Hudson Karin Verspoor and Lawrence Cavedon. 2023. The Secondary Use of Electronic Health Records for Data Mining: Data Characteristics and Challenges. ACM Comput. Surv. 55 2 (Feb. 2023) 1\u201340. 10.1145\/3490234","DOI":"10.1145\/3490234"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","unstructured":"Merlijn Sevenster Rob Van\u00a0Ommering and Yuechen Qian. 2012. Algorithmic and user study of an autocompletion algorithm on a large medical vocabulary. Journal of Biomedical Informatics 45 1 (Feb. 2012) 107\u2013119. 10.1016\/j.jbi.2011.09.004","DOI":"10.1016\/j.jbi.2011.09.004"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-34586-923"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"publisher","unstructured":"Maria Taboada Diego Martinez Mohammed Arideh and Rosa Mosquera. 2025. Ontology Matching with Large Language Models and Prioritized Depth-First Search. 10.48550\/arXiv.2501.11441arXiv:https:\/\/arXiv.org\/abs\/2501.11441 [cs].","DOI":"10.48550\/arXiv.2501.11441"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","unstructured":"Navya Tyagi Naima Vahab and Sonika Tyagi. 2024. Genome Language Modeling (Glm): A Beginner\u2019s Cheat Sheet. 10.20944\/preprints202411.0285.v1","DOI":"10.20944\/preprints202411.0285.v1"},{"key":"e_1_3_3_2_39_2","first-page":"845","volume-title":"Proceedings of the AMIA Symposium","author":"Wang Amy\u00a0Y","year":"2002","unstructured":"Amy\u00a0Y Wang, Jeremiah\u00a0H Sable, and Kent\u00a0A Spackman. 2002. The SNOMED clinical terms development process: refinement and analysis of content.. In Proceedings of the AMIA Symposium. American Medical Informatics Association, 845."},{"key":"e_1_3_3_2_40_2","unstructured":"World Health Organisation. 2019. International Statistical Classification of Diseases and Related Health Problems 10th Revision. https:\/\/icd.who.int\/browse10\/2019\/en"},{"key":"e_1_3_3_2_41_2","unstructured":"World Health Organisation. 2022. International Statistical Classification of Diseases and Related Health Problems (ICD). https:\/\/www.who.int\/standards\/classifications\/classification-of-diseases"},{"key":"e_1_3_3_2_42_2","unstructured":"Qianqian Xie Qingyu Chen Aokun Chen Cheng Peng Yan Hu Fongci Lin Xueqing Peng Jimin Huang Jeffrey Zhang Vipina Keloth et\u00a0al. 2024. Me-llama: Foundation large language models for medical applications. Research square (2024) rs\u20133."},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","unstructured":"Xi Yang Nima PourNejatian Hoo\u00a0Chang Shin Kaleb\u00a0E Smith Christopher Parisien Colin Compas Cheryl Martin Mona\u00a0G Flores Ying Zhang Tanja Magoc Christopher\u00a0A Harle Gloria Lipori Duane\u00a0A Mitchell William\u00a0R Hogan Elizabeth\u00a0A Shenkman Jiang Bian and Yonghui Wu. 2022. GatorTron: A Large Language Model for Clinical Natural Language Processing. 10.1101\/2022.02.27.22271257","DOI":"10.1101\/2022.02.27.22271257"},{"key":"e_1_3_3_2_44_2","unstructured":"Hongjian Zhou Fenglin Liu Boyang Gu Xinyu Zou Jinfa Huang Jinge Wu Yiru Li Sam\u00a0S Chen Peilin Zhou Junling Liu Yining Hua Chengfeng Mao Chenyu You Xian Wu Yefeng Zheng Lei Clifton Zheng Li Jiebo Luo and David\u00a0A Clifton. 2024. A Survey of Large Language Models in Medicine: Progress Application and Challenge. Preprint (2024). https:\/\/doi.org\/arXiv.2303.18223"}],"event":{"name":"HIKM '25: Health Informatics Knowledge Management Conference 2025","location":"Online Australia","acronym":"HIKM 2025"},"container-title":["Proceedings of the 2025 18th Health Informatics Knowledge Management Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774816.3774838","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T09:50:47Z","timestamp":1773136247000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774816.3774838"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,16]]},"references-count":43,"alternative-id":["10.1145\/3774816.3774838","10.1145\/3774816"],"URL":"https:\/\/doi.org\/10.1145\/3774816.3774838","relation":{},"subject":[],"published":{"date-parts":[[2025,9,16]]},"assertion":[{"value":"2026-02-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}