{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:57:43Z","timestamp":1785488263012,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Data Sciences Institute (DSI) Seed Funding for Methodologists Grant","award":["0000"],"award-info":[{"award-number":["0000"]}]},{"name":"NSERC Alliance International Catalyst Grant","award":["0000"],"award-info":[{"award-number":["0000"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774546","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-8","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["ChestGPT: Integrating Large Language Models and Vision Transformers for Disease Detection and Localization in Chest X-Rays"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1195-4999","authenticated-orcid":false,"given":"Shehroz S.","family":"Khan","sequence":"first","affiliation":[{"name":"College of Engineering and Technology, American University of the Middle East, Kuwait, Egaila, 54200, Kuwait"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5326-8404","authenticated-orcid":false,"given":"Petar","family":"Przulj","sequence":"additional","affiliation":[{"name":"University of Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0463-4102","authenticated-orcid":false,"given":"Ahmed","family":"Ashraf","sequence":"additional","affiliation":[{"name":"University of Manitoba, Winnipeg, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7393-1362","authenticated-orcid":false,"given":"Ali","family":"Abedi","sequence":"additional","affiliation":[{"name":"University of Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Yasmeena Akhter Richa Singh and Mayank Vatsa. 2023. AI-based radiodiagnosis using chest X-rays: A review. Frontiers in Big Data 6 (2023) 1120989.","DOI":"10.3389\/fdata.2023.1120989"},{"key":"e_1_3_3_2_3_2","unstructured":"Asma Alkhaldi Raneem Alnajim Layan Alabdullatef Rawan Alyahya Jun Chen Deyao Zhu Ahmed Alsinan and Mohamed Elhoseiny. 2024. Minigpt-med: Large language model as a general interface for radiology diagnosis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.04106 (2024)."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Daniel\u00a0J Cao Casey Hurrell and Michael\u00a0N Patlas. 2023. Current status of burnout in Canadian radiology. Canadian Association of Radiologists Journal 74 1 (2023) 37\u201343.","DOI":"10.1177\/08465371221117282"},{"key":"e_1_3_3_2_5_2","unstructured":"Jun Chen Deyao Zhu Xiaoqian Shen Xiang Li Zechun Liu Pengchuan Zhang Raghuraman Krishnamoorthi Vikas Chandra Yunyang Xiong and Mohamed Elhoseiny. 2023. Minigpt-v2: large language model as a unified interface for vision-language multi-task learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.09478 (2023)."},{"key":"e_1_3_3_2_6_2","unstructured":"Yueru Chen and C-C\u00a0Jay Kuo. 2019. PixelHop: A successive subspace learning (SSL) method for object classification. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1909.08190 (2019)."},{"key":"e_1_3_3_2_7_2","unstructured":"Wei-Lin Chiang Zhuohan Li Zi Lin Ying Sheng Zhanghao Wu Hao Zhang Lianmin Zheng Siyuan Zhuang Yonghao Zhuang Joseph\u00a0E Gonzalez et\u00a0al. 2023. Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality March 2023. URL https:\/\/lmsys.org\/blog\/2023-03-30-vicuna 3 5 (2023)."},{"key":"e_1_3_3_2_8_2","unstructured":"Yeongjae Cho Taehee Kim Heejun Shin Sungzoon Cho and Dongmyung Shin. 2024. Pretraining Vision-Language Model for Difference Visual Question Answering in Longitudinal Chest X-rays. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.08966 (2024)."},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Kai Han Yunhe Wang Hanting Chen Xinghao Chen Jianyuan Guo Zhenhua Liu Yehui Tang An Xiao Chunjing Xu Yixing Xu et\u00a0al. 2022. A survey on vision transformer. IEEE transactions on pattern analysis and machine intelligence 45 1 (2022) 87\u2013110.","DOI":"10.1109\/TPAMI.2022.3152247"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"crossref","unstructured":"Ahmed Hosny Chintan Parmar John Quackenbush Lawrence\u00a0H Schwartz and Hugo\u00a0JWL Aerts. 2018. Artificial intelligence in radiology. Nature Reviews Cancer 18 8 (2018) 500\u2013510.","DOI":"10.1038\/s41568-018-0016-5"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599819"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.3301590"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Alistair\u00a0EW Johnson Tom\u00a0J Pollard Seth\u00a0J Berkowitz Nathaniel\u00a0R Greenbaum Matthew\u00a0P Lungren Chih-ying Deng Roger\u00a0G Mark and Steven Horng. 2019. MIMIC-CXR a de-identified publicly available database of chest radiographs with free-text reports. Scientific data 6 1 (2019) 317.","DOI":"10.1038\/s41597-019-0322-0"},{"key":"e_1_3_3_2_15_2","first-page":"1493","volume-title":"Medical Imaging with Deep Learning","author":"Keicher Matthias","year":"2024","unstructured":"Matthias Keicher, Kamilia Zaripova, Tobias Czempiel, Kristina Mach, Ashkan Khakzar, and Nassir Navab. 2024. FlexR: few-shot classification with language embeddings for structured reporting of chest x-rays. In Medical Imaging with Deep Learning. PMLR, 1493\u20131508."},{"key":"e_1_3_3_2_16_2","unstructured":"Hyungyung Lee Da\u00a0Young Lee Wonjae Kim Jin-Hwa Kim Tackeun Kim Jihang Kim Leonard Sunwoo and Edward Choi. 2023. UniXGen: A Unified Vision-Language Model for Multi-View Chest X-ray Generation and Report Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.12172 (2023)."},{"key":"e_1_3_3_2_17_2","unstructured":"Hongzhao Li Hongyu Wang Xia Sun Hua He and Jun Feng. 2024. Prompt-Guided Generation of Structured Chest X-Ray Report Using a Pre-trained LLM. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.11209 (2024)."},{"key":"e_1_3_3_2_18_2","unstructured":"Liunian\u00a0Harold Li Mark Yatskar Da Yin Cho-Jui Hsieh and Kai-Wei Chang. 2019. Visualbert: A simple and performant baseline for vision and language. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1908.03557 (2019)."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"crossref","unstructured":"Matthew Limb. 2022. Shortages of radiology and oncology staff putting cancer patients at risk college warns.","DOI":"10.1136\/bmj.o1430"},{"key":"e_1_3_3_2_20_2","first-page":"74","volume-title":"Text summarization branches out","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74\u201381."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Masoud Monajatipoor Mozhdeh Rouhsedaghat Liunian\u00a0Harold Li Aichi Chien C-C\u00a0Jay Kuo Fabien Scalzo and Kai-Wei Chang. 2021. Berthop: An effective vision-and-language model for chest x-ray disease diagnosis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2108.04938 (2021).","DOI":"10.1109\/ICCVW54120.2021.00372"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Subhash Nerella Sabyasachi Bandyopadhyay Jiaqing Zhang Miguel Contreras Scott Siegel Aysegul Bumin Brandon Silva Jessica Sena Benjamin Shickel Azra Bihorac et\u00a0al. 2024. Transformers and large language models in healthcare: A review. Artificial Intelligence in Medicine (2024) 102900.","DOI":"10.1016\/j.artmed.2024.102900"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"crossref","unstructured":"Ha\u00a0Q Nguyen Khanh Lam Linh\u00a0T Le Hieu\u00a0H Pham Dat\u00a0Q Tran Dung\u00a0B Nguyen Dung\u00a0D Le Chi\u00a0M Pham Hang\u00a0TT Tong Diep\u00a0H Dinh et\u00a0al. 2022. VinDr-CXR: An open dataset of chest X-rays with radiologist\u2019s annotations. Scientific Data 9 1 (2022) 429.","DOI":"10.1038\/s41597-022-01498-w"},{"key":"e_1_3_3_2_24_2","first-page":"311","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318."},{"key":"e_1_3_3_2_25_2","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_26_2","unstructured":"Alec Radford Karthik Narasimhan Tim Salimans Ilya Sutskever et\u00a0al. 2018. Improving language understanding by generative pre-training. (2018)."},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-95-0568-5_5"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.bionlp-1.35"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"crossref","unstructured":"Arun\u00a0James Thirunavukarasu Darren Shu\u00a0Jeng Ting Kabilan Elangovan Laura Gutierrez Ting\u00a0Fang Tan and Daniel Shu\u00a0Wei Ting. 2023. Large language models in medicine. Nature medicine 29 8 (2023) 1930\u20131940.","DOI":"10.1038\/s41591-023-02448-8"},{"key":"e_1_3_3_2_30_2","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et\u00a0al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09288 (2023)."},{"key":"e_1_3_3_2_31_2","unstructured":"Xiaosong Wang Ziyue Xu Leo Tam Dong Yang and Daguang Xu. 2021. Self-supervised image-text pre-training with mixed data in chest x-rays. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2103.16022 (2021)."},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"crossref","unstructured":"Zifeng Wang Zhenbang Wu Dinesh Agarwal and Jimeng Sun. 2022. Medclip: Contrastive learning from unpaired medical images and text. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.10163 (2022).","DOI":"10.18653\/v1\/2022.emnlp-main.256"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.aacl-srw.11"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"crossref","unstructured":"Linghua Wu Jing Zhang Yilin Wang Rong Ding Yueqin Cao Guiqin Liu Changsheng Liufu Baowei Xie Shanping Kang Rui Liu et\u00a0al. 2024. Pneumonia detection based on RSNA dataset and anchor-free deep learning detector. Scientific Reports 14 1 (2024) 1929.","DOI":"10.1038\/s41598-024-52156-7"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"crossref","unstructured":"An Yan Julian McAuley Xing Lu Jiang Du Eric\u00a0Y Chang Amilcare Gentili and Chun-Nan Hsu. 2022. RadBERT: adapting transformer-based language models to radiology. Radiology: Artificial Intelligence 4 4 (2022) e210258.","DOI":"10.1148\/ryai.210258"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774546","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:01:41Z","timestamp":1785484901000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774546"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":34,"alternative-id":["10.1145\/3774521.3774546","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774546","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}