{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:58:06Z","timestamp":1785488286847,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774604","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Medical Semantic-Aware Image-Text Alignment loss for Medical Visual Question Answering"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-7425-4559","authenticated-orcid":false,"given":"Vasudha","family":"Joshi","sequence":"first","affiliation":[{"name":"Computer Science and Engineering, Indian Institute of Technology Kharagpur, Kharagpur, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5895-4036","authenticated-orcid":false,"given":"Adrish","family":"Adhikari","sequence":"additional","affiliation":[{"name":"Electronics and Electrical Communication, Indian Institute of Technology Kharagpur, Kharagpur, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1908-9813","authenticated-orcid":false,"given":"Pabitra","family":"Mitra","sequence":"additional","affiliation":[{"name":"Computer Science and Engineering, Indian Institute of Technology Kharagpur, Kharagpur, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6756-1393","authenticated-orcid":false,"given":"Supratik","family":"Bose","sequence":"additional","affiliation":[{"name":"Varian Medical Systems, San Ramon, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_2_2_2","volume-title":"CLEF (working notes)","author":"Abacha Asma\u00a0Ben","year":"2018","unstructured":"Asma\u00a0Ben Abacha, Soumya Gayen, Jason\u00a0J Lau, Sivaramakrishnan Rajaraman, and Dina Demner-Fushman. 2018. NLM at ImageCLEF 2018 visual question answering in the medical domain.. In CLEF (working notes)."},{"key":"e_1_3_3_2_3_2","series-title":"(CEUR Workshop Proceedings)","volume-title":"Working Notes of CLEF 2019 - Conference and Labs of the Evaluation Forum, Lugano, Switzerland, September 9-12, 2019","author":"Abacha Asma\u00a0Ben","year":"2019","unstructured":"Asma\u00a0Ben Abacha, Sadid\u00a0A. Hasan, Vivek\u00a0V. Datla, Joey Liu, Dina Demner-Fushman, and Henning M\u00fcller. 2019. VQA-Med: Overview of the Medical Visual Question Answering Task at ImageCLEF 2019. In Working Notes of CLEF 2019 - Conference and Labs of the Evaluation Forum, Lugano, Switzerland, September 9-12, 2019(CEUR Workshop Proceedings)."},{"key":"e_1_3_3_2_4_2","volume-title":"Working Notes of CLEF 2019 - Conference and Labs of the Evaluation Forum, Lugano, Switzerland, September 9-12, 2019","author":"Allaouzi Imane","year":"2019","unstructured":"Imane Allaouzi, Mohamed\u00a0Ben Ahmed, and Badr Benamrou. 2019. An Encoder-Decoder Model for Visual Question Answering in the Medical Domain. In Working Notes of CLEF 2019 - Conference and Labs of the Evaluation Forum, Lugano, Switzerland, September 9-12, 2019. CEUR-WS.org."},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Olivier Bodenreider. 2004. The unified medical language system (UMLS): integrating biomedical terminology. Nucleic acids research (2004).","DOI":"10.1093\/nar\/gkh061"},{"key":"e_1_3_3_2_7_2","unstructured":"Xinlei Chen Haoqi Fan Ross Girshick and Kaiming He. 2020. Improved baselines with momentum contrastive learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2003.04297 (2020)."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-16443-9_65"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00359"},{"key":"e_1_3_3_2_10_2","unstructured":"Marco Cuturi. 2013. Sinkhorn distances: Lightspeed computation of optimal transport. Advances in neural information processing systems (2013)."},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87240-3_7"},{"key":"e_1_3_3_2_12_2","volume-title":"9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In 9th International Conference on Learning Representations, ICLR 2021, Virtual Event, Austria, May 3-7, 2021."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-eacl.88"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3460426.3463584"},{"key":"e_1_3_3_2_15_2","unstructured":"Sadid\u00a0A Hasan Yuan Ling Oladimeji Farri Joey Liu Henning M\u00fcller and Matthew Lungren. 2018. Overview of imageclef 2018 medical domain visual question answering task. Proceedings of CLEF 2018 Working Notes (2018)."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long Short-Term Memory. Neural Comput. (1997).","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Alistair\u00a0EW Johnson Tom\u00a0J Pollard Nathaniel\u00a0R Greenbaum Matthew\u00a0P Lungren Chih-ying Deng Yifan Peng Zhiyong Lu Roger\u00a0G Mark Seth\u00a0J Berkowitz and Steven Horng. 2019. MIMIC-CXR-JPG a large publicly available database of labeled chest radiographs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1901.07042 (2019).","DOI":"10.1038\/s41597-019-0322-0"},{"key":"e_1_3_3_2_19_2","first-page":"2","volume-title":"Proceedings of naacL-HLT","volume":"1","author":"Kenton Jacob Devlin Ming-Wei\u00a0Chang","year":"2019","unstructured":"Jacob Devlin Ming-Wei\u00a0Chang Kenton and Lee\u00a0Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of naacL-HLT , Vol.\u00a01. 2."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISBI48211.2021.9434063"},{"key":"e_1_3_3_2_21_2","unstructured":"Jin-Hwa Kim Jaehyun Jun and Byoung-Tak Zhang. 2018. Bilinear attention networks. Advances in neural information processing systems (2018)."},{"key":"e_1_3_3_2_22_2","unstructured":"Tomasz Kornuta Deepta Rajan Chaitanya Shivade Alexis Asseman and Ahmet\u00a0S Ozcan. 2019. Leveraging medical visual question answering with supporting facts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1905.12008 (2019)."},{"key":"e_1_3_3_2_23_2","unstructured":"Jason\u00a0J Lau Soumya Gayen Asma Ben\u00a0Abacha and Dina Demner-Fushman. 2018. A dataset of clinically generated visual questions and answers about radiology images. Scientific data (2018)."},{"key":"e_1_3_3_2_24_2","unstructured":"Junnan Li Ramprasaath Selvaraju Akhilesh Gotmare Shafiq Joty Caiming Xiong and Steven Chu\u00a0Hong Hoi. 2021. Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems (2021)."},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-43907-0_36"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87196-3_20"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISBI48211.2021.9434010"},{"key":"e_1_3_3_2_28_2","volume-title":"International Conference on Learning Representations","author":"Loshchilov Ilya","year":"2019","unstructured":"Ilya Loshchilov and Frank Hutter. 2019. Decoupled Weight Decay Regularization. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Bkg6RiCqY7"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-21735-7_7"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"crossref","unstructured":"Mark Neumann Daniel King Iz Beltagy and Waleed Ammar. 2019. ScispaCy: fast and robust models for biomedical natural language processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1902.07669 (2019).","DOI":"10.18653\/v1\/W19-5034"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-32251-9_57"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01364-6_20"},{"key":"e_1_3_3_2_33_2","volume-title":"CEUR Workshop Proceedings","author":"R\u00fcckert Johannes","year":"2022","unstructured":"Johannes R\u00fcckert, Asma Ben\u00a0Abacha, Alba Garc\u00eda Seco\u00a0de Herrera, Louise Bloch, Raphael Br\u00fcngel, Ahmad Idrissi-Yaghir, Henning Sch\u00e4fer, Henning M\u00fcller, and Christoph\u00a0M Friedrich. 2022. Overview of ImageCLEFmedical 2022\u2013caption prediction and concept detection. In CEUR Workshop Proceedings. CEUR Workshop Proceedings."},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.74"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.74"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Dhruv Sharma Sanjay Purushotham and Chandan\u00a0K Reddy. 2021. MedFuseNet: An attention-based multimodal deep learning model for visual question answering in the medical domain. Scientific Reports (2021).","DOI":"10.1038\/s41598-021-98390-1"},{"key":"e_1_3_3_2_37_2","volume-title":"CLEF (working notes)","author":"Shi Lei","year":"2019","unstructured":"Lei Shi, Feifan Liu, and Max\u00a0P Rosen. 2019. Deep Multimodal Learning for Medical Visual Question Answering.. In CLEF (working notes)."},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"crossref","unstructured":"Sanjay Subramanian Lucy\u00a0Lu Wang Sachin Mehta Ben Bogin Madeleine Van\u00a0Zuylen Sravanthi Parasa Sameer Singh Matt Gardner and Hannaneh Hajishirzi. 2020. Medicat: A dataset of medical images captions and textual references. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.06000 (2020).","DOI":"10.18653\/v1\/2020.findings-emnlp.191"},{"key":"e_1_3_3_2_39_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_3_2_40_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems (2017)."},{"key":"e_1_3_3_2_41_2","unstructured":"Minh\u00a0H Vu Tommy L\u00f6fstedt Tufve Nyholm and Raphael Sznitman. 2020. A question-centric model for visual question answering in medical imaging. IEEE transactions on medical imaging (2020)."},{"key":"e_1_3_3_2_42_2","volume-title":"CLEF 2019-Conference and Labs of the Evaluation Forum, Lugano, Switzerland, Sept 9-12, 2019","author":"Vu Minh\u00a0Hoang","year":"2019","unstructured":"Minh\u00a0Hoang Vu, Raphael Sznitman, Tufve Nyholm, and Tommy L\u00f6fstedt. 2019. Ensemble of streamlined bilinear visual question answering models for the imageclef 2019 challenge in the medical domain. In CLEF 2019-Conference and Labs of the Evaluation Forum, Lugano, Switzerland, Sept 9-12, 2019."},{"key":"e_1_3_3_2_43_2","unstructured":"Risto Vuorio Shao-Hua Sun Hexiang Hu and Joseph\u00a0J Lim. 2019. Multimodal model-agnostic meta-learning via task-aware modulation. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.369"},{"key":"e_1_3_3_2_45_2","unstructured":"Xin Yan Lin Li Chulin Xie Jun Xiao and Lin Gu. 2019. Zhejiang University at ImageCLEF 2019 Visual Question Answering in the Medical Domain. CLEF (Working Notes) (2019)."},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.10"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.202"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774604","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:07:53Z","timestamp":1785485273000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774604"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":46,"alternative-id":["10.1145\/3774521.3774604","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774604","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}