{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T16:55:03Z","timestamp":1779382503221,"version":"3.53.1"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2021ZD0113303"],"award-info":[{"award-number":["2021ZD0113303"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62432006"],"award-info":[{"award-number":["62432006"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276159"],"award-info":[{"award-number":["62276159"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Fundamental Research Program of Shanxi Province","award":["202303021223004"],"award-info":[{"award-number":["202303021223004"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Multimedia"],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/tmm.2025.3590930","type":"journal-article","created":{"date-parts":[[2025,7,23]],"date-time":"2025-07-23T18:44:14Z","timestamp":1753296254000},"page":"6780-6792","source":"Crossref","is-referenced-by-count":1,"title":["Multi-Grained Vision-and-Language Model for Medical Image and Text Alignment"],"prefix":"10.1109","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-8342-7605","authenticated-orcid":false,"given":"Huimin","family":"Yan","sequence":"first","affiliation":[{"name":"Institute of Intelligent Information Processing, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1496-8923","authenticated-orcid":false,"given":"Xian","family":"Yang","sequence":"additional","affiliation":[{"name":"Alliance Manchester Business School, The University of Manchester, Manchester, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0380-2995","authenticated-orcid":false,"given":"Liang","family":"Bai","sequence":"additional","affiliation":[{"name":"Institute of Intelligent Information Processing, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1663-9438","authenticated-orcid":false,"given":"Jiamin","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Intelligent Information Processing, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5887-9327","authenticated-orcid":false,"given":"Jiye","family":"Liang","sequence":"additional","affiliation":[{"name":"Institute of Intelligent Information Processing, Shanxi University, Taiyuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3338769"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448302"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3358411"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TBME.2023.3331305"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ISBI56570.2024.10635357"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3273924"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/JBHI.2024.3414413"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_1"},{"key":"ref10","first-page":"2","article-title":"Contrastive learning of medical visual representations from paired images and text","volume-title":"Proc. Mach. Learn. Healthcare Conf.","author":"Zhang","year":"2022"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.256"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00391"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3406897"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2024.3369968"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-022-30761-2"},{"key":"ref16","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2021"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2019.2896100"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.469"},{"key":"ref21","first-page":"13","article-title":"ViLBERT: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Lu","year":"2019"},{"key":"ref22","article-title":"Pixel-BERT: Aligning image pixels with text by deep multi-modal transformers","author":"Huang","year":"2020"},{"key":"ref23","first-page":"5583","article-title":"VILT: Vision-and-language transformer without convolution or region supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Kim","year":"2021"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3401031"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3280734"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3325965"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3235495"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3174413"},{"key":"ref30","article-title":"Beit: Bert pre-training of image transformers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bao","year":"2022"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16431"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01763"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"ref36","first-page":"91","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Ren","year":"2015"},{"key":"ref37","article-title":"Multimodal masked autoencoders learn transferable representations","author":"Geng","year":"2022"},{"key":"ref38","article-title":"Masked vision and language modeling for multi-modal representation learning","author":"Kwon","year":"2022"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.3301590"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-019-0322-0"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1016\/j.compbiomed.2021.104319"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1148\/ryai.2019180041"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-1909"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72390-2_44"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01954"},{"key":"ref48","first-page":"33536","article-title":"Multi-granularity cross-modal alignment for generalized medical visual representation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Wang","year":"2022"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.74"}],"container-title":["IEEE Transactions on Multimedia"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6046\/10844992\/11091540.pdf?arnumber=11091540","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T14:28:53Z","timestamp":1759242533000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11091540\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":49,"URL":"https:\/\/doi.org\/10.1109\/tmm.2025.3590930","relation":{},"ISSN":["1520-9210","1941-0077"],"issn-type":[{"value":"1520-9210","type":"print"},{"value":"1941-0077","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]}}}