{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:57:25Z","timestamp":1776887845138,"version":"3.51.2"},"publisher-location":"Cham","reference-count":42,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031730030","type":"print"},{"value":"9783031730047","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73004-7_8","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T17:02:14Z","timestamp":1730394134000},"page":"123-139","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Unified Medical Image Pre-training in\u00a0Language-Guided Common Semantic Space"],"prefix":"10.1007","author":[{"given":"Xiaoxuan","family":"He","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifan","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinyang","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xufang","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haoji","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siyun","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dongsheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuqing","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lili","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"8_CR1","doi-asserted-by":"crossref","unstructured":"Alsentzer, E., et al.: Publicly available clinical BERT embeddings. arXiv preprint arXiv:1904.03323 (2019)","DOI":"10.18653\/v1\/W19-1909"},{"issue":"2","key":"8_CR2","doi-asserted-by":"publisher","first-page":"915","DOI":"10.1118\/1.3528204","volume":"38","author":"SG Armato III","year":"2011","unstructured":"Armato, S.G., III., et al.: The lung image database consortium (LIDC) and image database resource initiative (IDRI): a completed reference database of lung nodules on CT scans. Med. Phys. 38(2), 915\u2013931 (2011)","journal-title":"Med. Phys."},{"key":"8_CR3","doi-asserted-by":"crossref","unstructured":"Bannur, S., et\u00a0al.: Learning to exploit temporal structure for biomedical vision-language processing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15016\u201315027 (2023)","DOI":"10.1109\/CVPR52729.2023.01442"},{"key":"8_CR4","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-031-20059-5_1","volume-title":"ECCV 2022","author":"B Boecking","year":"2022","unstructured":"Boecking, B., et al.: Making the most of text semantics to improve biomedical vision-language processing. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13696, pp. 1\u201321. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20059-5_1"},{"key":"8_CR5","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Chen, X., Xie, S., He, K.: An empirical study of training self-supervised vision transformers (2021)","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"8_CR7","unstructured":"Chen, Y., Liu, C., Huang, W., Cheng, S., Arcucci, R., Xiong, Z.: Generative text-guided 3D vision-language pretraining for unified medical image segmentation. arXiv preprint arXiv:2306.04811 (2023)"},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Cheng, P., Lin, L., Lyu, J., Huang, Y., Luo, W., Tang, X.: Prior: prototype representation joint learning from medical images and reports. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 21361\u201321371 (2023)","DOI":"10.1109\/ICCV51070.2023.01953"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10578\u201310587 (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Dong, X., et\u00a0al.: MaskCLIP: masked self-distillation advances contrastive language-image pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10995\u201311005 (2023)","DOI":"10.1109\/CVPR52729.2023.01058"},{"key":"8_CR11","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth $$16 \\times 16$$ words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Engilberge, M., Chevallier, L., P\u00e9rez, P., Cord, M.: Finding beans in burgers: deep semantic-visual embedding with localization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3984\u20133993 (2018)","DOI":"10.1109\/CVPR.2018.00419"},{"key":"8_CR13","unstructured":"Faghri, F., Fleet, D.J., Kiros, J.R., Fidler, S.: VSE++: improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612 (2017)"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"He, X., et al.: Automated model design and benchmarking of deep learning models for Covid-19 detection with chest CT scans. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a035, pp. 4821\u20134829 (2021)","DOI":"10.1609\/aaai.v35i6.16614"},{"key":"8_CR15","doi-asserted-by":"crossref","unstructured":"Huang, S.C., Shen, L., Lungren, M.P., Yeung, S.: Gloria: a multimodal global-local representation learning framework for label-efficient medical image recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3942\u20133951 (2021)","DOI":"10.1109\/ICCV48922.2021.00391"},{"key":"8_CR16","doi-asserted-by":"publisher","unstructured":"de\u00a0la Iglesia\u00a0Vay\u00e1, M., et al.: BIMCV Covid-19+: a large annotated dataset of RX and CT images from Covid-19 patients (2021). https:\/\/doi.org\/10.21227\/w3aw-rv39","DOI":"10.21227\/w3aw-rv39"},{"key":"8_CR17","doi-asserted-by":"crossref","unstructured":"Irvin, J., et\u00a0al.: CheXpert: a large chest radiograph dataset with uncertainty labels and expert comparison. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 590\u2013597 (2019)","DOI":"10.1609\/aaai.v33i01.3301590"},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"Johnson, A.E., et al.: MIMIC-CXR, a de-identified publicly available database of chest radiographs with free-text reports. Sci. Data 6(1), 317 (2019)","DOI":"10.1038\/s41597-019-0322-0"},{"key":"8_CR19","unstructured":"Landman, B., Xu, Z., Igelsias, J., Styner, M., Langerak, T., Klein, A.: Miccai multi-atlas labeling beyond the cranial vault\u2013workshop and challenge. In: Proceedings of the MICCAI Multi-Atlas Labeling Beyond Cranial Vault-Workshop Challenge, vol.\u00a05, p.\u00a012 (2015)"},{"key":"8_CR20","doi-asserted-by":"crossref","unstructured":"Li, Y., Fan, H., Hu, R., Feichtenhofer, C., He, K.: Scaling language-image pre-training via masking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23390\u201323400 (2023)","DOI":"10.1109\/CVPR52729.2023.02240"},{"key":"8_CR21","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"280","DOI":"10.1007\/978-3-031-20077-9_17","volume-title":"ECCV 2022","author":"Y Li","year":"2022","unstructured":"Li, Y., Mao, H., Girshick, R., He, K.: Exploring plain vision transformer backbones for object detection. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13669, pp. 280\u2013296. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_17"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Liu, J., et al.: Clip-driven universal model for organ segmentation and tumor detection. arXiv preprint arXiv:2301.00785 (2023)","DOI":"10.1109\/ICCV51070.2023.01934"},{"key":"8_CR23","unstructured":"Loshchilov, I., Hutter, F.: SGDR: stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016)"},{"key":"8_CR24","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"8_CR25","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"658","DOI":"10.1007\/978-3-031-19809-0_39","volume-title":"ECCV 2022","author":"P M\u00fcller","year":"2022","unstructured":"M\u00fcller, P., Kaissis, G., Zou, C., Rueckert, D.: Joint learning of localized representations from medical images and reports. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13686, pp. 658\u2013701. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19809-0_39"},{"key":"8_CR26","doi-asserted-by":"crossref","unstructured":"Nguyen, D.M., et al.: Joint self-supervised image-volume representation learning with intra-inter contrastive clustering. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a037, pp. 14426\u201314435 (2023)","DOI":"10.1609\/aaai.v37i12.26687"},{"key":"8_CR27","doi-asserted-by":"crossref","unstructured":"Noroozi, M., Favaro, P.: Unsupervised learning of visual representations by solving jigsaw puzzles (2016). arXiv preprint arXiv:1603.09246, 2","DOI":"10.1007\/978-3-319-46466-4_5"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Pathak, D., Krahenbuhl, P., Donahue, J., Darrell, T., Efros, A.A.: Context encoders: feature learning by inpainting. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2536\u20132544 (2016)","DOI":"10.1109\/CVPR.2016.278"},{"key":"8_CR29","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"8_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.media.2017.06.015","volume":"42","author":"AAA Setio","year":"2017","unstructured":"Setio, A.A.A., et al.: Validation, comparison, and combination of algorithms for automatic detection of pulmonary nodules in computed tomography images: the luna16 challenge. Med. Image Anal. 42, 1\u201313 (2017)","journal-title":"Med. Image Anal."},{"key":"8_CR31","doi-asserted-by":"crossref","unstructured":"Shih, G., et\u00a0al.: Augmenting the national institutes of health chest radiograph dataset with expert annotations of possible pneumonia. Radiol. Artif. Intell. 1(1), e180041 (2019)","DOI":"10.1148\/ryai.2019180041"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: LXMERT: learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490 (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"8_CR33","doi-asserted-by":"crossref","unstructured":"Tang, Y., et al.: Self-supervised pre-training of swin transformers for 3d medical image analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20730\u201320740 (2022)","DOI":"10.1109\/CVPR52688.2022.02007"},{"key":"8_CR34","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Ray, J., LeCun, Y., Paluri, M.: A closer look at spatiotemporal convolutions for action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6450\u20136459 (2018)","DOI":"10.1109\/CVPR.2018.00675"},{"key":"8_CR35","unstructured":"Wang, F., Zhou, Y., Wang, S., Vardhanabhuti, V., Yu, L.: Multi-granularity cross-modal alignment for generalized medical visual representation learning. In: Advances in Neural Information Processing Systems, vol. 35, pp. 33536\u201333549 (2022)"},{"issue":"1","key":"8_CR36","doi-asserted-by":"publisher","first-page":"19549","DOI":"10.1038\/s41598-020-76550-z","volume":"10","author":"L Wang","year":"2020","unstructured":"Wang, L., Lin, Z.Q., Wong, A.: Covid-net: a tailored deep convolutional neural network design for detection of Covid-19 cases from chest X-ray images. Sci. Rep. 10(1), 19549 (2020)","journal-title":"Sci. Rep."},{"key":"8_CR37","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"558","DOI":"10.1007\/978-3-031-19803-8_33","volume-title":"ECCV 2022","author":"Y Xie","year":"2022","unstructured":"Xie, Y., Zhang, J., Xia, Y., Wu, Q.: UniMiSS: universal medical self-supervised learning via breaking dimensionality barrier. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13681, pp. 558\u2013575. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19803-8_33"},{"key":"8_CR38","unstructured":"Xu, K., et al.: Show, attend and tell: neural image caption generation with visual attention. In: International Conference on Machine Learning, pp. 2048\u20132057. PMLR (2015)"},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Yang, Y., et al.: Attentive mask clip. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 2771\u20132781 (2023)","DOI":"10.1109\/ICCV51070.2023.00260"},{"issue":"6","key":"8_CR40","doi-asserted-by":"publisher","first-page":"1423","DOI":"10.1016\/j.cell.2020.04.045","volume":"181","author":"K Zhang","year":"2020","unstructured":"Zhang, K., et al.: Clinically applicable AI system for accurate diagnosis, quantitative measurements, and prognosis of Covid-19 pneumonia using computed tomography. Cell 181(6), 1423\u20131433 (2020)","journal-title":"Cell"},{"key":"8_CR41","unstructured":"Zhang, Y., Jiang, H., Miura, Y., Manning, C.D., Langlotz, C.P.: Contrastive learning of medical visual representations from paired images and text. In: Machine Learning for Healthcare Conference, pp. 2\u201325. PMLR (2022)"},{"key":"8_CR42","unstructured":"Zhou, J., et al.: iBOT: image BERT pre-training with online tokenizer. arXiv preprint arXiv:2111.07832 (2021)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73004-7_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T17:04:48Z","timestamp":1730394288000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73004-7_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031730030","9783031730047"],"references-count":42,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73004-7_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}