{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T15:02:26Z","timestamp":1782313346976,"version":"3.54.5"},"publisher-location":"Cham","reference-count":84,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726637","type":"print"},{"value":"9783031726644","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T00:00:00Z","timestamp":1729900800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72664-4_6","type":"book-chapter","created":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T17:02:04Z","timestamp":1729875724000},"page":"92-111","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["ChEX: Interactive Localization and\u00a0Region Description in\u00a0Chest X-Rays"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8186-6479","authenticated-orcid":false,"given":"Philip","family":"M\u00fcller","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8382-8062","authenticated-orcid":false,"given":"Georgios","family":"Kaissis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5683-5889","authenticated-orcid":false,"given":"Daniel","family":"Rueckert","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,26]]},"reference":[{"key":"6_CR1","unstructured":"Banerjee, S., Lavie, A.: Meteor: an automatic metric for mt evaluation with improved correlation with human judgments. In: ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp. 65\u201372 (2005)"},{"key":"6_CR2","doi-asserted-by":"publisher","unstructured":"Bannur, S., et al.: Learning to exploit temporal structure for biomedical vision-language processing. In: CVPR, pp. 15016\u201315027 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01442","DOI":"10.1109\/CVPR52729.2023.01442"},{"key":"6_CR3","doi-asserted-by":"publisher","unstructured":"Boecking, B., et al.: Making the most of text semantics to improve biomedical vision\u2013language processing. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV, pp. 1\u201321. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20059-5_1","DOI":"10.1007\/978-3-031-20059-5_1"},{"key":"6_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-End object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"key":"6_CR5","doi-asserted-by":"publisher","unstructured":"Chen, Z., et al.: Medical phrase grounding with region-phrase context contrastive alignment. In: Greenspan, H., et al. (eds.) MICCAI, pp. 371\u2013381. Springer, Cham (2023).https:\/\/doi.org\/10.1007\/978-3-031-43990-2_35","DOI":"10.1007\/978-3-031-43990-2_35"},{"issue":"11","key":"6_CR6","doi-asserted-by":"publisher","first-page":"13636","DOI":"10.1109\/TPAMI.2023.3296823","volume":"45","author":"J Deng","year":"2023","unstructured":"Deng, J., et al.: Transvg++: end-to-end visual grounding with language conditioned vision transformer. IEEE TPAMI 45(11), 13636\u201313652 (2023). https:\/\/doi.org\/10.1109\/TPAMI.2023.3296823","journal-title":"IEEE TPAMI"},{"key":"6_CR7","doi-asserted-by":"publisher","unstructured":"Deng, J., Yang, Z., Chen, T., Zhou, W., Li, H.: Transvg: end-to-end visual grounding with transformers. In: ICCV, pp. 1749\u20131759. IEEE (2021). https:\/\/doi.org\/10.1109\/ICCV48922.2021.00179","DOI":"10.1109\/ICCV48922.2021.00179"},{"key":"6_CR8","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"6_CR9","doi-asserted-by":"crossref","unstructured":"Du, Y., Fu, Z., Liu, Q., Wang, Y.: Visual grounding with transformers. In: ICME (2022)","DOI":"10.1109\/ICME52920.2022.9859880"},{"key":"6_CR10","unstructured":"Eslami, S., de\u00a0Melo, G., Meinel, C.: Does CLIP benefit visual question answering in the medical domain as much as it does in the general domain? ArXiv preprint arxiv:2112.13906 (2021)"},{"issue":"2","key":"6_CR11","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1148\/radiol.2019191586","volume":"293","author":"JR Geis","year":"2019","unstructured":"Geis, J.R., et al.: Ethics of artificial intelligence in radiology: summary of the joint European and North American multisociety statement. Radiology 293(2), 436\u2013440 (2019)","journal-title":"Radiology"},{"issue":"23","key":"6_CR12","first-page":"215","volume":"101","author":"A Goldberger","year":"2000","unstructured":"Goldberger, A., Amaral, L., Glass, L., Hausdorff, J., et al.: Physiobank, physiotoolkit, and physionet: components of a new research resource for complex physiologic signals. Circulation [Online] 101(23), 215\u2013220 (2000)","journal-title":"Circulation [Online]"},{"key":"6_CR13","doi-asserted-by":"crossref","unstructured":"Gu, T., Liu, D., Li, Z., Cai, W.: Complex organ mask guided radiology report generation. In: WACV, pp. 7995\u20138004 (2024)","DOI":"10.1109\/WACV57701.2024.00781"},{"key":"6_CR14","unstructured":"Gu, X., Lin, T., Kuo, W., Cui, Y.: Open-vocabulary object detection via vision and language knowledge distillation. In: ICLR (2022). https:\/\/openreview.net\/forum?id=lL3lnMbR4WU"},{"key":"6_CR15","doi-asserted-by":"publisher","unstructured":"Guo, M., Yi, H., Qin, Z., Wang, H., Men, A., Lao, Q.: Multiple prompt fusion for zero-shot lesion detection using vision-language models. In: Greenspan, H., Madabhushi, A., Mousavi, P., Salcudean, S., Duncan, J., Syeda-Mahmood, T., Taylor, R. (eds.) MICCAI, pp. 283\u2013292. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43904-9_28","DOI":"10.1007\/978-3-031-43904-9_28"},{"key":"6_CR16","unstructured":"He, J., Li, P., Liu, G., Zhao, Z., Zhong, S.: Pefomed: parameter efficient fine-tuning on multimodal large language models for medical visual question answering (2024)"},{"key":"6_CR17","doi-asserted-by":"publisher","unstructured":"Hou, W., Xu, K., Cheng, Y., Li, W., Liu, J.: ORGAN: observation-guided radiology report generation via tree reasoning. In: Rogers, A., Boyd-Graber, J., Okazaki, N. (eds.) Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, vol. 1: Long Papers, pp. 8108\u20138122. Association for Computational Linguistics, Toronto (2023). https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.451. https:\/\/aclanthology.org\/2023.acl-long.451","DOI":"10.18653\/v1\/2023.acl-long.451"},{"key":"6_CR18","doi-asserted-by":"publisher","unstructured":"Huang, S., Shen, L., Lungren, M.P., Yeung, S.: Gloria: a multimodal global-local representation learning framework for label-efficient medical image recognition. In: ICCV, pp. 3922\u20133931. IEEE (2021). https:\/\/doi.org\/10.1109\/ICCV48922.2021.00391","DOI":"10.1109\/ICCV48922.2021.00391"},{"key":"6_CR19","doi-asserted-by":"publisher","unstructured":"Huang, Y., et al.: Segment anything model for medical images? Med. Image Anal. 92, 103061 (2024). https:\/\/doi.org\/10.1016\/j.media.2023.103061. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S1361841523003213","DOI":"10.1016\/j.media.2023.103061"},{"key":"6_CR20","unstructured":"Hyland, S.L., et al.: Maira-1: a specialised large multimodal model for radiology report generation (2023)"},{"key":"6_CR21","doi-asserted-by":"publisher","unstructured":"Ichinose, A., et al.: Visual grounding of whole radiology reports for 3d ct images. In: Greenspan, H., et al. (eds.) MICCAI, pp. 611\u2013621. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43904-9_59","DOI":"10.1007\/978-3-031-43904-9_59"},{"key":"6_CR22","doi-asserted-by":"crossref","unstructured":"Jin, H., Che, H., Lin, Y., Chen, H.: Promptmrg: diagnosis-driven prompts for medical report generation (2024)","DOI":"10.1609\/aaai.v38i3.28038"},{"key":"6_CR23","doi-asserted-by":"publisher","unstructured":"Johnson, A., Pollard, T., Berkowitz, S., et\u00a0al.: Mimic-cxr, a de-identified publicly available database of chest radiographs with free-text reports. Sci. Data 6(317) (2019). https:\/\/doi.org\/10.1038\/s41597-019-0322-0","DOI":"10.1038\/s41597-019-0322-0"},{"key":"6_CR24","doi-asserted-by":"publisher","unstructured":"Johnson, A., Pollard, T., Mark, R., Berkowitz, S., Horng, S.: Mimic-cxr database (version 2.0.0). PhysioNet (2019). https:\/\/doi.org\/10.13026\/C2JT1Q","DOI":"10.13026\/C2JT1Q"},{"key":"6_CR25","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment anything (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"6_CR26","unstructured":"Li, C., et al.: Llava-med: training a large language-and-vision assistant for biomedicine in one day (2023)"},{"key":"6_CR27","doi-asserted-by":"publisher","unstructured":"Li, L.H., et al.: Grounded language-image pre-training. In: CVPR, pp. 10955\u201310965 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.01069","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"6_CR28","doi-asserted-by":"publisher","unstructured":"Li, M., Lin, B., Chen, Z., Lin, H., Liang, X., Chang, X.: Dynamic graph enhanced contrastive learning for chest x-ray report generation. In: CVPR, pp. 3334\u20133343 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.00325","DOI":"10.1109\/CVPR52729.2023.00325"},{"key":"6_CR29","unstructured":"Li, M., Sigal, L.: Referring transformer: a one-step approach to multi-task visual grounding. In: Ranzato, M., Beygelzimer, A., Dauphin, Y.N., Liang, P., Vaughan, J.W. (eds.) NeurIPS, pp. 19652\u201319664 (2021). https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/a376802c0811f1b9088828288eb0d3f0-Abstract.html"},{"key":"6_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1007\/978-3-030-87196-3_26","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021","author":"R Liao","year":"2021","unstructured":"Liao, R., et al.: Multimodal representation learning via maximization of local mutual information. In: de Bruijne, M., et al. (eds.) MICCAI 2021. LNCS, vol. 12902, pp. 273\u2013283. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-87196-3_26"},{"issue":"2","key":"6_CR31","doi-asserted-by":"publisher","first-page":"318","DOI":"10.1109\/TPAMI.2018.2858826","volume":"42","author":"TY Lin","year":"2020","unstructured":"Lin, T.Y., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. IEEE TPAMI 42(2), 318\u2013327 (2020). https:\/\/doi.org\/10.1109\/TPAMI.2018.2858826","journal-title":"IEEE TPAMI"},{"key":"6_CR32","doi-asserted-by":"publisher","unstructured":"Liu, J., et al.: Parameter-efficient transfer learning for medical visual question answering. IEEE Trans. Emerg. Topics Comput. Intell. 1\u201311 (2023). https:\/\/doi.org\/10.1109\/TETCI.2023.3311333","DOI":"10.1109\/TETCI.2023.3311333"},{"key":"6_CR33","unstructured":"Liu, S., et al.: DAB-DETR: dynamic anchor boxes are better queries for DETR. In: ICLR (2022). https:\/\/openreview.net\/forum?id=oMI9PjOb9Jl"},{"key":"6_CR34","doi-asserted-by":"crossref","unstructured":"Liu, S., et al.: Grounding dino: marrying dino with grounded pre-training for open-set object detection (2023)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"6_CR35","doi-asserted-by":"crossref","unstructured":"Liu, X., Ji, K., Fu, Y., Du, Z., Yang, Z., Tang, J.: P-tuning v2: prompt tuning can be comparable to fine-tuning universally across scales and tasks. CoRR arxiv:2110.07602 (2021)","DOI":"10.18653\/v1\/2022.acl-short.8"},{"issue":"1","key":"6_CR36","doi-asserted-by":"publisher","first-page":"654","DOI":"10.1038\/s41467-024-44824-z","volume":"15","author":"J Ma","year":"2024","unstructured":"Ma, J., He, Y., Li, F., Han, L., You, C., Wang, B.: Segment anything in medical images. Nat. Commun. 15(1), 654 (2024)","journal-title":"Nat. Commun."},{"key":"6_CR37","doi-asserted-by":"publisher","unstructured":"Maaz, M., Rasheed, H., Khan, S., Khan, F.S., Anwer, R.M., Yang, M.H.: Class-agnostic object detection with multi-modal transformer. In: ECCV. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-20080-9_30","DOI":"10.1007\/978-3-031-20080-9_30"},{"key":"6_CR38","doi-asserted-by":"publisher","unstructured":"Meng, D., et al.: Conditional detr for fast training convergence. In: ICCV, pp. 3631\u20133640 (2021). https:\/\/doi.org\/10.1109\/ICCV48922.2021.00363","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"6_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.artint.2018.07.007","volume":"267","author":"T Miller","year":"2019","unstructured":"Miller, T.: Explanation in artificial intelligence: insights from the social sciences. Artif. Intell. 267, 1\u201338 (2019)","journal-title":"Artif. Intell."},{"key":"6_CR40","doi-asserted-by":"crossref","unstructured":"Miura, Y., Zhang, Y., Tsai, E., Langlotz, C., Jurafsky, D.: Improving factual completeness and consistency of image-to-text radiology report generation. In: NAACL, pp. 5288\u20135304 (2021)","DOI":"10.18653\/v1\/2021.naacl-main.416"},{"key":"6_CR41","doi-asserted-by":"publisher","first-page":"685","DOI":"10.1007\/978-3-031-19809-0_39","volume-title":"ECCV","author":"P M\u00fcller","year":"2022","unstructured":"M\u00fcller, P., Kaissis, G., Zou, C., Rueckert, D.: Joint learning of localized representations from medical images and reports. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV, pp. 685\u2013701. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19809-0_39"},{"key":"6_CR42","doi-asserted-by":"publisher","unstructured":"M\u00fcller, P., Meissen, F., Brandt, J., Kaissis, G., Rueckert, D.: Anatomy-driven pathology detection on chest x-rays. In: Greenspan, H., et al. (eds.) MICCAI, pp. 57\u201366. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43907-0_6","DOI":"10.1007\/978-3-031-43907-0_6"},{"key":"6_CR43","doi-asserted-by":"crossref","unstructured":"M\u00fcller, P., Meissen, F., Kaissis, G., Rueckert, D.: Weakly supervised object detection in chest x-rays with differentiable roi proposal networks and soft roi pooling (2024)","DOI":"10.1109\/TMI.2024.3435015"},{"key":"6_CR44","doi-asserted-by":"publisher","unstructured":"Nguyen, H.Q., Pham, H.H., Tuan\u00a0Linh, L., Dao, M., Khanh, L.: Vindr-cxr: an open dataset of chest x-rays with radiologist annotations (version 1.0.0). PhysioNet (2021).https:\/\/doi.org\/10.13026\/3akn-b287","DOI":"10.13026\/3akn-b287"},{"issue":"1","key":"6_CR45","doi-asserted-by":"publisher","first-page":"429","DOI":"10.1038\/s41597-022-01498-w","volume":"9","author":"HQ Nguyen","year":"2022","unstructured":"Nguyen, H.Q., et al.: Vindr-cxr: an open dataset of chest x-rays with radiologist\u2019s annotations. Sci. Data 9(1), 429 (2022). https:\/\/doi.org\/10.1038\/s41597-022-01498-w","journal-title":"Sci. Data"},{"key":"6_CR46","doi-asserted-by":"publisher","unstructured":"Nicolson, A., Dowling, J., Koopman, B.: Improving chest x-ray report generation by leveraging warm starting. Artif. Intell. Med. 144, 102633 (2023). https:\/\/doi.org\/10.1016\/j.artmed.2023.102633. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0933365723001471","DOI":"10.1016\/j.artmed.2023.102633"},{"key":"6_CR47","unstructured":"van\u00a0den Oord, A., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv: 1807.03748 (2019)"},{"key":"6_CR48","unstructured":"Pellegrini, C., \u00d6zsoy, E., Busam, B., Navab, N., Keicher, M.: Radialog: a large vision-language model for radiology report generation and conversational assistance (2023)"},{"key":"6_CR49","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) ICML. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763. PMLR (2021). http:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"issue":"8","key":"6_CR50","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI blog 1(8), 9 (2019)","journal-title":"OpenAI blog"},{"key":"6_CR51","doi-asserted-by":"publisher","unstructured":"Rajpurkar, P., et\u00a0al.: Chexnet: radiologist-level pneumonia detection on chest x-rays with deep learning. arXiv preprint arXiv:1711.05225 (2017). https:\/\/doi.org\/10.48550\/arXiv.1711.05225","DOI":"10.48550\/arXiv.1711.05225"},{"key":"6_CR52","unstructured":"Ramesh, V., Chi, N.A., Rajpurkar, P.: Improving radiology report generation systems by removing hallucinated references to non-existent priors. In: Machine Learning for Health, pp. 456\u2013473. PMLR (2022)"},{"key":"6_CR53","unstructured":"Rasheed, H., Maaz, M., Khattak, M.U., Khan, S., Khan, F.S.: Bridging the gap between object and image-level representations for open-vocabulary detection. In: NIPS (2022)"},{"key":"6_CR54","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. NIPS 28 (2015)"},{"key":"6_CR55","doi-asserted-by":"publisher","unstructured":"Seibold, C., Rei\u00df, S., Sarfraz, M.S., Stiefelhagen, R., Kleesiek, J.: Breaking with fixed set pathology recognition through report-guided contrastive training. In: Wang, L., Dou, Q., Fletcher, P.T., Speidel, S., Li, S. (eds.) MICCAI, pp. 690\u2013700. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-16443-9_66","DOI":"10.1007\/978-3-031-16443-9_66"},{"key":"6_CR56","doi-asserted-by":"crossref","unstructured":"Smit, A., Jain, S., Rajpurkar, P., Pareek, A., Ng, A.Y., Lungren, M.P.: Chexbert: combining automatic labelers and expert annotations for accurate radiology report labeling using bert (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.117"},{"key":"6_CR57","doi-asserted-by":"publisher","unstructured":"van Sonsbeek, T., Derakhshani, M.M., Najdenkoska, I., Snoek, C.G.M., Worring, M.: Open-ended medical visual question answering through prefix tuning of language models. In: Greenspan, H., et al. (eds.) MICCAI, pp. 726\u2013736. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43904-9_70","DOI":"10.1007\/978-3-031-43904-9_70"},{"key":"6_CR58","doi-asserted-by":"publisher","unstructured":"Sun, J., Wei, D., Wang, L., Zheng, Y.: Lesion guided explainable few weak-shot medical report generation, pp. 615\u2013625. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-16443-9_59","DOI":"10.1007\/978-3-031-16443-9_59"},{"key":"6_CR59","doi-asserted-by":"publisher","unstructured":"Tanida, T., M\u00fcller, P., Kaissis, G., Rueckert, D.: Interactive and explainable region-guided radiology report generation. In: CVPR, pp. 7433\u20137442 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.00718","DOI":"10.1109\/CVPR52729.2023.00718"},{"issue":"12","key":"6_CR60","doi-asserted-by":"publisher","first-page":"1399","DOI":"10.1038\/s41551-022-00936-9","volume":"6","author":"E Tiu","year":"2022","unstructured":"Tiu, E., Talius, E., Patel, P., Langlotz, C.P., Ng, A.Y., Rajpurkar, P.: Expert-level detection of pathologies from unannotated chest x-ray images via self-supervised learning. Nat. Biomed. Eng. 6(12), 1399\u20131406 (2022)","journal-title":"Nat. Biomed. Eng."},{"key":"6_CR61","doi-asserted-by":"publisher","unstructured":"Tu, T., et al.: Towards generalist biomedical ai. NEJM AI 1(3), AIoa2300138 (2024). https:\/\/doi.org\/10.1056\/AIoa2300138","DOI":"10.1056\/AIoa2300138"},{"key":"6_CR62","unstructured":"Wang, F., Zhou, Y., Wang, S., Vardhanabhuti, V., Yu, L.: Multi-granularity cross-modal alignment for generalized medical visual representation learning. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) NeurIPS (2022)"},{"key":"6_CR63","doi-asserted-by":"publisher","unstructured":"Wang, L., Ning, M., Lu, D., Wei, D., Zheng, Y., Chen, J.: An inclusive task-aware framework for radiology report generation. In: MICCAI, pp. 568\u2013577. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-16452-1_54","DOI":"10.1007\/978-3-031-16452-1_54"},{"key":"6_CR64","doi-asserted-by":"publisher","unstructured":"Wang, W., et al.: Image as a foreign language: beit pretraining for vision and vision-language tasks. In: CVPR, pp. 19175\u201319186 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01838","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"6_CR65","doi-asserted-by":"publisher","unstructured":"Wang, X., Peng, Y., Lu, L., Lu, Z., Bagheri, M., Summers, R.M.: Chestx-ray8: hospital-scale chest x-ray database and benchmarks on weakly-supervised classification and localization of common thorax diseases. In: CVPR, pp. 2097\u20132106 (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.369","DOI":"10.1109\/CVPR.2017.369"},{"key":"6_CR66","doi-asserted-by":"publisher","unstructured":"Wang, Z., Liu, L., Wang, L., Zhou, L.: Metransformer: radiology report generation by transformer with multiple learnable expert tokens. In: CVPR, pp. 11558\u201311567 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01112","DOI":"10.1109\/CVPR52729.2023.01112"},{"key":"6_CR67","doi-asserted-by":"crossref","unstructured":"Wang, Z., Wu, Z., Agarwal, D., Sun, J.: MedCLIP: contrastive learning from unpaired medical images and text. In: Conference on Empirical Methods in Natural Language Processing. pp. 3876\u20133887. Association for Computational Linguistics, Abu Dhabi (2022). https:\/\/aclanthology.org\/2022.emnlp-main.256","DOI":"10.18653\/v1\/2022.emnlp-main.256"},{"key":"6_CR68","unstructured":"Wu, J., et\u00a0al.: Chest imagenome dataset for clinical reasoning. In: NIPS (2021)"},{"key":"6_CR69","doi-asserted-by":"publisher","unstructured":"Wu, J.T., et\u00a0al.: Chest imagenome dataset (version 1.0.0). PhysioNet (2021). https:\/\/doi.org\/10.13026\/wv01-y230","DOI":"10.13026\/wv01-y230"},{"key":"6_CR70","doi-asserted-by":"publisher","unstructured":"Wu, X., Zhu, F., Zhao, R., Li, H.: Cora: adapting clip for open-vocabulary detection with region prompting and anchor pre-matching. In: CVPR, pp. 7031\u20137040 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.00679","DOI":"10.1109\/CVPR52729.2023.00679"},{"key":"6_CR71","doi-asserted-by":"publisher","unstructured":"Wu, Y., et al.: Zero-shot nuclei detection via visual-language pre-trained models. In: Greenspan, H., et al. (eds.) MICCAI, pp. 693\u2013703. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43987-2_67","DOI":"10.1007\/978-3-031-43987-2_67"},{"key":"6_CR72","unstructured":"Xu, L., Ni, Z., Liu, X., Wang, X., Li, H., Zhang, S.: Learning a multi-task transformer via unified and customized instruction tuning for chest radiograph interpretation (2023)"},{"key":"6_CR73","unstructured":"Xu, S., et al.: Elixr: towards a general purpose x-ray artificial intelligence system through alignment of large language models and radiology vision encoders (2023)"},{"key":"6_CR74","doi-asserted-by":"publisher","unstructured":"Yang, Z., et al.: Unitab: unifying text and box outputs for grounded vision-language modeling. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV, pp. 521\u2013539. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20059-5_30","DOI":"10.1007\/978-3-031-20059-5_30"},{"key":"6_CR75","doi-asserted-by":"publisher","unstructured":"Zang, Y., Li, W., Zhou, K., Huang, C., Loy, C.C.: Open-vocabulary detr with conditional matching. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV, pp. 106\u2013122. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_7","DOI":"10.1007\/978-3-031-20077-9_7"},{"key":"6_CR76","doi-asserted-by":"publisher","unstructured":"Zareian, A., Rosa, K.D., Hu, D.H., Chang, S.F.: Open-vocabulary object detection using captions. In: CVPR, pp. 14388\u201314397 (2021). https:\/\/doi.org\/10.1109\/CVPR46437.2021.01416","DOI":"10.1109\/CVPR46437.2021.01416"},{"key":"6_CR77","doi-asserted-by":"publisher","unstructured":"Zhang, G., Luo, Z., Yu, Y., Cui, K., Lu, S.: Accelerating detr convergence via semantic-aligned matching. In: CVPR, pp. 939\u2013948 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.00102","DOI":"10.1109\/CVPR52688.2022.00102"},{"key":"6_CR78","unstructured":"Zhang, H., et al.: Glipv2: Unifying localization and vision-language understanding. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) NeurIPS, vol.\u00a035, pp. 36067\u201336080. Curran Associates, Inc. (2022). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/ea370419760b421ce12e3082eb2ae1a8-Paper-Conference.pdf"},{"key":"6_CR79","unstructured":"Zhang, K., et al.: Biomedgpt: a unified and generalist biomedical generative pre-trained transformer for vision, language, and multimodal tasks (2024)"},{"key":"6_CR80","unstructured":"Zhang, Y., Jiang, H., Miura, Y., Manning, C.D., Langlotz, C.P.: Contrastive learning of medical visual representations from paired images and text. In: Lipton, Z., Ranganath, R., Sendak, M., Sjoding, M., Yeung, S. (eds.) Machine Learning for Healthcare Conference. Proceedings of Machine Learning Research, vol.\u00a0182, pp. 2\u201325. PMLR (2022)"},{"key":"6_CR81","doi-asserted-by":"publisher","unstructured":"Zhong, Y., et al.: Regionclip: region-based language-image pretraining. In: CVPR, pp. 16772\u201316782 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.01629","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"6_CR82","doi-asserted-by":"publisher","unstructured":"Zhou, X., Girdhar, R., Joulin, A., Kr\u00e4henb\u00fchl, P., Misra, I.: Detecting twenty-thousand classes using image-level supervision. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022, pp. 350\u2013368. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_21","DOI":"10.1007\/978-3-031-20077-9_21"},{"key":"6_CR83","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. In: ICLR (2021), https:\/\/openreview.net\/forum?id=gZ9hCDWe6ke"},{"key":"6_CR84","doi-asserted-by":"publisher","unstructured":"Zong, Z., Song, G., Liu, Y.: Detrs with collaborative hybrid assignments training. In: ICCV, pp. 6725\u20136735 (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.00621","DOI":"10.1109\/ICCV51070.2023.00621"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72664-4_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T08:27:52Z","timestamp":1732955272000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72664-4_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,26]]},"ISBN":["9783031726637","9783031726644"],"references-count":84,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72664-4_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,26]]},"assertion":[{"value":"26 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}