{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:01:11Z","timestamp":1777654871076,"version":"3.51.4"},"publisher-location":"Cham","reference-count":36,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030871956","type":"print"},{"value":"9783030871963","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-87196-3_9","type":"book-chapter","created":{"date-parts":[[2021,9,23]],"date-time":"2021-09-23T06:19:41Z","timestamp":1632377981000},"page":"90-101","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["Self-supervised Multi-modal Alignment for Whole Body Medical Imaging"],"prefix":"10.1007","author":[{"given":"Rhydian","family":"Windsor","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Amir","family":"Jamaludin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Timor","family":"Kadir","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andrew","family":"Zisserman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,9,21]]},"reference":[{"key":"9_CR1","unstructured":"Alwassel, H., Mahajan, D., Korbar, B., Torresani, L., Ghanem, B., Tran, D.: Self-supervised learning by cross-modal audio-video clustering. In: NeurIPS (2020)"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Arandjelovi\u0107, R., Zisserman, A.: Look, listen and learn. In: Proceedings of the ICCV (2017)","DOI":"10.1109\/ICCV.2017.73"},{"key":"9_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"451","DOI":"10.1007\/978-3-030-01246-5_27","volume-title":"Computer Vision \u2013 ECCV 2018","author":"R Arandjelovi\u0107","year":"2018","unstructured":"Arandjelovi\u0107, R., Zisserman, A.: Objects that sound. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11205, pp. 451\u2013466. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01246-5_27"},{"key":"9_CR4","unstructured":"Asano, Y.M., Rupprecht, C., Vedaldi, A.: Self-labelling via simultaneous clustering and representation learning. In: Proceedings of the ICLR (2020)"},{"key":"9_CR5","doi-asserted-by":"crossref","unstructured":"Borga, M.: MRI adipose tissue and muscle composition analysis\u2013a review of automation techniques. Br. J. Radiol. 91(1089), 20180252 (2018)","DOI":"10.1259\/bjr.20180252"},{"key":"9_CR6","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"key":"9_CR7","unstructured":"Caron, M., Misra, I., Mairal, J., Goyal, P., Bojanowski, P., Joulin, A.: Unsupervised learning of visual features by contrasting cluster assignments. In: NeurIPS (2020)"},{"key":"9_CR8","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: Proceedings of the ICLR (2020)"},{"key":"9_CR9","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the NAACL, pp. 4171\u20134186 (2019)"},{"key":"9_CR10","unstructured":"Ghorbani, A., Natarajan, V., Coz, D., Liu, Y.: DermGAN: synthetic generation of clinical skin images with pathology. In: Machine Learning for Health NeurIPS Workshop, pp. 155\u2013170 (2019)"},{"key":"9_CR11","unstructured":"Grill, J.B., et al.: Bootstrap your own latent - a new approach to self-supervised learning. In: NeurIPS (2020)"},{"key":"9_CR12","unstructured":"Gutmann, M.U., Hyv\u00e4rinen, A.: Noise-contrastive estimation of unnormalized statistical models, with applications to natural image statistics. J. Mach. Learn. Res. 13(11), 307\u2013361 (2012)"},{"key":"9_CR13","unstructured":"Han, T., Xie, W., Zisserman, A.: Self-supervised co-training for video representation learning. In: NeurIPS (2020)"},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., Girshick, R.: Momentum contrast for unsupervised visual representation learning. In: Proceedings of the CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Heinrich, M.P., et al.: MIND: modality independent neighbourhood descriptor for multi-modal deformable registration. Med. Image Anal. 16(7), 1423\u20131435 (2012)","DOI":"10.1016\/j.media.2012.05.008"},{"key":"9_CR16","unstructured":"H\u00e9naff, O., et al.: Data-efficient image recognition with contrastive predictive coding. In: Proceedings of the ICLR (2020)"},{"key":"9_CR17","doi-asserted-by":"crossref","unstructured":"Howard, J., Ruder, S.: Universal language model fine-tuning for text classification. In: Proceedings of the ACL (2018)","DOI":"10.18653\/v1\/P18-1031"},{"key":"9_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1007\/978-3-030-13736-6_2","volume-title":"Computational Methods and Clinical Applications for Spine Imaging","author":"A Jamaludin","year":"2019","unstructured":"Jamaludin, A., Kadir, T., Clark, E., Zisserman, A.: Predicting scoliosis in DXA scans using intermediate representations. In: Zheng, G., Belavy, D., Cai, Y., Li, S. (eds.) CSI 2018. LNCS, vol. 11397, pp. 15\u201328. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-13736-6_2"},{"key":"9_CR19","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"294","DOI":"10.1007\/978-3-319-67558-9_34","volume-title":"Deep Learning in Medical Image Analysis and Multimodal Learning for Clinical Decision Support","author":"A Jamaludin","year":"2017","unstructured":"Jamaludin, A., Kadir, T., Zisserman, A.: Self-supervised learning for spinal MRIs. In: Cardoso, M.J., et al. (eds.) DLMIA\/ML-CDS-2017. LNCS, vol. 10553, pp. 294\u2013302. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-67558-9_34"},{"key":"9_CR20","doi-asserted-by":"crossref","unstructured":"Johnson, A.E.W., et al.: MIMIC-CXR, a de-identified publicly available database of chest radiographs with free-text reports. Sci. Data 6(1), 317 (2019)","DOI":"10.1038\/s41597-019-0322-0"},{"key":"9_CR21","unstructured":"Korbar, B., Tran, D., Torresani, L.: Cooperative learning of audio and video models from self-supervised synchronization. In: NeurIPS, vol. 31 (2018)"},{"key":"9_CR22","doi-asserted-by":"crossref","unstructured":"Lowe, D.: Object recognition from local scale-invariant features. In: Proceedings of the ICCV, pp. 1150\u20131157, September 1999","DOI":"10.1109\/ICCV.1999.790410"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Lowe, D.: Distinctive image features from scale-invariant keypoints. IJCV 60(2), 91\u2013110 (2004)","DOI":"10.1023\/B:VISI.0000029664.99615.94"},{"key":"9_CR24","doi-asserted-by":"publisher","unstructured":"Mattes, D., Haynor, D.R., Vesselle, H., Lewellyn, T.K., Eubank, W.: Nonrigid multimodality image registration. In: Sonka, M., Hanson, K.M. (eds.) Medical Imaging 2001: Image Processing, vol. 4322, pp. 1609\u20131620. International Society for Optics and Photonics, SPIE (2001). https:\/\/doi.org\/10.1117\/12.431046","DOI":"10.1117\/12.431046"},{"key":"9_CR25","unstructured":"Mikolov, T., Sutskever, I., Chen, K., Corrado, G.S., Dean, J.: Distributed representations of words and phrases and their compositionality. In: NeurIPS (2013)"},{"key":"9_CR26","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"639","DOI":"10.1007\/978-3-030-01231-1_39","volume-title":"Computer Vision \u2013 ECCV 2018","author":"A Owens","year":"2018","unstructured":"Owens, A., Efros, A.A.: Audio-visual scene analysis with self-supervised multisensory features. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11210, pp. 639\u2013658. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01231-1_39"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Qian, R., et al.: Spatiotemporal contrastive video representation learning. In: Proceedings of the CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00689"},{"key":"9_CR28","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I.: Language models are unsupervised multitask learners. Technical report, OpenAI (2019)"},{"key":"9_CR29","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W.M., Frangi, A.F. (eds.) MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28"},{"key":"9_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"288","DOI":"10.1007\/978-3-642-23626-6_36","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2011","author":"K Simonyan","year":"2011","unstructured":"Simonyan, K., Zisserman, A., Criminisi, A.: Immediate structured visual search for medical images. In: Fichtinger, G., Martel, A., Peters, T. (eds.) MICCAI 2011. LNCS, vol. 6893, pp. 288\u2013296. Springer, Heidelberg (2011). https:\/\/doi.org\/10.1007\/978-3-642-23626-6_36"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"Sudlow, C., et al.: UK Biobank: an open access resource for identifying the causes of a wide range of complex diseases of middle and old age. PLoS Med. 12(3), 1\u201310 (2015)","DOI":"10.1371\/journal.pmed.1001779"},{"key":"9_CR32","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"661","DOI":"10.1007\/978-3-030-78191-0_51","volume-title":"Information Processing in Medical Imaging","author":"A Taleb","year":"2021","unstructured":"Taleb, A., Lippert, C., Klein, T., Nabi, M.: Multimodal self-supervised learning for medical image analysis. In: Feragen, A., Sommer, S., Schnabel, J., Nielsen, M. (eds.) IPMI 2021. LNCS, vol. 12729, pp. 661\u2013673. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-78191-0_51"},{"key":"9_CR33","unstructured":"Taleb, A., et al.: 3D self-supervised methods for medical imaging. In: NeurIPS (2020)"},{"key":"9_CR34","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1007\/978-3-642-38868-2_3","volume-title":"Information Processing in Medical Imaging","author":"M Toews","year":"2013","unstructured":"Toews, M., Z\u00f6llei, L., Wells, W.M.: Feature-based alignment of volumetric multi-modal images. In: Gee, J.C., Joshi, S., Pohl, K.M., Wells, W.M., Z\u00f6llei, L. (eds.) IPMI 2013. LNCS, vol. 7917, pp. 25\u201336. Springer, Heidelberg (2013). https:\/\/doi.org\/10.1007\/978-3-642-38868-2_3"},{"key":"9_CR35","doi-asserted-by":"crossref","unstructured":"Viola, P., Wells, W.: Alignment by maximization of mutual information. In: Press, I.C.S. (ed.) Proceedings of the ICCV, pp. 16\u201323, June 1995","DOI":"10.21236\/ADA299525"},{"key":"9_CR36","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"712","DOI":"10.1007\/978-3-030-59725-2_69","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"R Windsor","year":"2020","unstructured":"Windsor, R., Jamaludin, A., Kadir, T., Zisserman, A.: A Convolutional approach to vertebrae detection and labelling in whole spine\u00a0MRI. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12266, pp. 712\u2013722. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59725-2_69"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-87196-3_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,10]],"date-time":"2023-01-10T00:34:56Z","timestamp":1673310896000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-87196-3_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030871956","9783030871963"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-87196-3_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"21 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Strasbourg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 October 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/miccai2021.org\/en\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Microsoft CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1622","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"531","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"33% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}