{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T14:02:03Z","timestamp":1783605723604,"version":"3.55.0"},"publisher-location":"Cham","reference-count":102,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031733895","type":"print"},{"value":"9783031733901","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73390-1_24","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T16:24:01Z","timestamp":1730305441000},"page":"409-427","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":40,"title":["OmniSat: Self-supervised Modality Fusion for\u00a0Earth Observation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-6242-0939","authenticated-orcid":false,"given":"Guillaume","family":"Astruc","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nicolas","family":"Gonthier","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Clement","family":"Mallet","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Loic","family":"Landrieu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"24_CR1","unstructured":"PyTorch: reduceLROnPlateau. org\/docs\/stable\/generated\/torch.optim.lr_scheduler.ReduceLROnPlateau.html#torch.optim.lr_scheduler.ReduceLROnPlateau. Accessed 29 Feb 2024"},{"key":"24_CR2","doi-asserted-by":"crossref","unstructured":"Ahlswede, S., et al.: TreeSatAI Benchmark Archive: a multi-sensor, multi-label dataset for tree species classification in remote sensing. Earth Syst. Sci. Data Discuss. 15(2), 681\u2013695 (2022)","DOI":"10.5194\/essd-15-681-2023"},{"key":"24_CR3","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: NeurIPS (2022)"},{"key":"24_CR4","doi-asserted-by":"crossref","unstructured":"Amitrano, D., et al.: Earth environmental monitoring using multi-temporal synthetic aperture radar: a critical review of selected applications. Remote Sens. 13(4), 604 (2021)","DOI":"10.3390\/rs13040604"},{"key":"24_CR5","doi-asserted-by":"crossref","unstructured":"Anderson, K., Ryan, B., Sonntag, W., Kavvada, A., Friedl, L.: Earth observation in service of the 2030 agenda for sustainable development. Geo-spat. Inf. Sci. 20(2), 77\u201396 (2017)","DOI":"10.1080\/10095020.2017.1333230"},{"key":"24_CR6","doi-asserted-by":"crossref","unstructured":"Assran, M., et al.: Self-supervised learning from images with a joint-embedding predictive architecture. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01499"},{"key":"24_CR7","doi-asserted-by":"crossref","unstructured":"Ayush, K., et al.: Geography-aware self-supervised learning. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01002"},{"key":"24_CR8","doi-asserted-by":"crossref","unstructured":"Badrinarayanan, V., Kendall, A., Cipolla, R.: SegNet: a deep convolutional encoder-decoder architecture for image segmentation. IEEE TPAMI 39(12), 2481\u20132495 (2017)","DOI":"10.1109\/TPAMI.2016.2644615"},{"key":"24_CR9","unstructured":"Baevski, A., Hsu, W.N., Xu, Q., Babu, A., Gu, J., Auli, M.: Data2vec: a general framework for self-supervised learning in speech, vision and language. In: ICML (2022)"},{"key":"24_CR10","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.: BEiT: BERT pre-training of image transformers. In: ICLR (2021)"},{"key":"24_CR11","doi-asserted-by":"crossref","unstructured":"Bao, X., et al.: Vegetation descriptors from Sentinel-1 SAR data for crop growth monitoring. ISPRS J. Photogrammetry Remote Sens. 203, 86\u2013114 (2023)","DOI":"10.1016\/j.isprsjprs.2023.07.023"},{"key":"24_CR12","doi-asserted-by":"crossref","unstructured":"Bastani, F., Wolters, P., Gupta, R., Ferdinando, J., Kembhavi, A.: SatlasPretrain: a large-scale dataset for remote sensing image understanding. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01538"},{"issue":"8","key":"24_CR13","doi-asserted-by":"publisher","first-page":"2939","DOI":"10.1007\/s00371-021-02166-7","volume":"38","author":"K Bayoudh","year":"2022","unstructured":"Bayoudh, K., Knani, R., Hamdaoui, F., Mtibaa, A.: A survey on deep multimodal learning for computer vision: advances, trends, applications, and datasets. Vis. Comput. 38(8), 2939\u20132970 (2022). https:\/\/doi.org\/10.1007\/s00371-021-02166-7","journal-title":"Vis. Comput."},{"key":"24_CR14","doi-asserted-by":"crossref","unstructured":"Benedetti, P., Ienco, D., Gaetano, R., Ose, K., Pensa, R.G., Dupuy, S.: M$$^{3}$$-Fusion: a deep learning architecture for multiscale multimodal multitemporal satellite data fusion. IEEE J. Sel. Topics Appl. Earth Observations Remote Sens. 11(12), 4939\u20134949 (2018)","DOI":"10.1109\/JSTARS.2018.2876357"},{"key":"24_CR15","unstructured":"Caron, M., Misra, I., Mairal, J., Goyal, P., Bojanowski, P., Joulin, A.: Unsupervised learning of visual features by contrasting cluster assignments. In: NeurIPS (2020)"},{"key":"24_CR16","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"24_CR17","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: ICML (2020)"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Chen, X., He, K.: Exploring simple siamese representation learning. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"24_CR19","doi-asserted-by":"crossref","unstructured":"Christie, G., Fendley, N., Wilson, J., Mukherjee, R.: Functional map of the world. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00646"},{"key":"24_CR20","unstructured":"Cong, Y., et al.: SatMAE: pre-training transformers for temporal and multi-spectral satellite imagery. In: NeurIPS (2022)"},{"key":"24_CR21","doi-asserted-by":"crossref","unstructured":"Coppin, P., Lambin, E., Jonckheere, I., Muys, B.: Digital change detection methods in natural ecosystem monitoring: A review. Analysis of multi-temporal remote sensing images (2002)","DOI":"10.1142\/9789812777249_0001"},{"key":"24_CR22","doi-asserted-by":"crossref","unstructured":"Corley, I., Robinson, C., Dodhia, R., Ferres, J.M.L., Najafirad, P.: Revisiting pre-trained remote sensing model benchmarks: resizing and normalization matters. arXiv preprint arXiv:2305.13456 (2023)","DOI":"10.1109\/CVPRW63382.2024.00322"},{"key":"24_CR23","doi-asserted-by":"crossref","unstructured":"Dai, A., Nie\u00dfner, M.: 3DMV: joint 3D-multi-view prediction for 3D semantic scene segmentation. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01249-6_28"},{"key":"24_CR24","unstructured":"DataTerra Dinamis: diffusion OpenData Dinamis. https:\/\/dinamis.data-terra.org\/opendata\/. Accessed 15 Dec 2023"},{"key":"24_CR25","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR (2020)"},{"key":"24_CR26","doi-asserted-by":"crossref","unstructured":"Drusch, M., et al.: Sentinel-2: ESA\u2019s optical high-resolution mission for GMES operational services. Remote Sens. Environ. 120, 25\u201336 (2012)","DOI":"10.1016\/j.rse.2011.11.026"},{"key":"24_CR27","doi-asserted-by":"crossref","unstructured":"Ebel, P., Xu, Y., Schmitt, M., Zhu, X.X.: SEN12MS-CR-TS: a remote-sensing data set for multimodal multitemporal cloud removal. IEEE TGRS 60, 1\u201314 (2022)","DOI":"10.1109\/TGRS.2022.3146246"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"Ekim, B., Stomberg, T.T., Roscher, R., Schmitt, M.: MapInWild: a remote sensing dataset to address the question of what makes nature wild. IEEE Geosci. Remote Sens. Magazine 11(1), 103\u2013114 (2023)","DOI":"10.1109\/MGRS.2022.3226525"},{"key":"24_CR29","unstructured":"Fuller, A., Millard, K., Green, J.R.: CROMA: remote sensing representations with contrastive radar-optical masked autoencoders. In: NeurIPS (2023)"},{"key":"24_CR30","doi-asserted-by":"crossref","unstructured":"Gao, Y., Sun, X., Liu, C.: A general self-supervised framework for remote sensing image classification. Remote Sens. 14(9), 4824 (2022)","DOI":"10.3390\/rs14194824"},{"key":"24_CR31","unstructured":"Garioud, A., et al.: FLAIR: a country-scale land cover semantic segmentation dataset from multi-source optical imagery. In: NeurIPS Dataset and Benchmark (2023)"},{"key":"24_CR32","doi-asserted-by":"crossref","unstructured":"Garnot, V.S.F., Landrieu, L.: Lightweight temporal self-attention for classifying satellite images time series. In: Advanced Analytics and Learning on Temporal Data: ECML PKDD Workshop (2020)","DOI":"10.1007\/978-3-030-65742-0_12"},{"key":"24_CR33","unstructured":"Garnot, V.S.F., Landrieu, L.: Panoptic segmentation of satellite image time series with convolutional temporal attention networks. In: ICCV (2021)"},{"key":"24_CR34","doi-asserted-by":"crossref","unstructured":"Garnot, V.S.F., Landrieu, L., Chehata, N.: Multi-modal temporal attention models for crop mapping from satellite time series. ISPRS J. Photogrammetry Remote Sens. 187, 294\u2013305 (2022)","DOI":"10.1016\/j.isprsjprs.2022.03.012"},{"key":"24_CR35","unstructured":"Garnot, V.S.F., Landrieu, L., Giordano, S., Chehata, N.: Satellite image time series classification with pixel-set encoders and temporal self-attention. In: CVPR (2020)"},{"key":"24_CR36","doi-asserted-by":"crossref","unstructured":"Ghamisi, P., et\u00a0al.: Multisource and multitemporal data fusion in remote sensing: a comprehensive review of the state of the art. IEEE Geosci. Remote Sens. Magazine 7(1), 6\u201339 (2019)","DOI":"10.1109\/MGRS.2018.2890023"},{"key":"24_CR37","unstructured":"Gidaris, S., Singh, P., Komodakis, N.: Unsupervised representation learning by predicting image rotations. In: ICLR (2018)"},{"key":"24_CR38","doi-asserted-by":"crossref","unstructured":"Girdhar, R., et al.: ImageBind: one embedding space to bind them all. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"24_CR39","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Singh, M., Ravi, N., van\u00a0der Maaten, L., Joulin, A., Misra, I.: Omnivore: a single model for many visual modalities. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01563"},{"key":"24_CR40","doi-asserted-by":"crossref","unstructured":"Goldberg, H.R., Ratto, C.R., Banerjee, A., Kelbaugh, M.T., Giglio, M., Vermote, E.F.: Automated global-scale detection and characterization of anthropogenic activity using multi-source satellite-based remote sensing imagery. In: Geospatial Informatics XIII. SPIE (2023)","DOI":"10.1117\/12.2663071"},{"key":"24_CR41","doi-asserted-by":"crossref","unstructured":"Greenwell, C., et al.: WATCH: wide-area terrestrial change hypercube. In: WACV (2024)","DOI":"10.1109\/WACV57701.2024.00809"},{"key":"24_CR42","unstructured":"Grill, J.B., et\u00a0al.: Bootstrap your own latent-a new approach to self-supervised learning. In: NeurIPS (2020)"},{"key":"24_CR43","unstructured":"Hackstein, J., Sumbul, G., Clasen, K.N., Demir, B.: Exploring masked autoencoders for sensor-agnostic image retrieval in remote sensing. arXiv preprint arXiv:2401.07782 (2024)"},{"key":"24_CR44","doi-asserted-by":"crossref","unstructured":"Hazirbas, C., Ma, L., Domokos, C., Cremers, D.: FuseNet: incorporating depth into semantic segmentation via fusion-based CNN architecture. In: ACCV (2017)","DOI":"10.1007\/978-3-319-54181-5_14"},{"key":"24_CR45","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"24_CR46","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., Girshick, R.: Momentum contrast for unsupervised visual representation learning. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"24_CR47","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"24_CR48","doi-asserted-by":"crossref","unstructured":"Hu, J., et al.: MDAS: a new multimodal benchmark dataset for remote sensing. Earth Syst. Sci. Data Discuss. 15(1), 113\u2013131 (2022)","DOI":"10.5194\/essd-15-113-2023"},{"key":"24_CR49","unstructured":"Huang, P.Y., et al.: MAViL: masked audio-video learners. In: NeurIPS (2023)"},{"key":"24_CR50","doi-asserted-by":"crossref","unstructured":"Ibanez, D., Fernandez-Beltran, R., Pla, F., Yokoya, N.: Masked auto-encoding spectral\u2013spatial transformer for hyperspectral image classification. IEEE TGRS 60, 1\u201314 (2022)","DOI":"10.1109\/TGRS.2022.3217892"},{"key":"24_CR51","unstructured":"Irvin, J., et al.: USat: a unified self-supervised encoder for multi-sensor satellite imagery. arXiv preprint arXiv:2312.02199 (2023)"},{"key":"24_CR52","unstructured":"Kenton, J.D.M.W.C., Toutanova, L.K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: NAACL (2019)"},{"key":"24_CR53","unstructured":"Kingma, D.P., Ba, J.: ADAM: a method for stochastic optimization. ICLR (2015)"},{"key":"24_CR54","doi-asserted-by":"crossref","unstructured":"Krispel, G., Opitz, M., Waltner, G., Possegger, H., Bischof, H.: FuseSeg: LiDAR point cloud segmentation fusing multi-modal data. In: WACV (2020)","DOI":"10.1109\/WACV45572.2020.9093584"},{"key":"24_CR55","doi-asserted-by":"crossref","unstructured":"Kuffer, M., et\u00a0al.: The role of Earth observation in an integrated deprived area mapping \u201csystem\u201d for low-to-middle income countries. Remote Sens. 12(6), 982 (2020)","DOI":"10.3390\/rs12060982"},{"key":"24_CR56","unstructured":"Lacoste, A., et\u00a0al.: Toward foundation models for Earth monitoring: proposal for a climate change benchmark. arXiv preprint arXiv:2112.00570 (2021)"},{"key":"24_CR57","doi-asserted-by":"crossref","unstructured":"Li, D., Tong, Q., Li, R., Gong, J., Zhang, L.: Current issues in high-resolution Earth observation technology. Sci. China Earth Sci. 55(7), 1043\u20131051 (2012)","DOI":"10.1007\/s11430-012-4445-9"},{"key":"24_CR58","volume":"112","author":"J Li","year":"2022","unstructured":"Li, J., et al.: Deep learning in multimodal remote sensing data fusion: a comprehensive review. Int. J. Appl. Earth Obs. Geoinf. 112, 102926 (2022)","journal-title":"Int. J. Appl. Earth Obs. Geoinf."},{"key":"24_CR59","doi-asserted-by":"crossref","unstructured":"Liao, Y., Xie, J., Geiger, A.: KITTI-360: a novel dataset and benchmarks for urban scene understanding in 2D and 3D. IEEE TPAMI 45(3), 3292\u20133310 (2022)","DOI":"10.1109\/TPAMI.2022.3179507"},{"key":"24_CR60","doi-asserted-by":"crossref","unstructured":"Liu, Y., Li, X., Hua, Z., Xia, C., Zhao, L.: A band selection method with masked convolutional autoencoder for hyperspectral image. IEEE Geosci. Remote Sens. Lett. 19, 1\u20135 (2022)","DOI":"10.1109\/LGRS.2022.3178824"},{"key":"24_CR61","doi-asserted-by":"crossref","unstructured":"Ma, Y., et\u00a0al.: The outcome of the 2021 IEEE GRSS data fusion contest-Track DSE: detection of settlements without electricity. IEEE J. Sel. Top Appl. Earth Observations Remote Sens.14, 12375\u201312385 (2021)","DOI":"10.1109\/JSTARS.2021.3130446"},{"key":"24_CR62","unstructured":"Mai, G., et\u00a0al.: On the opportunities and challenges of foundation models for geospatial artificial intelligence. arXiv preprint arXiv:2304.06798 (2023)"},{"key":"24_CR63","doi-asserted-by":"crossref","unstructured":"Manas, O., Lacoste, A., Gir\u00f3-i Nieto, X., Vazquez, D., Rodriguez, P.: Seasonal contrast: unsupervised pre-training from uncurated remote sensing data. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00928"},{"key":"24_CR64","unstructured":"Manfreda, S., et\u00a0al.: On the use of unmanned aerial systems for environmental monitoring. Remote sens. 10(4), 641 (2018)"},{"key":"24_CR65","doi-asserted-by":"crossref","unstructured":"Moreira, A., Prats-Iraola, P., Younis, M., Krieger, G., Hajnsek, I., Papathanassiou, K.P.: A tutorial on synthetic aperture radar. IEEE Geosci. Remote Sens. Mag. 1(1), 6\u201343 (2013)","DOI":"10.1109\/MGRS.2013.2248301"},{"key":"24_CR66","doi-asserted-by":"crossref","unstructured":"Nakalembe, C.: Urgent and critical need for sub-Saharan African countries to invest in Earth observation-based agricultural early warning and monitoring systems. Environ. Res. Lett. 15(12), 121002 (2020)","DOI":"10.1088\/1748-9326\/abc0bb"},{"key":"24_CR67","doi-asserted-by":"crossref","unstructured":"Nathan\u00a0Silberman, Derek\u00a0Hoiem, P.K., Fergus, R.: Indoor segmentation and support inference from RGBD images. In: ECCV (2012)","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"24_CR68","doi-asserted-by":"crossref","unstructured":"Noroozi, M., Favaro, P.: Unsupervised learning of visual representations by solving jigsaw puzzles. In: ECCV (2016)","DOI":"10.1007\/978-3-319-46466-4_5"},{"key":"24_CR69","unstructured":"Oord, A.v.d., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"24_CR70","unstructured":"Oquab, M., et\u00a0al.: DINOv2: learning robust visual features without supervision. TLMR (2023)"},{"key":"24_CR71","doi-asserted-by":"crossref","unstructured":"Pohl, C., Van\u00a0Genderen, J.L.: Multisensor image fusion in remote sensing: concepts, methods and applications. Int. J. Remote Sens. 19(5), 823\u2013854 (1998)","DOI":"10.1080\/014311698215748"},{"key":"24_CR72","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"24_CR73","unstructured":"Recasens, A., et\u00a0al.: Zorro: The masked multimodal transformer. arXiv preprint arXiv:2301.09595 (2023)"},{"key":"24_CR74","doi-asserted-by":"crossref","unstructured":"Reed, C.J., et al.: Scale-MAE: a scale-aware masked autoencoder for multiscale geospatial representation learning. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00378"},{"key":"24_CR75","doi-asserted-by":"crossref","unstructured":"Robert, D., Vallet, B., Landrieu, L.: Learning multi-view aggregation in the wild for large-scale 3D semantic segmentation. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00549"},{"key":"24_CR76","doi-asserted-by":"crossref","unstructured":"Robinson, C., et al.: Global land-cover mapping with weak supervision: outcome of the 2020 IEEE GRSS data fusion contest. IEEE J. Sel. Topics Appl. Earth Observations Remote Sens. 14, 3185\u20133199 (2021)","DOI":"10.1109\/JSTARS.2021.3063849"},{"key":"24_CR77","doi-asserted-by":"crossref","unstructured":"Rolf, E., et al.: A generalizable and accessible approach to machine learning with global satellite imagery. Nat. Commun. 12(1), 4392 (2021)","DOI":"10.1038\/s41467-021-24638-z"},{"key":"24_CR78","doi-asserted-by":"crossref","unstructured":"Ru\u00dfwurm, M., K\u00f6rner, M.: Self-attention for raw optical satellite time series classification. ISPRS J. Photogrammetry Remote Sens. 169, 421\u2013435 (2020)","DOI":"10.1016\/j.isprsjprs.2020.06.006"},{"key":"24_CR79","doi-asserted-by":"crossref","unstructured":"Schmitt, M., Zhu, X.X.: Data fusion and remote sensing: an ever-growing relationship. IEEE Geosci. Remote Sens. Magazine 4(4), 6\u201323 (2016)","DOI":"10.1109\/MGRS.2016.2561021"},{"key":"24_CR80","unstructured":"Secades, C.,et\u00a0al.: Earth observation for biodiversity monitoring: a review of current approaches and future opportunities for tracking progress towards the Aichi biodiversity targets. CBD technical series (2014)"},{"key":"24_CR81","doi-asserted-by":"crossref","unstructured":"Shermeyer, J., et\u00a0al.: SpaceNet 6: multi-sensor all weather mapping dataset. In: CVPR Workshop EarthVision (2020)","DOI":"10.1109\/CVPRW50498.2020.00106"},{"key":"24_CR82","unstructured":"Shukor, M., Dancette, C., Rame, A., Cord, M.: UnIVAL: unified model for image, video, audio and language tasks. TMLR (2023)"},{"key":"24_CR83","doi-asserted-by":"crossref","unstructured":"Skidmore, A.K., et\u00a0al.: Priority list of biodiversity metrics to observe from space. Nat. Ecol. Evol. 6(5), 506\u2013519 (2021)","DOI":"10.1038\/s41559-021-01451-x"},{"key":"24_CR84","doi-asserted-by":"crossref","unstructured":"Srivastava, S., Sharma, G.: OmniVec: learning robust representations with cross modal sharing. In: WACV (2024)","DOI":"10.1109\/WACV57701.2024.00127"},{"key":"24_CR85","doi-asserted-by":"crossref","unstructured":"Sudmanns, M., Tiede, D., Augustin, H., Lang, S.: Assessing global Sentinel-2 coverage dynamics and data availability for operational Earth observation (EO) applications using the EO-Compass. Int. J. Digital Earth 3(7), 768\u2013784 (2019)","DOI":"10.1080\/17538947.2019.1572799"},{"key":"24_CR86","doi-asserted-by":"crossref","unstructured":"Sumbul, G., et al.: BigEarthNet-MM: a large-scale, multimodal, multilabel benchmark archive for remote sensing image classification and retrieval. IEEE Geosci. Remote Sens. Magaz. 9(3), 174\u2013180 (2021)","DOI":"10.1109\/MGRS.2021.3089174"},{"key":"24_CR87","doi-asserted-by":"crossref","unstructured":"Tarasiou, M., Chavez, E., Zafeiriou, S.: ViTs for SITS: Vision transformers for satellite image time series. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01004"},{"key":"24_CR88","unstructured":"Tseng, G., Zvonkov, I., Purohit, M., Rolnick, D., Kerner, H.: Lightweight, pre-trained transformers for remote sensing timeseries. arXiv preprint arXiv:2304.14065 (2023)"},{"key":"24_CR89","volume-title":"CROCO: cross-modal contrastive learning for localization of Earth observation data","author":"WH Tseng","year":"2022","unstructured":"Tseng, W.H., L\u00ea, H.\u00c2., Boulch, A., Lef\u00e8vre, S., Tiede, D.: CROCO: cross-modal contrastive learning for localization of Earth observation data. ISPRS Annals of the Photogrammetry, Remote Sensing and Spatial Information Sciences (2022)"},{"key":"24_CR90","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS (2017)"},{"key":"24_CR91","doi-asserted-by":"crossref","unstructured":"Vrieling, A., et al.: Vegetation phenology from Sentinel-2 and field cameras for a Dutch barrier island. Remote Sens. Environ. 215, 517\u2013529 (2018)","DOI":"10.1016\/j.rse.2018.03.014"},{"key":"24_CR92","doi-asserted-by":"crossref","unstructured":"Wang, Y., Braham, N.A.A., Xiong, Z., Liu, C., Albrecht, C.M., Zhu, X.X.: SSL4EO-S12: a large-scale multi-modal, multi-temporal dataset for self-supervised learning in Earth observation. IEEE Geosci. Remote Sens. Magaz. 11(3), 98\u2013106 (2023)","DOI":"10.1109\/MGRS.2023.3281651"},{"key":"24_CR93","volume-title":"MultiSenGE: a multimodal and multitemporal benchmark dataset for land use\/land cover remote sensing applications","author":"R Wenger","year":"2022","unstructured":"Wenger, R., Puissant, A., Weber, J., Idoumghar, L., Forestier, G.: MultiSenGE: a multimodal and multitemporal benchmark dataset for land use\/land cover remote sensing applications. ISPRS Annals of the Photogrammetry, Remote Sensing and Spatial Information Sciences (2022)"},{"key":"24_CR94","doi-asserted-by":"crossref","unstructured":"Wu, K., Peng, H., Chen, M., Fu, J., Chao, H.: Rethinking and improving relative position encoding for vision transformer. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00988"},{"key":"24_CR95","doi-asserted-by":"crossref","unstructured":"Xie, Z., et al.: SimMim: a simple framework for masked image modeling. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"24_CR96","unstructured":"Xiong, Z., et al.: Neural plasticity-inspired foundation model for observing the Earth crossing modalities. arXiv preprint arXiv:2403.15356 (2024)"},{"key":"24_CR97","doi-asserted-by":"crossref","unstructured":"Yang, J., et al.: The role of satellite remote sensing in climate change studies. Nat. Clim. Change 3(10), 875\u2013883 (2013)","DOI":"10.1038\/nclimate1908"},{"key":"24_CR98","doi-asserted-by":"crossref","unstructured":"Yang, M.Y., Landrieu, L., Tuia, D., Toth, C.: Muti-modal learning in photogrammetry and remote sensing. ISPRS J. Photogrammetry Remote Sens. 176, 54\u201354 (2021)","DOI":"10.1016\/j.isprsjprs.2021.03.022"},{"key":"24_CR99","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Lin, L., Liu, Q., Hang, R., Zhou, Z.G.: SITS-former: a pre-trained spatio-spectral-temporal representation model for sentinel-2 time series classification. Int. J. Appl. Earth Obs. Geoinf. 106, 102651 (2022)","DOI":"10.1016\/j.jag.2021.102651"},{"key":"24_CR100","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A.: Colorful image colorization. In: ECCV (2016)","DOI":"10.1007\/978-3-319-46487-9_40"},{"key":"24_CR101","unstructured":"Zhou, J., et al.: Image BERT pre-training with online tokenizer. In: ICLR (2022)"},{"key":"24_CR102","doi-asserted-by":"crossref","unstructured":"Zong, Y., Mac\u00a0Aodha, O., Hospedales, T.: Self-supervised multimodal learning: A survey. arXiv preprint arXiv:2304.01008 (2023)","DOI":"10.1109\/TPAMI.2024.3429301"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73390-1_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T16:36:46Z","timestamp":1730306206000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73390-1_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031733895","9783031733901"],"references-count":102,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73390-1_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}