{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T01:10:24Z","timestamp":1755825024557,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3731715.3733396","type":"proceedings-article","created":{"date-parts":[[2025,6,25]],"date-time":"2025-06-25T18:31:04Z","timestamp":1750876264000},"page":"861-870","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["MR4SseC: A Multimodal Representation Learning Framework for Space Science Experiment of China's Space Station"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9362-3447","authenticated-orcid":false,"given":"Yunfei","family":"Liu","sequence":"first","affiliation":[{"name":"Technology and Engineering Center for Space Utilization, Chinese Academy of Sciences, Beijing, China, Key Laboratory of Space Utilization, Chinese Academy of Sciences, Beijing, China, and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9033-6701","authenticated-orcid":false,"given":"Anqi","family":"Liu","sequence":"additional","affiliation":[{"name":"Technology and Engineering Center for Space Utilization, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9607-3397","authenticated-orcid":false,"given":"Yanan","family":"Liu","sequence":"additional","affiliation":[{"name":"Technology and Engineering Center for Space Utilization, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9007-1311","authenticated-orcid":false,"given":"Yunziwei","family":"Deng","sequence":"additional","affiliation":[{"name":"Technology and Engineering Center for Space Utilization, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2027-1495","authenticated-orcid":false,"given":"Yizhao","family":"Wang","sequence":"additional","affiliation":[{"name":"Technology and Engineering Center for Space Utilization, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9888-9869","authenticated-orcid":false,"given":"Shengyang","family":"Li","sequence":"additional","affiliation":[{"name":"Technology and Engineering Center for Space Utilization, Chinese Academy of Sciences, Beijing, China, Key Laboratory of Space Utilization, Chinese Academy of Sciences, Beijing, China, and University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,6,30]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"8748","volume-title":"International Conference on Machine Learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning, pages 8748--8763. PMLR."},{"key":"e_1_3_2_1_2_1","volume-title":"A survey of vision-language pre-trained models. arXiv preprint arXiv:2202.10936","author":"Du Yifan","year":"2022","unstructured":"Yifan Du, Zikang Liu, Junyi Li, and Wayne Xin Zhao. A survey of vision-language pre-trained models. arXiv preprint arXiv:2202.10936, 2022."},{"key":"e_1_3_2_1_3_1","volume-title":"Similarity reasoning and filtration for image-text matching. arXiv preprint arXiv:2101.01368","author":"Diao Haiwen","year":"2021","unstructured":"Haiwen Diao, Ying Zhang, Lin Mafdevlin2018bert, and Huchuan Lu. Similarity reasoning and filtration for image-text matching. arXiv preprint arXiv:2101.01368, 2021."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00475"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01267"},{"key":"e_1_3_2_1_7_1","volume-title":"Learning semantic concepts and order for image and sentence matching. CoRR, abs\/1712.02036","author":"Huang Yan","year":"2017","unstructured":"Yan Huang, Qi Wu, and Liang Wang. Learning semantic concepts and order for image and sentence matching. CoRR, abs\/1712.02036, 2017."},{"key":"e_1_3_2_1_8_1","volume-title":"Deep visual-semantic alignments for generating image descriptions. CoRR, abs\/1412.2306","author":"Karpathy Andrej","year":"2014","unstructured":"Andrej Karpathy and Li Fei-Fei. Deep visual-semantic alignments for generating image descriptions. CoRR, abs\/1412.2306, 2014."},{"key":"e_1_3_2_1_9_1","volume-title":"Oscar: Object semantics aligned pre-training for vision-language tasks. CoRR, abs\/2004.06165","author":"Li Xiujun","year":"2020","unstructured":"Xiujun Li, Xi Yin, Chunyuan Li, Pengchuan Zhang, Xiaowei Hu, Lei Zhang, Lijuan Wang, Houdong Hu, Li Dong, Furu Wei, Yejin Choi, and Jianfeng Gao. Oscar: Object semantics aligned pre-training for vision-language tasks. CoRR, abs\/2004.06165, 2020."},{"key":"e_1_3_2_1_10_1","volume-title":"Self-supervised learning: The dark matter of intelligence","author":"LeCun Y","year":"2021","unstructured":"Y LeCun and I Misra. Self-supervised learning: The dark matter of intelligence, 2021."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_1_12_1","first-page":"1597","volume-title":"International conference on machine learning","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. A simple framework for contrastive learning of visual representations. In International conference on machine learning, pp. 1597--1607. PMLR, 2020a."},{"key":"e_1_3_2_1_13_1","volume-title":"Unsupervised learning of visual features by contrasting cluster assignments. arXiv preprint arXiv:2006.09882","author":"Caron Mathilde","year":"2020","unstructured":"Mathilde Caron, Ishan Misra, Julien Mairal, Priya Goyal, Piotr Bojanowski, and Armand Joulin. Unsupervised learning of visual features by contrasting cluster assignments. arXiv preprint arXiv:2006.09882, 2020."},{"key":"e_1_3_2_1_14_1","first-page":"67","volume-title":"European Conference on Computer Vision","author":"Joulin Armand","unstructured":"Armand Joulin, Laurens van der Maaten, Allan Jabri, and Nicolas Vasilache. 2016. Learning visual features from large weakly supervised data. In European Conference on Computer Vision, pages 67--84. Springer."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.449"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58598-3_10"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01101"},{"key":"e_1_3_2_1_18_1","first-page":"5583","volume-title":"International Conference on Machine Learning","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. Vilt: Vision-and-language transformer without convolution or region supervision. In International Conference on Machine Learning, pages 5583--5594. PMLR."},{"key":"e_1_3_2_1_19_1","unstructured":"Wenhui Wang Hangbo Bao Li Dong and Furu Wei. 2021a. VLMo: Unified vision-language pretraining with mixture-of-modality-experts. arXiv preprint arXiv:2111.02358"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_2_1_21_1","volume-title":"Daylen Yang, Anna Rohrbach, Trevor Darrell, and Marcus Rohrbach. Multimodal compact bilinear pooling for visual question answering and visual grounding. arXiv preprint arXiv:1606.01847","author":"Fukui Akira","year":"2016","unstructured":"Akira Fukui, Dong Huk Park, Daylen Yang, Anna Rohrbach, Trevor Darrell, and Marcus Rohrbach. Multimodal compact bilinear pooling for visual question answering and visual grounding. arXiv preprint arXiv:1606.01847, 2016."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.93"},{"key":"e_1_3_2_1_23_1","volume-title":"Learning to count objects in natural images for visual question answering. arXiv preprint arXiv:1802.05766","author":"Zhang Yan","year":"2018","unstructured":"Yan Zhang, Jonathon Hare, and Adam Pru\u00a8gel-Bennett. Learning to count objects in natural images for visual question answering. arXiv preprint arXiv:1802.05766, 2018"},{"key":"e_1_3_2_1_24_1","first-page":"2048","volume-title":"International conference on machine learning","author":"Xu Kelvin","year":"2015","unstructured":"Kelvin Xu, Jimmy Ba, Ryan Kiros, Kyunghyun Cho, Aaron Courville, Ruslan Salakhudinov, Rich Zemel, and Yoshua Bengio. Show, attend and tell: Neural image caption generation with visual attention. In International conference on machine learning, pages 2048--2057. PMLR, 2015."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.494"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00133"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00601"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00419"},{"key":"e_1_3_2_1_29_1","volume-title":"Jamie Ryan Kiros, and Sanja Fidler. Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612","author":"Faghri Fartash","year":"2017","unstructured":"Fartash Faghri, David J Fleet, Jamie Ryan Kiros, and Sanja Fidler. Vse++: Improving visual-semantic embeddings with hard negatives. arXiv preprint arXiv:1707.05612, 2017."},{"key":"e_1_3_2_1_30_1","volume-title":"STAIR: Learning Sparse Text and Image Representation in Grounded Tokens.","author":"Chen Chen","year":"2023","unstructured":"Chen Chen,Bowen Zhang,Liangliang Cao,Jiguang Shen,Tom Gunter,Albin Madappally Jose,Alexander Toshev,Jonathon Shlens,Ruoming Pang Yinfei Yang. 2023. STAIR: Learning Sparse Text and Image Representation in Grounded Tokens."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00143"},{"key":"e_1_3_2_1_32_1","volume-title":"Ales Leonardis and Steven McDonagh. Improving Object Detection via Local-global Contrastive Learning. arXiv preprint arXiv:2410.05058","author":"Triantafyllidou Danai","year":"2024","unstructured":"Danai Triantafyllidou, Sarah Parisot, Ales Leonardis and Steven McDonagh. Improving Object Detection via Local-global Contrastive Learning. arXiv preprint arXiv:2410.05058, 2024."},{"key":"e_1_3_2_1_33_1","first-page":"8372","volume-title":"Li Zhenguo and Luo Ping. DetCo: Unsupervised Contrastive Learning for Object Detection. In Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Enze Xie","year":"2021","unstructured":"Xie Enze,Ding Jian,Wang Wenhai, Zhan Xiaohang, Xu Hang, Sun Peize,Li Zhenguo and Luo Ping. DetCo: Unsupervised Contrastive Learning for Object Detection. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8372--8381,2021."},{"key":"e_1_3_2_1_34_1","volume-title":"Attention is all you need. Advances in neural information processing systems, 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. Attention is all you need. Advances in neural information processing systems, 30, 2017"},{"key":"e_1_3_2_1_35_1","volume-title":"SSuieBERT: Domain Adaptation Model for Chinese Space Science Text Mining and Information Extraction. Electronics","author":"L.","year":"2024","unstructured":"Liu, Y.; Li, S.; Deng, Y.; Hao, S.; Wang, L. SSuieBERT: Domain Adaptation Model for Chinese Space Science Text Mining and Information Extraction. Electronics 2024, 13, 2949."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"e_1_3_2_1_37_1","volume-title":"Simcse: Simple contrastive learning of sentence embeddings. arXiv preprint arXiv:2104.08821","author":"Gao Tianyu","year":"2021","unstructured":"Tianyu Gao, Xingcheng Yao, and Danqi Chen. Simcse: Simple contrastive learning of sentence embeddings. arXiv preprint arXiv:2104.08821, 2021."},{"key":"e_1_3_2_1_38_1","volume-title":"Revisiting contrastive methods for unsupervised learning of visual representations. arXiv preprint arXiv:2106.05967","author":"Gansbeke Wouter Van","year":"2021","unstructured":"Wouter Van Gansbeke, Simon Vandenhende, Stamatios Georgoulis, and Luc Van Gool. Revisiting contrastive methods for unsupervised learning of visual representations. arXiv preprint arXiv:2106.05967, 2021."},{"key":"e_1_3_2_1_39_1","volume-title":"Eda: Easy data augmentation techniques for boosting performance on text classi?cation tasks. arXiv preprint arXiv:1901.11196","author":"Wei Jason","year":"2019","unstructured":"Jason Wei and Kai Zou. Eda: Easy data augmentation techniques for boosting performance on text classi?cation tasks. arXiv preprint arXiv:1901.11196, 2019."},{"key":"e_1_3_2_1_40_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2015. Deep residual learning for image recognition."},{"key":"e_1_3_2_1_41_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et al. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929, 2020."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"}],"event":{"name":"ICMR '25: International Conference on Multimedia Retrieval","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Chicago IL USA","acronym":"ICMR '25"},"container-title":["Proceedings of the 2025 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731715.3733396","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T04:07:18Z","timestamp":1755749238000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731715.3733396"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":42,"alternative-id":["10.1145\/3731715.3733396","10.1145\/3731715"],"URL":"https:\/\/doi.org\/10.1145\/3731715.3733396","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-06-30","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}