{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T22:00:41Z","timestamp":1783116041992,"version":"3.54.6"},"reference-count":42,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,4,17]],"date-time":"2025-04-17T00:00:00Z","timestamp":1744848000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2025,4,17]],"date-time":"2025-04-17T00:00:00Z","timestamp":1744848000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21A20472"],"award-info":[{"award-number":["U21A20472"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Plan of China","award":["2021YFB3600503"],"award-info":[{"award-number":["2021YFB3600503"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Artif Intell Rev"],"DOI":"10.1007\/s10462-025-11224-8","type":"journal-article","created":{"date-parts":[[2025,4,17]],"date-time":"2025-04-17T17:50:51Z","timestamp":1744912251000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Scalable multi-modal representation learning networks"],"prefix":"10.1007","volume":"58","author":[{"given":"Zihan","family":"Fang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Zou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiyang","family":"Lan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shide","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanchao","family":"Tan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiping","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,4,17]]},"reference":[{"key":"11224_CR1","doi-asserted-by":"crossref","unstructured":"Fan Y, Xu W, Wang H et\u00a0al (2023) PMR: prototypical modal rebalance for multimodal learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 20029\u201320038","DOI":"10.1109\/CVPR52729.2023.01918"},{"key":"11224_CR2","doi-asserted-by":"publisher","first-page":"8889","DOI":"10.1109\/TMM.2024.3383295","volume":"26","author":"Z Fang","year":"2024","unstructured":"Fang Z, Du S, Cai Z et al (2024) Representation learning meets optimization-derived networks: from single-view to multi-view. IEEE Trans Multimed 26:8889\u20138901","journal-title":"IEEE Trans Multimed"},{"key":"11224_CR3","doi-asserted-by":"crossref","unstructured":"Feng Y, You H, Zhang Z et\u00a0al (2019) Hypergraph neural networks. In: Proceedings of the AAAI conference on artificial intelligence. pp 3558\u20133565","DOI":"10.1609\/aaai.v33i01.33013558"},{"issue":"5","key":"11224_CR4","first-page":"2548","volume":"44","author":"Y Gao","year":"2020","unstructured":"Gao Y, Zhang Z, Lin H et al (2020) Hypergraph learning: methods and practices. IEEE Trans Pattern Anal Mach Intell 44(5):2548\u20132566","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"11224_CR5","unstructured":"Gregor K, LeCun Y (2010) Learning fast approximations of sparse coding. In: International conference on machine learning. pp 399\u2013406"},{"key":"11224_CR6","unstructured":"Han Z, Zhang C, Fu H et\u00a0al (2020) Trusted multi-view classification. In: International conference on learning representations. pp 1\u201311"},{"key":"11224_CR7","doi-asserted-by":"crossref","unstructured":"Han Z, Yang F, Huang J et\u00a0al (2022) Multimodal dynamics: dynamical fusion for trustworthy multimodal classification. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 20707\u201320717","DOI":"10.1109\/CVPR52688.2022.02005"},{"key":"11224_CR8","doi-asserted-by":"crossref","unstructured":"Hu P, Zhu H, Peng X et\u00a0al (2020) Semi-supervised multi-modal learning with balanced spectral decomposition. In: Proceedings of the AAAI conference on artificial intelligence. pp 99\u2013106","DOI":"10.1609\/aaai.v34i01.5339"},{"key":"11224_CR9","doi-asserted-by":"crossref","unstructured":"Huang J, Yang J (2021) UniGNN: a unified framework for graph and hypergraph neural networks. In: Proceedings of the international joint conference on artificial intelligence. pp 2563\u20132569","DOI":"10.24963\/ijcai.2021\/353"},{"key":"11224_CR10","doi-asserted-by":"crossref","unstructured":"Huang Y, Liu Q, Zhang S et\u00a0al (2010) Image retrieval via probabilistic hypergraph ranking. In: IEEE computer society conference on computer vision and pattern recognition. pp 3376\u20133383","DOI":"10.1109\/CVPR.2010.5540012"},{"issue":"9","key":"11224_CR11","doi-asserted-by":"publisher","first-page":"9471","DOI":"10.1007\/s10462-023-10397-4","volume":"56","author":"Z Ibrahim","year":"2023","unstructured":"Ibrahim Z, Bosaghzadeh A, Dornaika F (2023) Joint graph and reduced flexible manifold embedding for scalable semi-supervised learning. Artif Intell Rev 56(9):9471\u20139495","journal-title":"Artif Intell Rev"},{"key":"11224_CR12","doi-asserted-by":"crossref","unstructured":"Jia X, Han K, Zhu Y et\u00a0al (2021) Joint representation learning and novel category discovery on single- and multi-modal data. In: Proceedings of the IEEE\/CVF international conference on computer vision. pp 610\u2013619","DOI":"10.1109\/ICCV48922.2021.00065"},{"key":"11224_CR13","doi-asserted-by":"crossref","unstructured":"Jiang J, Wei Y, Feng Y et\u00a0al (2019) Dynamic hypergraph neural networks. In: Proceedings of the international joint conference on artificial intelligence. pp 2635\u20132641","DOI":"10.24963\/ijcai.2019\/366"},{"key":"11224_CR14","doi-asserted-by":"crossref","unstructured":"Jiang Q, Chen C, Zhao H et\u00a0al (2023) Understanding and constructing latent modality structures in multi-modal representation learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 7661\u20137671","DOI":"10.1109\/CVPR52729.2023.00740"},{"key":"11224_CR15","doi-asserted-by":"crossref","unstructured":"Jin T, Cao L, Zhang B et\u00a0al (2019) Hypergraph induced convolutional manifold networks. In: Proceedings of the international joint conference on artificial intelligence. pp 2670\u20132676","DOI":"10.24963\/ijcai.2019\/371"},{"key":"11224_CR16","doi-asserted-by":"crossref","unstructured":"Kiela D, Grave E, Joulin A et\u00a0al (2018) Efficient large-scale multi-modal classification. In: Proceedings of the AAAI conference on artificial intelligence. pp 1\u20137","DOI":"10.1609\/aaai.v32i1.11945"},{"key":"11224_CR17","doi-asserted-by":"crossref","unstructured":"Kim ES, Kang WY, On KW et\u00a0al (2020) Hypergraph attention networks for multimodal learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 14581\u201314590","DOI":"10.1109\/CVPR42600.2020.01459"},{"key":"11224_CR18","doi-asserted-by":"crossref","unstructured":"Li S, Li WT, Wang W (2020) Co-GCN for multi-view semi-supervised learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 04. pp 4691\u20134698","DOI":"10.1609\/aaai.v34i04.5901"},{"key":"11224_CR19","doi-asserted-by":"crossref","unstructured":"Li L, Fan K, Yuan C (2022) Cross-modal representation learning and relation reasoning for bidirectional adaptive manipulation. In: Proceedings of the international joint conference on artificial intelligence. pp 3222\u20133228","DOI":"10.24963\/ijcai.2022\/447"},{"key":"11224_CR20","first-page":"17612","volume":"35","author":"VW Liang","year":"2022","unstructured":"Liang VW, Zhang Y, Kwon Y et al (2022) Mind the gap: understanding the modality gap in multi-modal contrastive representation learning. Adv Neural Inf Process Syst 35:17612\u201317625","journal-title":"Adv Neural Inf Process Syst"},{"issue":"1","key":"11224_CR21","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1109\/TPAMI.2012.88","volume":"35","author":"G Liu","year":"2012","unstructured":"Liu G, Lin Z, Yan S et al (2012) Robust recovery of subspace structures by low-rank representation. IEEE Trans Pattern Anal Mach Intell 35(1):171\u2013184","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"11224_CR22","doi-asserted-by":"crossref","unstructured":"Lu X, Zhu L, Liu L et\u00a0al (2021) Graph convolutional multi-modal hashing for flexible multimedia retrieval. In: Proceedings of the ACM international conference on multimedia. pp 1414\u20131422","DOI":"10.1145\/3474085.3475598"},{"key":"11224_CR23","doi-asserted-by":"crossref","unstructured":"Lu J, Wu Z, Chen Z et\u00a0al (2024) Towards multi-view consistent graph diffusion. In: Proceedings of the ACM international conference on multimedia. pp 186\u2013195","DOI":"10.1145\/3664647.3681258"},{"key":"11224_CR24","doi-asserted-by":"crossref","unstructured":"Mao Y, Yan X, Guo Q et\u00a0al (2021) Deep mutual information maximin for cross-modal clustering. In: Proceedings of the AAAI conference on artificial intelligence. pp 8893\u20138901","DOI":"10.1609\/aaai.v35i10.17076"},{"key":"11224_CR25","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.101947","volume":"100","author":"E Pan","year":"2023","unstructured":"Pan E, Kang Z (2023) High-order multi-view clustering for generic data. Inf Fusion 100:101947","journal-title":"Inf Fusion"},{"key":"11224_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102338","volume":"108","author":"T Pan","year":"2024","unstructured":"Pan T, Ye Y, Zhang Y et al (2024) Online multi-hypergraph fusion learning for cross-subject emotion recognition. Inf Fusion 108:102338","journal-title":"Inf Fusion"},{"issue":"7","key":"11224_CR27","doi-asserted-by":"publisher","first-page":"178","DOI":"10.1007\/s10462-024-10730-5","volume":"57","author":"F Shafizadegan","year":"2024","unstructured":"Shafizadegan F, Naghsh-Nilchi AR, Shabaninia E (2024) Multimodal vision-based human action recognition using deep learning: a review. Artif Intell Rev 57(7):178","journal-title":"Artif Intell Rev"},{"issue":"10","key":"11224_CR28","doi-asserted-by":"publisher","first-page":"4705","DOI":"10.1109\/TKDE.2020.3048678","volume":"34","author":"C Tang","year":"2021","unstructured":"Tang C, Zheng X, Liu X et al (2021) Cross-view locality preserved diversity and consensus learning for multi-view unsupervised feature selection. IEEE Trans Knowl Data Eng 34(10):4705\u20134716","journal-title":"IEEE Trans Knowl Data Eng"},{"issue":"9","key":"11224_CR29","doi-asserted-by":"publisher","first-page":"4283","DOI":"10.1109\/TIP.2017.2717191","volume":"26","author":"H Tao","year":"2017","unstructured":"Tao H, Hou C, Nie F et al (2017) Scalable multi-view semi-supervised classification via adaptive regression. IEEE Trans Image Process 26(9):4283\u20134296","journal-title":"IEEE Trans Image Process"},{"key":"11224_CR30","doi-asserted-by":"crossref","unstructured":"Wei S, Luo C, Luo Y (2023) MMANet: margin-aware distillation and modality-aware regularization for incomplete multimodal learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp 20039\u201320049","DOI":"10.1109\/CVPR52729.2023.01919"},{"key":"11224_CR31","doi-asserted-by":"publisher","first-page":"8593","DOI":"10.1109\/TMM.2023.3260649","volume":"25","author":"Z Wu","year":"2023","unstructured":"Wu Z, Lin X, Lin Z et al (2023) Interpretable graph convolutional network for multi-view semi-supervised learning. IEEE Trans Multimed 25:8593\u20138606","journal-title":"IEEE Trans Multimed"},{"key":"11224_CR32","doi-asserted-by":"publisher","first-page":"1170","DOI":"10.1109\/TIP.2023.3240863","volume":"32","author":"W Xia","year":"2023","unstructured":"Xia W, Wang T, Gao Q et al (2023) Graph embedding contrastive multi-modal representation learning for clustering. IEEE Trans Image Process 32:1170\u20131183","journal-title":"IEEE Trans Image Process"},{"issue":"2","key":"11224_CR33","doi-asserted-by":"publisher","first-page":"572","DOI":"10.1109\/TCYB.2018.2869789","volume":"50","author":"Y Xie","year":"2018","unstructured":"Xie Y, Zhang W, Qu Y et al (2018) Hyper-Laplacian regularized multilinear multiview self-representations for clustering and semisupervised learning. IEEE Trans Cybern 50(2):572\u2013586","journal-title":"IEEE Trans Cybern"},{"key":"11224_CR34","unstructured":"Xie X, Wu J, Liu G et\u00a0al (2019) Differentiable linearized ADMM. In: International conference on machine learning. pp 6902\u20136911"},{"key":"11224_CR35","doi-asserted-by":"crossref","unstructured":"Xu C, Zhao W, Zhao J et\u00a0al (2023) Progressive deep multi-view comprehensive representation learning. In: Proceedings of the AAAI conference on artificial intelligence. pp 10557\u201310565","DOI":"10.1609\/aaai.v37i9.26254"},{"key":"11224_CR36","doi-asserted-by":"crossref","unstructured":"Xu C, Si J, Guan Z et\u00a0al (2024) Reliable conflictive multi-view learning. In: Proceedings of the AAAI conference on artificial intelligence. pp 16129\u201316137","DOI":"10.1609\/aaai.v38i14.29546"},{"issue":"6","key":"11224_CR37","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1007\/s10462-024-10785-4","volume":"57","author":"Z Yang","year":"2024","unstructured":"Yang Z, Tan Y (2024) The methods for improving large-scale multi-view clustering efficiency: a survey. Artif Intell Rev 57(6):153","journal-title":"Artif Intell Rev"},{"issue":"3","key":"11224_CR38","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1109\/TPAMI.2015.2462360","volume":"38","author":"M Yin","year":"2015","unstructured":"Yin M, Gao J, Lin Z (2015) Laplacian regularized low-rank representation and its applications. IEEE Trans Pattern Anal Mach Intell 38(3):504\u2013517","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"12","key":"11224_CR39","doi-asserted-by":"publisher","first-page":"5957","DOI":"10.1109\/TIP.2018.2862625","volume":"27","author":"Z Zhang","year":"2018","unstructured":"Zhang Z, Lin H, Zhao X et al (2018) Inductive multi-hypergraph learning and its application on view-based 3D object classification. IEEE Trans Image Process 27(12):5957\u20135968","journal-title":"IEEE Trans Image Process"},{"issue":"Suppl 1","key":"11224_CR40","doi-asserted-by":"publisher","first-page":"421","DOI":"10.1007\/s10462-023-10529-w","volume":"56","author":"Y Zhang","year":"2023","unstructured":"Zhang Y, Nie R, Cao J et al (2023) SS-SSAN: a self-supervised subspace attentional network for multi-modal medical image fusion. Artif Intell Rev 56(Suppl 1):421\u2013443","journal-title":"Artif Intell Rev"},{"key":"11224_CR41","doi-asserted-by":"publisher","first-page":"198","DOI":"10.1016\/j.inffus.2022.08.014","volume":"89","author":"Q Zheng","year":"2023","unstructured":"Zheng Q, Zhu J, Li Z et al (2023) Comprehensive multi-view representation learning. Inf Fusion 89:198\u2013209","journal-title":"Inf Fusion"},{"key":"11224_CR42","first-page":"1","volume":"19","author":"D Zhou","year":"2006","unstructured":"Zhou D, Huang J, Sch\u00f6lkopf B (2006) Learning with hypergraphs: clustering, classification, and embedding. Adv Neural Inf Process Syst 19:1\u20138","journal-title":"Adv Neural Inf Process Syst"}],"container-title":["Artificial Intelligence Review"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11224-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10462-025-11224-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10462-025-11224-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,13]],"date-time":"2025-05-13T17:43:51Z","timestamp":1747158231000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10462-025-11224-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,17]]},"references-count":42,"journal-issue":{"issue":"7","published-online":{"date-parts":[[2025,7]]}},"alternative-id":["11224"],"URL":"https:\/\/doi.org\/10.1007\/s10462-025-11224-8","relation":{},"ISSN":["1573-7462"],"issn-type":[{"value":"1573-7462","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,4,17]]},"assertion":[{"value":"4 April 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"209"}}