{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T00:15:01Z","timestamp":1783728901193,"version":"3.55.0"},"reference-count":100,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s10489-026-07378-9","type":"journal-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T23:47:50Z","timestamp":1783727270000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A survey of transformers based on their input modalities"],"prefix":"10.1007","volume":"56","author":[{"given":"Priyanka","family":"Sharma","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sanjay Kumar","family":"Jain","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,11]]},"reference":[{"key":"7378_CR1","first-page":"5998","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani A et al (2017) Attention is all you need. In: Guyon I, von Luxburg U, Bengio S, Wallach H, Fergus R, Vishwanathan S, Garnett R (eds) Advances in Neural information processing systems. Curran Associates, Inc., pp 5998\u20136008","journal-title":"Advances in Neural Information Processing Systems"},{"key":"7378_CR2","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2019) BERT: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics pp. 4171\u20134186. https:\/\/doi.org\/10.18653\/v1\/N19-1423","DOI":"10.18653\/v1\/N19-1423"},{"key":"7378_CR3","unstructured":"Radford A, Narasimhan K, Salimans T, Sutskever I (2018) Improving language understanding by generative pre-training, OpenAI Technical Report"},{"issue":"140","key":"7378_CR4","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel C et al (2020) Exploring the limits of transfer learning with a unified text-to-text transformer. J Mach Learn Res 21(140):1\u201367","journal-title":"J Mach Learn Res"},{"key":"7378_CR5","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"7378_CR6","unstructured":"Ba JL, Kiros JR, Hinton GE (2016) Layer normalization, arXiv Prepr. arXiv1607.06450"},{"key":"7378_CR7","unstructured":"Liu Y et al (2019) RoBERTa: A robustly optimized bert pretraining approach. CoRR abs\/1907.11692"},{"issue":"8","key":"7378_CR8","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford A et al (2019) Language models are unsupervised multitask learners. OpenAI blog 1(8):9","journal-title":"OpenAI blog"},{"key":"7378_CR9","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown T et al (2020) Language models are few-shot learners. Adv Neural Inf Process Syst 33:1877\u20131901","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR10","doi-asserted-by":"publisher","unstructured":"Lewis M et al (2020) BART: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, Online: Association for Computational Linguistics, pp 7871\u20137880. https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.703","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"7378_CR11","unstructured":"OpenAI (2023) GPT-4 Technical Report. arXiv:2303.08774"},{"key":"7378_CR12","unstructured":"Touvron H et al (2023) Llama: Open and efficient foundation language models, arXiv Prepr.arXiv2302.13971"},{"key":"7378_CR13","unstructured":"Grattafiori A et al (2024) The llama 3 herd of models, arXiv Prepr. arXiv2407.21783"},{"key":"7378_CR14","unstructured":"Dosovitskiy A et al (2021) An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations"},{"key":"7378_CR15","doi-asserted-by":"publisher","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","volume":"29","author":"W-N Hsu","year":"2021","unstructured":"Hsu W-N, Bolte B, Tsai Y-HH, Lakhotia K, Salakhutdinov R, Mohamed A (2021) Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans Audio Speech Lang Process 29:3451\u20133460","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"7378_CR16","unstructured":"Radford A, Kim JW, Xu T, Brockman G, McLeavey C, Sutskever I (2023) Robust speech recognition via large-scale weak supervision, in International conference on machine learning, pp 28492\u201328518"},{"key":"7378_CR17","doi-asserted-by":"publisher","first-page":"23716","DOI":"10.52202\/068431-1723","volume":"35","author":"J-B Alayrac","year":"2022","unstructured":"Alayrac J-B et al (2022) Flamingo: a visual language model for few-shot learning. Adv Neural Inf Process Syst 35:23716\u201323736","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR18","unstructured":"Driess D et al (2023) Palm-e: An embodied multimodal language model, arXiv Prepr. arXiv2303.03378"},{"key":"7378_CR19","doi-asserted-by":"crossref","unstructured":"Liu H, Li C, Li Y, Lee YJ (2024) Improved baselines with visual instruction tuning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 26296\u201326306","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"7378_CR20","unstructured":"Yang Z, Dai Z, Yang Y, Carbonell J, Salakhutdinov RR, Le QV (2019) Xlnet: Generalized autoregressive pretraining for language understanding. Adv Neural Inf Process Syst 32"},{"key":"7378_CR21","unstructured":"Lan Z, Chen M, Goodman S, Gimpel K, Sharma P, Soricut R (2020) ALBERT : A lite BERT for self-supervised learning of language representations. In: International Conference on Learning Representations"},{"key":"7378_CR22","unstructured":"Touvron H et al (2023) Llama 2: Open foundation and fine-tuned chat models, arXiv Prepr. arXiv2307.09288"},{"key":"7378_CR23","unstructured":"Jiang AQ et al (2023) Mistral 7B, arXiv Prepr. arXiv2310.06825"},{"key":"7378_CR24","unstructured":"Jiang AQ et al (2024) Mixtral of Experts, arXiv Prepr. arXiv2401.04088"},{"key":"7378_CR25","unstructured":"DeepSeek-AI, DeepSeek LLM (2024) Scaling open-source language models with 2 trillion tokens. arXiv Prepr. arXiv2401.02954"},{"key":"7378_CR26","unstructured":"Guo D et al (2025) Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, arXiv Prepr. arXiv2501.12948"},{"key":"7378_CR27","unstructured":"Yang A et al (2025) Qwen3 technical report, arXiv Prepr. arXiv2505.09388"},{"key":"7378_CR28","doi-asserted-by":"publisher","unstructured":"Feng Z et al (2020) CodeBERT: A pre-trained model for programming and natural languages. In: Findings of the Association for Computational Linguistics: EMNLP 2020. Association for Computational Linguistics. Online, pp 1536\u20131547. https:\/\/doi.org\/10.18653\/v1\/2020.findings-emnlp.139","DOI":"10.18653\/v1\/2020.findings-emnlp.139"},{"key":"7378_CR29","unstructured":"Chen M et al (2021) Evaluating large language models trained on code, arXiv Prepr. arXiv2107.03374"},{"key":"7378_CR30","unstructured":"Roziere B et al (2023) Code llama: Open foundation models for code.\u00a0arXiv Prepr. arXiv2308.12950"},{"key":"7378_CR31","unstructured":"Guo D et al (2024) DeepSeek-Coder: When the large language model meets programming\u2013the rise of code intelligence, arXiv Prepr. arXiv2401.14196"},{"key":"7378_CR32","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G, Usunier N, Kirillov A, Zagoruyko S (2020) End-to-end object detection with transformers, in European conference on computer vision, pp. 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"7378_CR33","unstructured":"Touvron H, Cord M, Douze M, Massa F, Sablayrolles A, J\u00e9gou H (2021) Training data-efficient image transformers & distillation through attention. In: International conference on machine learning, pp 10347\u201310357"},{"key":"7378_CR34","doi-asserted-by":"crossref","unstructured":"Liu Z et al (2021) Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"7378_CR35","unstructured":"Bao H, Dong L, Piao S, Wei F (2022) Beit: Bert pre-training of image transformers. Int Conf Learn. Represent"},{"key":"7378_CR36","unstructured":"Bertasius G, Wang H, Torresani L (2021) Is space-time attention all you need for video understanding? In: ICML"},{"key":"7378_CR37","doi-asserted-by":"crossref","unstructured":"He K, Chen X, Xie S, Li Y, Doll\u00e1r P, Girshick R (2022) Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 16000\u201316009","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"7378_CR38","doi-asserted-by":"crossref","unstructured":"Caron M et al (2021) Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 9650\u20139660","DOI":"10.1109\/ICCV48922.2021.00951"},{"issue":"10","key":"7378_CR39","doi-asserted-by":"publisher","first-page":"12581","DOI":"10.1109\/TPAMI.2023.3282631","volume":"45","author":"K Li","year":"2023","unstructured":"Li K et al (2023) Uniformer: Unifying convolution and self-attention for visual recognition. IEEE Trans Pattern Anal Mach Intell 45(10):12581\u201312600","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"7378_CR40","doi-asserted-by":"crossref","unstructured":"Wu H et al (2021) Cvt: Introducing convolutions to vision transformers. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 22\u201331","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"7378_CR41","doi-asserted-by":"crossref","unstructured":"Kirillov A et al (2023) Segment anything. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4015\u20134026","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"7378_CR42","unstructured":"Wang Y et al (2022) Internvideo: General video foundation models via generative and discriminative learning, in European conference on computer vision"},{"key":"7378_CR43","unstructured":"Oquab M et al (2024) DINOv2: Learning robust visual features without supervision. Transaction on Machine Learning Research 2024"},{"key":"7378_CR44","doi-asserted-by":"crossref","unstructured":"Dong L, Xu S, Xu B (2018) Speech-transformer: a no-recurrence sequence-to-sequence model for speech recognition. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP) pp 5884\u20135888","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"7378_CR45","first-page":"12449","volume":"33","author":"A Baevski","year":"2020","unstructured":"Baevski A, Zhou Y, Mohamed A, Auli M (2020) wav2vec 2.0: A framework for self-supervised learning of speech representations. Adv Neural Inf Process Syst 33:12449\u201312460","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR46","doi-asserted-by":"crossref","unstructured":"Ao J et al (2022) Speecht5: Unified-modal encoder-decoder pre-training for spoken language processing. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (vol 1: Long Papers), pp 5723\u20135738","DOI":"10.18653\/v1\/2022.acl-long.393"},{"key":"7378_CR47","doi-asserted-by":"crossref","unstructured":"Chang H-J, Yang S, Lee H (2022) Distilhubert: Speech representation learning by layer-wise distillation of hidden-unit bert. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp 7087\u20137091","DOI":"10.1109\/ICASSP43922.2022.9747490"},{"key":"7378_CR48","doi-asserted-by":"publisher","first-page":"28708","DOI":"10.52202\/068431-2081","volume":"35","author":"P-Y Huang","year":"2022","unstructured":"Huang P-Y et al (2022) Masked autoencoders that listen. Adv Neural Inf Process Syst 35:28708\u201328720","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR49","doi-asserted-by":"crossref","unstructured":"Bain M, Huh J, Han T, Zisserman A (2023) Whisperx: Time-accurate speech transcription of long-form audio. In: Proc. INTERSPEECH, pp 4489\u20134493","DOI":"10.21437\/Interspeech.2023-78"},{"key":"7378_CR50","unstructured":"Barrault L et al (2023) SeamlessM4T: Massively Multilingual \\& Multimodal Machine Translation, arXiv Prepr. arXiv2308.11596"},{"key":"7378_CR51","unstructured":"Lu J, Batra D, Parikh D, Lee S (2019) ViLBERT: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in Neural Information Processing Systems, vol 32, pp 13\u201323"},{"key":"7378_CR52","unstructured":"Li LH, Yatskar M, Yin D, Hsieh C-J, Chang K-W (2019) Visualbert: A simple and performant baseline for vision and language, arXiv Prepr. arXiv1908.03557"},{"key":"7378_CR53","unstructured":"Radford A et al (2021) Learning transferable visual models from natural language supervision, in International conference on machine learning, pp. 8748\u20138763"},{"key":"7378_CR54","unstructured":"Li J, Li D, Xiong C, Hoi S (2022) Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International conference on machine learning, pp 12888\u201312900"},{"key":"7378_CR55","unstructured":"Li J, Li D, Savarese S, Hoi S (2023) Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International conference on machine learning, pp 19730\u201319742"},{"issue":"240","key":"7378_CR56","first-page":"1","volume":"24","author":"A Chowdhery","year":"2023","unstructured":"Chowdhery A et al (2023) Palm: Scaling language modeling with pathways. J Mach Learn Res 24(240):1\u2013113","journal-title":"J Mach Learn Res"},{"key":"7378_CR57","unstructured":"Team G et al (2024) Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context, arXiv Prepr. arXiv2403.05530"},{"key":"7378_CR58","unstructured":"Ramesh A, Dhariwal P, Nichol A, Chu C, Chen M (2022) Hierarchical text-conditional image generation with clip latents. arXiv Prepr. arXiv2204.06125 1(2);3"},{"key":"7378_CR59","unstructured":"OpenAI Hello GPT-4o, 2024. [Online]. Available: https:\/\/openai.com\/index\/hello-gpt-4o\/"},{"key":"7378_CR60","unstructured":"OpenAI Introducing GPT-5, 2025. [Online]. Available: https:\/\/openai.com\/index\/introducing-gpt-5\/"},{"key":"7378_CR61","doi-asserted-by":"publisher","first-page":"27730","DOI":"10.52202\/068431-2011","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang L et al (2022) Training language models to follow instructions with human feedback. Adv Neural Inf Process Syst 35:27730\u201327744","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR62","unstructured":"Longpre S et al (2023) The flan collection: Designing data and methods for effective instruction tuning. In: International Conference on Machine Learning, pp 22631\u201322648"},{"key":"7378_CR63","doi-asserted-by":"crossref","unstructured":"Kwon W et al (2023) Efficient memory management for large language model serving with paged attention. In: Proceedings of the 29th symposium on operating systems principles, pp 611\u2013626","DOI":"10.1145\/3600006.3613165"},{"key":"7378_CR64","doi-asserted-by":"publisher","first-page":"16344","DOI":"10.52202\/068431-1189","volume":"35","author":"T Dao","year":"2022","unstructured":"Dao T, Fu D, Ermon S, Rudra A, R\u00e9 C (2022) Flashattention: Fast and memory-efficient exact attention with io-awareness. Adv Neural Inf Process Syst 35:16344\u201316359","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR65","unstructured":"Choromanski K et al (2020) Rethinking attention with performers, arXiv Prepr. arXiv2009.14794"},{"key":"7378_CR66","unstructured":"Gu A, Dao T (2024) Mamba: Linear-time sequence modeling with selective state spaces. In: First Conference on Language Modeling (COLM)"},{"key":"7378_CR67","unstructured":"Wang A, Singh A, Michael J, Hill F, Levy O, Bowman SR (2018) GLUE: A multi-task benchmark and analysis platform for natural language understanding.\u00a0Int Conf Learn Represent\u00a0[Online]. Available: https:\/\/openreview.net\/forum?id=rJ4km2R5t7"},{"key":"7378_CR68","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: A large-scale hierarchical image database. In: 2009 IEEE conference on computer vision and pattern recognition, pp 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"7378_CR69","doi-asserted-by":"crossref","unstructured":"Panayotov V, Chen G, Povey D, Khudanpur S (2015) Librispeech: an asr corpus based on public domain audio books. In: 2015 IEEE international conference on acoustics, speech and signal processing (ICASSP) pp 5206\u20135210","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"7378_CR70","unstructured":"Hendrycks D et al (2021) Measuring Massive Multitask Language Understanding. Int Conf Learn Represent"},{"key":"7378_CR71","doi-asserted-by":"crossref","unstructured":"Yue X et al (2024) Mmmu: A massive multi-discipline multimodal understanding and reasoning benchmark for expert agi. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 9556\u20139567","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"7378_CR72","unstructured":"Wang C, Pino J, Wu A, Gu J (2020) Covost: A diverse multilingual speech-to-text translation corpus, arXiv Prepr. arXiv2002.01320"},{"key":"7378_CR73","doi-asserted-by":"crossref","unstructured":"Conneau A et al (2023) Fleurs: Few-shot learning evaluation of universal representations of speech. In: 2022 IEEE Spoken Language Technology Workshop (SLT), pp 798\u2013805","DOI":"10.1109\/SLT54892.2023.10023141"},{"key":"7378_CR74","doi-asserted-by":"crossref","unstructured":"Lin T-Y et al (2014) Microsoft coco: Common objects in context. In: European conference on computer vision, pp 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"7378_CR75","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"7378_CR76","doi-asserted-by":"crossref","unstructured":"Antol S et al (2015) Vqa: Visual question answering. In: Proceedings of the IEEE international conference on computer vision, pp 2425\u20132433","DOI":"10.1109\/ICCV.2015.279"},{"key":"7378_CR77","doi-asserted-by":"crossref","unstructured":"Hudson DA, Manning CD (2019) Gqa: A new dataset for real-world visual reasoning and compositional question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6700\u20136709","DOI":"10.1109\/CVPR.2019.00686"},{"key":"7378_CR78","unstructured":"Lu P et al (2023) Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts, arXiv Prepr. arXiv2310.02255"},{"key":"7378_CR79","doi-asserted-by":"crossref","unstructured":"Lin S, Hilton J, Evans O (2022) Truthfulqa: Measuring how models mimic human falsehoods. In: Proceedings of the 60th annual meeting of the association for computational linguistics (vol 1: long papers), pp 3214\u20133252","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"7378_CR80","doi-asserted-by":"publisher","first-page":"635","DOI":"10.1162\/tacl_a_00566","volume":"11","author":"F Liu","year":"2023","unstructured":"Liu F, Emerson G, Collier N (2023) Visual spatial reasoning. Trans Assoc Comput Linguist 11:635\u2013651","journal-title":"Trans Assoc Comput Linguist"},{"key":"7378_CR81","doi-asserted-by":"publisher","unstructured":"Gehman S, Gururangan S, Sap M, Choi Y, Smith NA (2020) RealToxicityPrompts: Evaluating neural toxic degeneration in language models. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp 3356\u2013336 https:\/\/doi.org\/10.18653\/v1\/2020.findings-emnlp.301","DOI":"10.18653\/v1\/2020.findings-emnlp.301"},{"key":"7378_CR82","unstructured":"Liang P et al (2023) Holistic evaluation of language models. Trans Mach Learn Res"},{"key":"7378_CR83","doi-asserted-by":"publisher","first-page":"140632","DOI":"10.52202\/079017-4464","volume":"37","author":"T Lee","year":"2024","unstructured":"Lee T et al (2024) Vhelm: A holistic evaluation of vision language models. Adv Neural Inf Process Syst 37:140632\u2013140666","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR84","doi-asserted-by":"publisher","first-page":"69981","DOI":"10.52202\/075280-3067","volume":"36","author":"T Lee","year":"2023","unstructured":"Lee T et al (2023) Holistic evaluation of text-to-image models. Adv Neural Inf Process Syst 36:69981\u201370011","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR85","unstructured":"Lee T et al (2025) Ahelm: A holistic evaluation of audio-language models, arXiv Prepr. arXiv2508.21376"},{"key":"7378_CR86","doi-asserted-by":"crossref","unstructured":"Dai D et al (2024) Deepseekmoe: Towards ultimate expert specialization in mixture-of-experts language models, arXiv Prepr. arXiv2401.06066","DOI":"10.18653\/v1\/2024.acl-long.70"},{"key":"7378_CR87","unstructured":"Bai Y et al (2022) Constitutional ai: Harmlessness from ai feedback, arXiv Prepr. arXiv2212.08073"},{"key":"7378_CR88","doi-asserted-by":"publisher","first-page":"24824","DOI":"10.52202\/068431-1800","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei J et al (2022) Chain-of-thought prompting elicits reasoning in large language models. Adv Neural Inf Process Syst 35:24824\u201324837","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR89","unstructured":"Wang X et al (2023) Self-consistency improves chain of thought reasoning in language models. Int Conf Learn Represent"},{"key":"7378_CR90","doi-asserted-by":"publisher","first-page":"11809","DOI":"10.52202\/075280-0517","volume":"36","author":"S Yao","year":"2023","unstructured":"Yao S et al (2023) Tree of thoughts: Deliberate problem solving with large language models. Adv Neural Inf Process Syst 36:11809\u201311822","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR91","unstructured":"Kadavath S et al (2022) Language models (mostly) know what they know, arXiv Prepr. arXiv2207.05221"},{"key":"7378_CR92","first-page":"9459","volume":"33","author":"P Lewis","year":"2020","unstructured":"Lewis P et al (2020) Retrieval-augmented generation for knowledge-intensive nlp tasks. Adv Neural Inf Process Syst 33:9459\u20139474","journal-title":"Adv Neural Inf Process Syst"},{"key":"7378_CR93","unstructured":"Askell A et al (2021) A general language assistant as a laboratory for alignment, arXiv Prepr. arXiv2112.00861"},{"key":"7378_CR94","doi-asserted-by":"publisher","first-page":"755","DOI":"10.1038\/s41586-024-07566-y","volume":"631","author":"I Shumailov","year":"2024","unstructured":"Shumailov I, Shumaylov Z, Zhao Y, Gal Y, Papernot N, Anderson R (2024) The curse of recursion: Training on generated data makes models forget. Nature 631:755\u2013759","journal-title":"Nature"},{"key":"7378_CR95","unstructured":"LeCun Y (2022) A path towards autonomous machine intelligence. Version 0.9.2.OpenReview"},{"key":"7378_CR96","doi-asserted-by":"crossref","unstructured":"Zhao H, Jiang L, Jia J, Torr PHS, Koltun V (2021) Point transformer. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 16259\u201316268","DOI":"10.1109\/ICCV48922.2021.01595"},{"key":"7378_CR97","doi-asserted-by":"crossref","unstructured":"Yu X, Tang L, Rao Y, Huang T, Zhou J, Lu J (2022) Point-bert: Pre-training 3d point cloud transformers with masked point modeling. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19313\u201319322","DOI":"10.1109\/CVPR52688.2022.01871"},{"key":"7378_CR98","first-page":"2440001","volume":"1","author":"Y Pang","year":"2023","unstructured":"Pang Y, Tay EHF, Yuan L, Chen Z (2023) Masked autoencoders for 3d point cloud self-supervised learning. World Sci Annu Rev Artif Intell 1:2440001","journal-title":"World Sci Annu Rev Artif Intell"},{"key":"7378_CR99","doi-asserted-by":"crossref","unstructured":"Wu X et al (2024) Point transformer v3: Simpler faster stronger. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4840\u20134851","DOI":"10.1109\/CVPR52733.2024.00463"},{"key":"7378_CR100","unstructured":"Lu D, Xie Q, Wei M, Gao K, Xu L, Li J (2022) Transformers in 3D point clouds: A survey, arXiv Prepr. arXiv2205.07417"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-026-07378-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-026-07378-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-026-07378-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T23:48:01Z","timestamp":1783727281000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-026-07378-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":100,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["7378"],"URL":"https:\/\/doi.org\/10.1007\/s10489-026-07378-9","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"27 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 July 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}}],"article-number":"348"}}