{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T05:43:52Z","timestamp":1785822232466,"version":"3.56.0"},"reference-count":113,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Innovate UK Knowledge Transfer Partnership","award":["10082602"],"award-info":[{"award-number":["10082602"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1109\/tai.2026.3670235","type":"journal-article","created":{"date-parts":[[2026,3,4]],"date-time":"2026-03-04T20:52:56Z","timestamp":1772657576000},"page":"4267-4281","source":"Crossref","is-referenced-by-count":3,"title":["Ensemble Learning for Large Language Models in Text and Code Generation: A Survey"],"prefix":"10.1109","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-6972-1799","authenticated-orcid":false,"given":"Mari","family":"Ashiga","sequence":"first","affiliation":[{"name":"University of West London","place":["London, U.K."],"department":["School of Computing and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5392-0009","authenticated-orcid":false,"given":"Wei","family":"Jie","sequence":"additional","affiliation":[{"name":"University of West London","place":["London, U.K."],"department":["School of Computing and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3734-7855","authenticated-orcid":false,"given":"Fan","family":"Wu","sequence":"additional","affiliation":[{"name":"Turing Intelligence Technology Limited","place":["London, U.K."]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3645-7710","authenticated-orcid":false,"given":"Vardan","family":"Voskanyan","sequence":"additional","affiliation":[{"name":"Turing Intelligence Technology Limited","place":["London, U.K."]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4697-8325","authenticated-orcid":false,"given":"Fateme","family":"Dinmohammadi","sequence":"additional","affiliation":[{"name":"University of West London","place":["London, U.K."],"department":["School of Computing and Engineering"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8776-821X","authenticated-orcid":false,"given":"Paul","family":"Brookes","sequence":"additional","affiliation":[{"name":"Turing Intelligence Technology Limited","place":["London, U.K."]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4551-0701","authenticated-orcid":false,"given":"Jingzhi","family":"Gong","sequence":"additional","affiliation":[{"name":"University of Leeds","place":["Leeds, U.K."],"department":["School of Computer Science"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6157-0662","authenticated-orcid":false,"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Leeds","place":["Leeds, U.K."],"department":["School of Computer Science"]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5270-8065","authenticated-orcid":false,"given":"Rafail","family":"Giavrimis","sequence":"additional","affiliation":[{"name":"Turing Intelligence Technology Limited","place":["London, U.K."]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mike","family":"Basios","sequence":"additional","affiliation":[{"name":"Turing Intelligence Technology Limited","place":["London, U.K."]}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3501-6862","authenticated-orcid":false,"given":"Leslie","family":"Kanthan","sequence":"additional","affiliation":[{"name":"Turing Intelligence Technology Limited","place":["London, U.K."]}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Evaluating large language models trained on code","year":"2021"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2011"},{"key":"ref3","article-title":"Claude 3.5 sonnet","year":"2024"},{"key":"ref4","article-title":"Alpacaeval: An automatic evaluator for instruction-following language models","author":"Lab","year":"2023"},{"key":"ref5","article-title":"Calibrating language models via augmented prompt ensembles","author":"Jiang","year":"2023","journal-title":"Proc. ICML Workshop Deployable Generative AI"},{"issue":"120","key":"ref6","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"Proc. JMLR"},{"key":"ref7","article-title":"From decoding to meta-generation: Inference-time algorithms for large language models","author":"Welleck","year":"2024","journal-title":"Proc. TMLR"},{"key":"ref8","first-page":"3008","article-title":"Learning to summarize with human feedback","volume":"33","author":"Stiennon","year":"2020","journal-title":"Proc. NeurIPS"},{"key":"ref9","first-page":"21691","article-title":"Large language model cascades with mixture of thoughts representations for cost-efficient reasoning","author":"Yue","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref10","first-page":"12697","article-title":"Calibrate before use: Improving few-shot performance of language models","volume":"139","author":"Zhao","year":"2021","journal-title":"Proc. ICML"},{"key":"ref11","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Proc. NeurIPS"},{"key":"ref12","article-title":"Outrageously large neural networks: The sparsely-gated mixture-of-experts layer","author":"Shazeer","year":"2017","journal-title":"Proc. ICLR"},{"key":"ref13","article-title":"The Llama 3 herd of models","year":"2024"},{"key":"ref14","article-title":"GPT-4 technical report","year":"2023"},{"key":"ref15","article-title":"Gemini: A family of highly capable multimodal models","year":"2024"},{"key":"ref16","first-page":"32694","article-title":"MiniLLM: Knowledge distillation of large language models","author":"Gu","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref17","first-page":"7690","article-title":"The expressive power of transformers with chain of thought","author":"Merrill","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref18","first-page":"23965","article-title":"Model soups: Averaging weights of multiple fine-tuned models improves accuracy without increasing inference time","volume":"162","author":"Wortsman","year":"2022","journal-title":"Proc. ICML"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2122"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1023\/A:1018054314350"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1006\/jcss.1997.1504"},{"key":"ref22","article-title":"RouterBench: A benchmark for multi-LLM routing system","author":"Hu","year":"2024"},{"key":"ref23","first-page":"18303","article-title":"Knowledge fusion of large language models","author":"Wan","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref24","article-title":"Mixtral of experts","author":"Jiang","year":"2024"},{"key":"ref25","first-page":"50905","article-title":"Reward model ensembles help mitigate overoptimization","author":"Coste","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref26","article-title":"Mixture-of-agents enhances large language model capabilities","author":"Wang","year":"2024"},{"key":"ref27","article-title":"Purifying large language models by ensembling a small language model","author":"Li","year":"2024"},{"key":"ref28","article-title":"Fusing finetuned models for better pretraining","author":"Choshen","year":"2022"},{"key":"ref29","article-title":"Editing models with task arithmetic","author":"Ilharco","year":"2023","journal-title":"Proc. ICLR"},{"key":"ref30","first-page":"22743","article-title":"AdaMerging: Adaptive model merging for multi-task learning","author":"Yang","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0310"},{"key":"ref32","first-page":"57755","article-title":"Language models are super Mario: Absorbing abilities from homologous models as a free lunch","author":"Yu","year":"2024","journal-title":"Proc. ICML"},{"key":"ref33","article-title":"Pack of LLMs: Model fusion at test-time via perplexity optimization","author":"Mavromatis","year":"2024","journal-title":"Proc. Colm"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-industry.36"},{"key":"ref35","article-title":"Checkpoint merging via Bayesian optimization in LLM pretraining","author":"Liu","year":"2024"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.175"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.762"},{"key":"ref38","article-title":"DPPA: Pruning method for large language model to model merging","author":"Zhu","year":"2024"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.248"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.35"},{"key":"ref41","article-title":"Concrete subspace learning based interference elimination for multi-task model fusion","author":"Tang","year":"2023"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3661357"},{"key":"ref43","article-title":"Evolutionary optimization of model merging recipes","author":"Akiba","year":"2024"},{"key":"ref44","article-title":"Knowledge fusion of chat LLMs: A preliminary technical report","author":"Wan","year":"2024"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.35"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.99"},{"key":"ref47","first-page":"42048","article-title":"Warm: On the benefits of weight averaged reward models","author":"Ram\u00e9","year":"2024","journal-title":"Proc. ICML"},{"key":"ref48","article-title":"Uncertainty-penalized reinforcement learning from human feedback with diverse reward Lora ensembles","author":"Zhai","year":"2023"},{"key":"ref49","article-title":"Scalable ensembling for mitigating reward overoptimisation","author":"Ahmed","year":"2024"},{"key":"ref50","article-title":"Helping or herding? Reward model ensembles mitigate but do not eliminate reward hacking","author":"Eisenstein","year":"2024","journal-title":"Proc. Colm"},{"key":"ref51","article-title":"Improving reinforcement learning from human feedback with efficient reward model ensemble","author":"Zhang","year":"2024"},{"key":"ref52","first-page":"45284","article-title":"Fusing models with complementary expertise","author":"Wang","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.792"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.261"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.552"},{"key":"ref56","article-title":"Large language models vote: Prompting for rare disease identification","author":"Oniani","year":"2024"},{"key":"ref57","article-title":"LLM routing with benchmark datasets","author":"Shnitzer","year":"2023","journal-title":"Proc. NeurIPS"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.109"},{"key":"ref59","first-page":"41348","article-title":"Hybrid LLM: Cost-efficient and quality-aware query routing","author":"Ding","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.insights-1.15"},{"key":"ref61","article-title":"Routoo: Learning to route to large language models effectively","author":"Mohammadshahi","year":"2024"},{"key":"ref62","article-title":"FrugalGPT: How to use large language models while reducing cost and improving performance","author":"Chen","year":"2023"},{"key":"ref63","article-title":"Language model cascades","author":"Dohan","year":"2022"},{"key":"ref64","first-page":"18858","article-title":"Mixture-of-experts meets instruction tuning: A winning combination for large language models","author":"Shen","year":"2024","journal-title":"Proc. ICLR"},{"key":"ref65","first-page":"606","article-title":"Fly-swat or cannon? Cost-effective language model choice via meta-modeling","volume":"35","author":"Sakota","year":"2024","journal-title":"ACM WSDM"},{"key":"ref67","article-title":"Merge, ensemble, and cooperate! A survey on collaborative strategies in the era of large language models","author":"Lu","year":"2024"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3787849"},{"key":"ref69","first-page":"131762","article-title":"A systematic literature review on AI safety: Identifying trends, challenges and future directions","volume":"12","author":"Zia","year":"2024","journal-title":"IEEE Trans. Artif. Intell."},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.24818\/ida-ql\/2019.5"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"ref73","article-title":"XLNet: Generalized autoregressive pretraining for language understanding","author":"Yang","year":"2019","journal-title":"Proc. NIPS"},{"key":"ref74","article-title":"LongFormer: The long-document transformer","author":"Beltagy","year":"2020"},{"issue":"70","key":"ref75","first-page":"1","article-title":"Scaling instruction-finetuned language models","volume":"25","author":"Chung","year":"2024","journal-title":"JMLR"},{"key":"ref76","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"ref77","article-title":"Generating long sequences with sparse transformers","author":"Child","year":"2019"},{"key":"ref78","article-title":"Language models are unsupervised multitask learners","author":"Radford","year":"2019"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1194"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1161"},{"key":"ref81","article-title":"Finetuned language models are zero-shot learners","author":"Wei","year":"2022","journal-title":"Proc. ICLR"},{"key":"ref82","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657740"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1056\/AIcs2400502"},{"key":"ref85","article-title":"Distilling the knowledge in a neural network","author":"Hinton","year":"2015","journal-title":"Proc. Nips"},{"key":"ref86","first-page":"10421","article-title":"Specializing smaller language models towards multi-step reasoning","volume":"202","author":"Fu","year":"2023","journal-title":"Proc. ICML"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3808"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.395"},{"key":"ref89","article-title":"GShard: Scaling giant models with conditional computation and automatic sharding","author":"Lepikhin","year":"2021","journal-title":"Proc. ICLR"},{"key":"ref90","article-title":"Glu variants improve transformer","author":"Shazeer","year":"2020"},{"key":"ref91","article-title":"Concrete problems in AI safety","author":"Amodei","year":"2016"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441"},{"key":"ref93","article-title":"Disagreement-regularized imitation learning","author":"Brantley","year":"2020","journal-title":"Proc. ICLR"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.309"},{"key":"ref95","article-title":"BARTScore: Evaluating generated text as text generation","author":"Yuan","year":"2021","journal-title":"NeurIPS"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.556"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1800"},{"key":"ref99","article-title":"Program of thoughts prompting: Disentangling computation from reasoning for numerical reasoning tasks","author":"Chen","year":"2023","journal-title":"Proc. TMLR"},{"key":"ref100","article-title":"Prompting GPT-3 to be reliable","author":"Si","year":"2023","journal-title":"Proc. ICLR"},{"key":"ref101","article-title":"Least-to-most prompting enables complex reasoning in large language models","author":"Zhou","year":"2023","journal-title":"Proc. ICLR"},{"key":"ref102","first-page":"62061","article-title":"OLMOE: Open mixture-of-experts language models","author":"Muennighoff","year":"2025","journal-title":"Proc. ICLR"},{"key":"ref103","article-title":"Measuring massive multitask language understanding","author":"Hendrycks","year":"2021","journal-title":"Proc. ICLR"},{"key":"ref104","article-title":"Training verifiers to solve math word problems","author":"Cobbe","year":"2021"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2020"},{"key":"ref106","article-title":"Program synthesis with large language models","author":"Austin","year":"2021"},{"key":"ref107","first-page":"92233","article-title":"Taid: Temporally adaptive interpolated distillation for efficient knowledge transfer in language models","author":"Shing","year":"2025","journal-title":"Proc. ICLR"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.clinicalnlp-1.35"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1056\/AIcs2400502"},{"key":"ref110","article-title":"BERTScore: Evaluating text generation with BERT","author":"Zhang","year":"2020","journal-title":"Proc. ICLR"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.859"},{"key":"ref112","article-title":"CodeBLEU: A method for automatic evaluation of code synthesis","author":"Ren","year":"2020"},{"key":"ref113","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. ICML","author":"Li","year":"2023"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9078688\/11635975\/11421084.pdf?arnumber=11421084","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T04:45:56Z","timestamp":1785818756000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11421084\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":113,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tai.2026.3670235","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8]]}}}