{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T12:14:01Z","timestamp":1774440841322,"version":"3.50.1"},"reference-count":79,"publisher":"China Science Publishing & Media Ltd.","issue":"1","content-domain":{"domain":["engine.scichina.com"],"crossmark-restriction":false},"short-container-title":["DI"],"published-print":{"date-parts":[[2026,3,1]]},"DOI":"10.3724\/2096-7004.di.2026.0055","type":"journal-article","created":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T09:05:43Z","timestamp":1774429543000},"page":"15-55","update-policy":"https:\/\/doi.org\/10.1360\/scp-crossmark-policy-page","source":"Crossref","is-referenced-by-count":0,"title":["Data Mixing for Large Language Models Pretraining: A Survey and Outlook"],"prefix":"10.3724","volume":"8","author":[{"given":"Zhuo","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxuan","family":"Miao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Deyi","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"2026","published-online":{"date-parts":[[2026,3,3]]},"reference":[{"key":"","unstructured":"Grattafiori A., Dubey A., Jauhri A., Pandey A., Kadian A., Al-Dahle A., Letman A., Mathur A., Schelten A., Vaughan A., Yang A., Fan A., Goyal A., Hartshorn A., Yang A., Mitra A., Sravankumar A., Korenev A., Hinsvark A., Rao A., Zhang A., Rodriguez A., Gregerson A., Spataru A., Roziere B., Biron B., Tang B., Caucheteux C., Nayak C., Bi C., Marra C., McConnell C., Keller C., Touret C., Wu C., Wong C., Ferrer C. C., Nikolaidis C., Allonsius D., Song D., Pintz D., Livshits D., Wyatt D., Esiobu D., Choudhary D., Mahajan D., Garcia-Olano D., Perino D., Hupkes D., Lakomkin E., AlBadawy E., Lobanova E., Dinan E., Smith E. M., Radenovic F., Guzm\u00e1n F., Zhang F., Synnaeve G., Lee G., Anderson G. L., Thattai G., Nail G., Mialon G., Pang G., Cucurell G., Nguyen H., Korevaar H., Xu H., Touvron H., Zarov I., Ibarra I. A., Kloumann I., Misra I., Evtimov I., Zhang J., Copet J., Lee J., Geffert J., Vranes J., Park J., Mahadeokar J., Shah J., van der Linde J., Billock J., Hong J., Lee J., Fu J., Chi J., Huang J., Liu J., Wang J., Yu J., Bitton J., Spisak J., Park J., Rocca J., Johnstun J., Saxe J., Jia J., et al., \u201cThe llama 3 herd of models,\u201d arXiv preprint arXiv: 2407.21783, 2024. [Online]. Available: https:\/\/arxiv.org\/abs\/2407.21783."},{"key":"","unstructured":"Yang A., Li A., Yang B., Zhang B., Hui B., Zheng B., Yu B., Gao C., Huang C., Lv C., Zheng C., Liu D., Zhou F., Huang F., Hu F., Ge H., Wei H., Lin H., Tang J., Yang J., Tu J., Zhang J., Yang J., Yang J., Zhou J., Zhou J., Lin J., Dang K., Bao K., Yang K., Yu L., Deng L., Li M., Xue M., Li M., Zhang P., Wang P., Zhu Q., Men R., Gao R., Liu S., Luo S., Li T., Tang T., Yin W., Ren X., Wang X., Zhang X., Ren X., Fan Y., Su Y., Zhang Y., Zhang Y., Wan Y., Liu Y., Wang Z., Cui Z., Zhang Z., Zhou Z., and Qiu Z., \u201cQwen3 technical report,\u201d arXiv preprint arXiv: 2505.09388, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2505.09388."},{"key":"","unstructured":"Bi X., Zang Y., Wang Y., Mao Y., Wang Y., Guo Y., Liu H., et al., \u201cDeepseek llm: Scaling open-source language models with longtermism,\u201d arXiv preprint arXiv: 2401.02954, 2024. [Online]. Available: https:\/\/arxiv.org\/abs\/2401.02954."},{"key":"","unstructured":"Achiam J., Anadkat S., Ang S. B. M., et al., \u201cGpt-4 technical report,\u201d arXiv preprint arXiv: 2303.08774, 2023. [Online]. Available: https:\/\/arxiv.org\/abs\/2303.08774."},{"key":"","unstructured":"Sun H., Jin R., Xu S., Pan L., Supryadi, Cui M., Du J., Lei Y., Yang L., Shi L., Xiao J., Zhu S., and Xiong D., \u201cFuxiTranyu: A multilingual large language model trained with balanced data,\u201d in Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing: Industry Track, Dernoncourt F., Preo\u0163iuc-Pietro D., and Shimorina A., Eds. Miami, Florida, US: Association for Computational Linguistics, Nov. 2024, pp. 1499\u20131522. [Online]. Available: https:\/\/aclanthology.org\/2024.emnlp-industry.110\/."},{"key":"","unstructured":"Vaswani A., Shazeer N., Parmar N., Uszkoreit J., Jones L., Gomez A. N., Kaiser \u0141., and Polosukhin I., \u201cAttention is all you need,\u201d in Advances in Neural Information Processing Systems, Guyon I., von Luxburg U., Bengio S., Wallach H., Fergus R., Vishwanathan S. V. N., and Garnett R., Eds., vol. 30. Curran Associates, Inc., 2017, pp. 5998\u20136008. [Online]. Available: https:\/\/papers.neurips.cc\/paper\/7181-attention-is-all-you-need."},{"key":"","unstructured":"Li J., Fang A., Smyrnis G., Ivgi M., Jordan M., Gadre S. Y., Bansal H., Guha E. K., Keh S., Arora K., Garg S., Xin R., Muennighoff N., Heckel R., Mercat J., Chen M., Gururangan S., Wortsman M., Albalak A., Bitton Y., Nezhurina M., Abbas A., Hsieh C.-Y., Ghosh D., Gardner J., Kilian M., Zhang H., Shao R., Pratt S., Daras G., Marathe K., Gokaslan A., Zhang J., Chandu K., Nguyen T., Vasiljevic I., Kakade S., Song S., Sanghavi S., Faghri F., Oh S., Zettlemoyer L., Lo K., El-Nouby A., Pouransari H., Toshev A., Wang S., Groeneveld D., Soldaini L., Koh P. W., Jitsev J., Kollar T., Dimakis A. G., Carmon Y., Dave A., Schmidt L., and Shankar V., \u201cDatacomp-lm: In search of the next generation of training sets for language models,\u201d in Advances in Neural Information Processing Systems (Datasets and Benchmarks Track), vol. 37, 2024. [Online]. Available: https:\/\/papers.nips.cc\/paper_files\/paper\/2024\/hash\/19e4ea30dded58259665db375885e412-Abstract-Datasets_and_Benchmarks_Track.html."},{"key":"","unstructured":"Parmar J., Prabhumoye S., Jennings J., Liu B., Jhunjhunwala A., Wang Z., Patwary M., Shoeybi M., and Catanzaro B., \u201cData, data everywhere: A guide for pretraining dataset construction,\u201d in Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, Al-Onaizan Y., Bansal M., and Chen Y.-N., Eds. Miami, Florida, USA: Association for Computational Linguistics, Nov. 2024, pp. 10671\u201310695. [Online]. Available: https:\/\/aclanthology.org\/2024.emnlp-main.596\/."},{"key":"","unstructured":"Albalak A., Elazar Y., Xie S. M., Longpre S., Lambert N., Wang X., Muennighoff N., Hou B., Pan L., Jeong H., Raffel C., Chang S., Hashimoto T., and Wang W. Y., \u201cA survey on data selection for language models,\u201d Transactions on Machine Learning Research, July 2024, tMLR (certified Survey Featured). [Online]. Available: https:\/\/openreview.net\/forum?id=XfHWcNTSHp."},{"key":"","unstructured":"Longpre S., Yauney G., Reif E., Lee K., Roberts A., Zoph B., Zhou D., Wei J., Robinson K., Mimno D., and Ippolito D., \u201cA pretrainer\u2019s guide to training data: Measuring the effects of data age, domain coverage, quality, & toxicity,\u201d in Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), Duh K., Gomez H., and Bethard S., Eds. Mexico City, Mexico: Association for Computational Linguistics, Jun. 2024, pp. 3245\u20133276. [Online]. Available: https:\/\/aclanthology.org\/2024.naacl-long.179\/."},{"key":"","unstructured":"Du N., Huang Y., Dai A. M., Tong S., Lepikhin D., Xu Y., Krikun M., Zhou Y., Yu A. W., Firat O., Zoph B., Fedus L., Bosma M. P., Zhou Z., Wang T., Wang E., Webster K., Pellat M., Robinson K., Meier-Hellstern K., Duke T., Dixon L., Zhang K., Le Q., Wu Y., Chen Z., and Cui C., \u201cGLaM: Efficient scaling of language models with mixture-of-experts,\u201d in Proceedings of the 39 th  International Conference on Machine Learning, ser. Proceedings of Machine Learning Research, Chaudhuri K., Jegelka S., Song L., Szepesvari C., Niu G., and Sabato S., Eds., vol. 162. PMLR, 17\u201323 Jul 2022, pp. 5547\u20135569. [Online]. Available: https:\/\/proceedings.mlr.press\/v162\/du22c.html."},{"key":"","unstructured":"Chung H. W., Constant N., Garcia X., Roberts A., Tay Y., Narang S., and Firat O., \u201cUnimax: Fairer and more effective language sampling for large-scale multilingual pretraining,\u201d arXiv preprint arXiv: 2304.09151, 2023. [Online]. Available: https:\/\/arxiv.org\/abs\/2304.09151."},{"key":"","unstructured":"Held W., Paranjape B., Koura P. S., Lewis M., Zhang F., and Mihaylov T., \u201cOptimizing pretraining data mixtures with llm-estimated utility,\u201d arXiv preprint arXiv: 2501.11747, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2501.11747."},{"key":"","unstructured":"Xie S. M., Pham H., Dong X., Du N., Liu H., Lu Y., Liang P., Le Q. V., Ma T., and Yu A. W., \u201cDoremi: Optimizing data mixtures speeds up language model pretraining,\u201d in Advances in Neural Information Processing Systems, vol. 36, 2023. [Online]. Available: https:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/dcba6be91359358c2355cd920da3fcbd-Abstract-Conference.html."},{"key":"","unstructured":"Fan S., Pagliardini M., and Jaggi M., \u201cDOGE: Domain reweighting with generalization estimation,\u201d in Proceedings of the 41 st  International Conference on Machine Learning, ser. Proceedings of Machine Learning Research, Salakhutdinov R., Kolter Z., Heller K., Weller A., Oliver N., Scarlett J., and Berkenkamp F., Eds., vol. 235. PMLR, 21\u201327 Jul 2024, pp. 12895\u201312915. [Online]. Available: https:\/\/proceedings.mlr.press\/v235\/fan24e.html."},{"key":"","unstructured":"Ye J., Liu P., Sun T., Zhan J., Zhou Y., and Qiu X., \u201cData mixing laws: Optimizing data mixtures by predicting language modeling performance,\u201d in The Thirteenth International Conference on Learning Representations (ICLR 2025), 2025. [Online]. Available: https:\/\/openreview.net\/forum?id=jjCB27TMK3."},{"key":"","unstructured":"Ge C., Ma Z., Chen D., Li Y., and Ding B., \u201cBimix: Bivariate data mixing law for language model pretraining,\u201d arXiv preprint arXiv: 2405.14908, 2024. [Online]. Available: https:\/\/arxiv.org\/abs\/2405.14908."},{"key":"","unstructured":"Kang F., Sun Y., Wen B., Chen S., Song D., Mahmood R., and Jia R., \u201cAutoscale: Scale-aware data mixing for pre-training llms,\u201d in Conference on Language Modeling (COLM 2025), 2025. [Online]. Available: https:\/\/openreview.net\/forum?id=rujwIvjooA."},{"key":"","unstructured":"Shukor M., Bethune L., Busbridge D., Grangier D., Fini E., El-Nouby A., and Ablin P., \u201cScaling laws for optimal data mixtures,\u201d arXiv preprint arXiv: 2507.09404, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2507.09404."},{"key":"","unstructured":"Liu Q., Zheng X., Muennighoff N., Zeng G., Dou L., Pang T., Jiang J., and Lin M., \u201cRegmix: Data mixture as regression for language model pre-training,\u201d in International Conference on Learning Representations (ICLR 2025), 2025. [Online]. Available: https:\/\/proceedings.iclr.cc\/paper_files\/paper\/2025\/hash\/5f67d864aae6115374fed7beddd119e0-Abstract-Conference.html."},{"key":"","unstructured":"Thudi A., Rovers E., Ruan Y., Thrush T., and Maddison C. J., \u201cMixmin: Finding data mixtures via convex minimization,\u201d in International Conference on Machine Learning (ICML 2025), Poster, 2025. [Online]. Available: https:\/\/openreview.net\/forum?id=wpaxYGgp2n."},{"key":"","unstructured":"Belenki L., Agarwal A., Shi T., and Toutanova K., \u201cOptimizing pre-training data mixtures with mixtures of data expert models,\u201d in Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Che W., Nabende J., Shutova E., and Pilehvar M. T., Eds. Vienna, Austria: Association for Computational Linguistics, Jul. 2025, pp. 32570\u201332587. [Online]. Available: https:\/\/aclanthology.org\/2025.acl-long.1564\/."},{"key":"","unstructured":"Yen T., Siah A. W. T., Chen H., Peng T., Guetta C. D., and Namkoong H., \u201cData mixture optimization: A multi-fidelity multi-scale bayesian framework,\u201d in NeurIPS 2025, Poster, 2025. [Online]. Available: https:\/\/openreview.net\/forum?id=Kvsa8ZXd0W."},{"key":"","unstructured":"Chen S., Ouyang X., Pearce M. A. L., Hartvigsen T., and Schwarz J. R., \u201cAdmire-bayesopt: Accelerated data mixture re-weighting for language models with bayesian optimization,\u201d arXiv preprint arXiv: 2508.11551, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2508.11551."},{"key":"","unstructured":"Albalak A., Pan L., Raffel C., and Wang W. Y., \u201cEfficient online data mixing for language model pre-training,\u201d arXiv preprint arXiv: 2312.02406, 2023. [Online]. Available: https:\/\/arxiv.org\/abs\/2312.02406."},{"key":"","unstructured":"Jiang Y., Zhou A., Feng Z., Malladi S., and Kolter J. Z., \u201cAdaptive data optimization: Dynamic sample selection with scaling laws,\u201d in The Thirteenth International Conference on Learning Representations, 2025, iCLR 2025 Poster. [Online]. Available: https:\/\/iclr.cc\/virtual\/2025\/poster\/29145."},{"key":"","unstructured":"Luo Z., Zhang X., Liu X., Li H., Gong Y., Chen Q., and Cheng P., \u201cVelocitune: A velocity-based dynamic domain reweighting method for continual pre-training,\u201d in Proceedings of the 63 rd  Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Vienna, Austria: Association for Computational Linguistics, Jul. 2025, pp. 16644\u201316656. [Online]. Available: https:\/\/aclanthology.org\/2025.acl-long.813\/."},{"key":"","unstructured":"Li Z., Deng Y., Zhong P., Razaviyayn M., and Mirrokni V., \u201cPike: Adaptive data mixing for multitask learning under low gradient conflicts,\u201d arXiv preprint arXiv: 2502.06244, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2502.06244."},{"key":"","unstructured":"Fan S., Glarou M. I., and Jaggi M., \u201cGrape: Optimize data mixture for group robust multi-target adaptive pretraining,\u201d in Thirty-ninth Conference on Neural Information Processing Systems, 2025, neurIPS 2025 Poster. [Online]. Available: https:\/\/openreview.net\/forum?id=JRmIvBcnWc."},{"key":"","unstructured":"Ma J., Dang C., and Liao M., \u201cActor-critic based online data mixing for language model pre-training,\u201d arXiv preprint arXiv: 2505.23878, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2505.23878."},{"key":"","unstructured":"Yang K., Liu X., Ji L., Li H., Gong Y., Cheng P., and Yang M., \u201cData mixing agent: Learning to re-weight domains for continual pre-training,\u201d arXiv preprint arXiv: 2507.15640, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2507.15640."},{"key":"","unstructured":"Wang Y., Liu B., Liu F., Guo Y., Deng J., Wu X., Zhou W., Zhou X., and Wang T., \u201cTikmix: Take data influence into dynamic mixture for language model pre-training,\u201d arXiv preprint arXiv: 2508.17677, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2508.17677."},{"key":"","unstructured":"Rae J. W., Borgeaud S., Cai T., Millican K., Hoffmann J., Song F., Aslanides J., Henderson S., Ring R., Young S., Rutherford E., Hennigan T., Menick J., Cassirer A., Powell R., van den Driessche G., Hendricks L. A., Rauh M., Huang P.-S., Glaese A., Welbl J., Dathathri S., Huang S., Uesato J., Mellor J., Higgins I., Creswell A., McAleese N., Wu A., Elsen E., Jayakumar S., Buchatskaya E., Budden D., Sutherland E., Simonyan K., Paganini M., Sifre L., Martens L., Li X. L., Kuncoro A., Nematzadeh A., Gribovskaya E., Donato D., Lazaridou A., Mensch A., Lespiau J.-B., Tsimpoukelli M., Grigorev N., Fritz D., Sottiaux T., Pajarskas M., Pohlen T., Gong Z., Toyama D., de Masson d\u2019Autume C., Li Y., Terzi T., Mikulik V., Babuschkin I., Clark A., de Las Casas D., Guy A., Jones C., Bradbury J., Johnson M., Hechtman B., Weidinger L., Gabriel I., Isaac W., Lockhart E., Osindero S., Rimell L., Dyer C., Vinyals O., Ayoub K., Stanway J., Bennett L., Hassabis D., Kavukcuoglu K., and Irving G., \u201cScaling language models: Methods, analysis & insights from training gopher,\u201d arXiv preprint arXiv: 2112.11446, 2021. [Online]. Available: https:\/\/arxiv.org\/abs\/2112.11446."},{"key":"","unstructured":"Groeneveld D., Beltagy I., Walsh E. P., Bhagia A., Kinney R., Tafjord O., Jha A. H., Ivison H., Magnusson I., Wang Y., Arora S., Atkinson D., Authur R., Chandu K., Cohan A., Dumas J., Elazar Y., Gu Y., Hessel J., Khot T., Merrill W., Morrison J., Muennighoff N., Naik A., Nam C., Peters M. E., Pyatkin V., Ravichander A., Schwenk D., Shah S., Saunders W. H., Schwenk D., et al., \u201cOLMo: Accelerating the science of language models,\u201d in Proceedings of the 62 nd  Annual Meeting of the Association for Computational Linguistics (Volume 3: System Demonstrations). Bangkok, Thailand: Association for Computational Linguistics, Aug. 2024, pp. 15789\u201315809. [Online]. Available: https:\/\/aclanthology.org\/2024.acl-long.841\/."},{"key":"","unstructured":"Devlin J., Chang M.-W., Lee K., and Toutanova K., \u201cBert: Pre-training of deep bidirectional transformers for language understanding,\u201d in Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Minneapolis, Minnesota: Association for Computational Linguistics, Jun. 2019, pp. 4171\u20134186. [Online]. Available: https:\/\/aclanthology.org\/N19-1423\/."},{"key":"","unstructured":"Luo F., Li W., Liu Y., Huang B., Wang Z., Guo J., Sun X.-L., and Liu W.-Y., \u201cVeco: Variable and flexible cross-lingual pre-training for language understanding and generation,\u201d in Proceedings of the 59 th  Annual Meeting of the Association for Computational Linguistics. Online: Association for Computational Linguistics, Aug. 2021, pp. 3980\u20133993. [Online]. Available: https:\/\/aclanthology.org\/2021.acl-long.308\/."},{"key":"","unstructured":"Conneau A., Khandelwal K., Goyal N., Chaudhary V., Wenzek G., Guzm\u00e1n F., Grave E., Ott M., Zettlemoyer L., and Stoyanov V., \u201cUnsupervised cross-lingual representation learning at scale,\u201d in Proceedings of the 58 th  Annual Meeting of the Association for Computational Linguistics. Online: Association for Computational Linguistics, Jul. 2020, pp. 8440\u20138451. [Online]. Available: https:\/\/aclanthology.org\/2020.acl-main.747\/."},{"key":"","unstructured":"Chi Z., Dong L., Wei F., Wang W., Yang N., Singhal S., Wang S., Song X., Ma S., Huang S., Zhou M., and Wei F., \u201cXlm-e: Cross-lingual language model pre-training via ELECTRA,\u201d in Proceedings of the 60 th  Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Dublin, Ireland: Association for Computational Linguistics, May 2022. [Online]. Available: https:\/\/aclanthology.org\/2022.acl-long.427\/."},{"key":"","unstructured":"Liu Y., Gu J., Goyal N., Li X., Edunov S., Ghazvininejad M., Lewis M., and Zettlemoyer L., \u201cMultilingual denoising pre-training for neural machine translation,\u201d Transactions of the Association for Computational Linguistics, vol. 8, pp. 726\u2013742, 2020. [Online]. Available: https:\/\/aclanthology.org\/2020.tacl-1.47\/."},{"key":"","unstructured":"Nussbaum Z. and Duderstadt B., \u201cTraining sparse mixture of experts text embedding models,\u201d arXiv preprint arXiv: 2502.07972, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2502.07972."},{"key":"","unstructured":"Xue L., Constant N., Roberts A., Kale M., Al-Rfou R., Siddhant A., Barua A., and Raffel C., \u201cmt5: A massively multilingual pre-trained text-to-text transformer,\u201d in Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Online: Association for Computational Linguistics, Jun. 2021, pp. 483\u2013498. [Online]. Available: https:\/\/aclanthology.org\/2021.naacl-main.41\/."},{"key":"","unstructured":"Marone M., Weller O., Fleshman W., Yang E., Lawrie D., and Van Durme B., \u201cmmbert: A modern multilingual encoder with annealed language learning,\u201d arXiv preprint arXiv: 2509.06888, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2509.06888."},{"key":"","unstructured":"Markowitz H., \u201cPortfolio selection,\u201d The Journal of Finance, vol. 7, no. 1, pp. 77\u201391, March 1952. [Online]. Available: https:\/\/onlinelibrary.wiley.com\/doi\/10.1111\/j.1540-6261.1952.tb01525.x."},{"key":"","unstructured":"Boyd S., Johansson K., Kahn R., Schiele P., and Schmelzer T., \u201cMarkowitz portfolio construction at seventy,\u201d The Journal of Portfolio Management, vol. 50, no. 8, pp. 117\u2013160, 2024, special Issue Dedicated to Harry Markowitz. [Online]. Available: https:\/\/www.pm-research.com\/content\/iijpormgmt\/50\/8\/117."},{"key":"","unstructured":"Nemirovski A., Juditsky A., Lan G., and Shapiro A., \u201cRobust stochastic approximation approach to stochastic programming,\u201d SIAM Journal on Optimization, vol. 19, no. 4, pp. 1574\u20131609, 2009. [Online]. Available: https:\/\/epubs.siam.org\/doi\/10.1137\/070704277."},{"key":"","unstructured":"Sagawa S., Koh P. W., Hashimoto T. B., and Liang P., \u201cDistributionally robust neural networks for group shifts: On the importance of regularization for worst-case generalization,\u201d in Proceedings of the International Conference on Learning Representations (ICLR), 2020. [Online]. Available: https:\/\/openreview.net\/forum?id=ryxGuJrFvS."},{"key":"","unstructured":"Oren Y., Sagawa S., Hashimoto T. B., and Liang P., \u201cDistributionally robust language modeling,\u201d in Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9 th  International Joint Conference on Natural Language Processing (EMNLP-IJCNLP). Hong Kong, China: Association for Computational Linguistics, Nov. 2019, pp. 4227\u20134237. [Online]. Available: https:\/\/aclanthology.org\/D19-1432\/."},{"key":"","unstructured":"Mindermann S., Bengs A., Hooper J., Gal Y., and Weller A., \u201cPrioritized training on points that are learnable, worth learning, and not yet learnt,\u201d in Proceedings of the 39 th  International Conference on Machine Learning, ser. Proceedings of Machine Learning Research, Chaudhuri K., Jegelka S., Song L., Szepesvari C., Niu G., and Sabato S., Eds., vol. 162. PMLR, 2022, pp. 15526\u201315544. [Online]. Available: https:\/\/proceedings.mlr.press\/v162\/mindermann22a.html."},{"key":"","unstructured":"Hoffmann J., Borgeaud S., Mensch A., Buchatskaya E., Cai T., Rutherford E., de Las Casas D., Hendricks L. A., Welbl J., Clark A., Hennigan T., Noland E., Millican K., van den Driessche G., Damoc B., Guy A., Osindero S., Simonyan K., Elsen E., Rae J., Vinyals O., and Sifre L., \u201cAn empirical analysis of compute-optimal large language model training,\u201d in Advances in Neural Information Processing Systems, vol. 35, 2022. [Online]. Available: https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/hash\/c1e2faff6f588870935f114ebe04a3e5-Abstract-Conference.html."},{"key":"","unstructured":"Ke G., Meng Q., Finley T., Wang T., Chen W., Ma W., Ye Q., and Liu T.-Y., \u201cLightgbm: A highly efficient gradient boosting decision tree,\u201d in Advances in Neural Information Processing Systems 30 (NIPS 2017), 2017, pp. 3146\u20133154. [Online]. Available: https:\/\/papers.nips.cc\/paper_files\/paper\/2017\/hash\/6449f44a102fde848669bdd9eb6b76fa-Abstract.html."},{"key":"","unstructured":"Rasmussen C. E. and Williams C. K. I., Gaussian Processes for Machine Learning. MIT Press, 2006. [Online]. Available: https:\/\/direct.mit.edu\/books\/oa-monograph\/2320\/Gaussian-Processes-for-Machine-Learning."},{"key":"","unstructured":"Broomhead D. S. and Lowe D., \u201cMultivariable functional interpolation and adaptive networks,\u201d Complex Systems, vol. 2, no. 3, pp. 321\u2013355, 1988. [Online]. Available: https:\/\/www.complex-systems.com\/abstracts\/v02i03a05\/."},{"key":"","unstructured":"Wu J., Toscano-Palmerin S., Frazier P. I., and Wilson A. G., \u201cPractical multi-fidelity bayesian optimization for hyperparameter tuning,\u201d in Proceedings of The 35 th  Uncertainty in Artificial Intelligence Conference, ser. Proceedings of Machine Learning Research, Adams R. P. and Gogate V., Eds., vol. 115. PMLR, 22\u201325 Jul 2020, pp. 788\u2013798. [Online]. Available: https:\/\/proceedings.mlr.press\/v115\/wu20a.html."},{"key":"","unstructured":"Auer P., Cesa-Bianchi N., Freund Y., and Schapire R. E., \u201cThe nonstochastic multiarmed bandit problem,\u201d SIAM Journal on Computing, vol. 32, no. 1, pp. 48\u201377, 2002. [Online]. Available: https:\/\/epubs.siam.org\/doi\/10.1137\/S0097539701398375."},{"key":"","unstructured":"Barto A. G., Sutton R. S., and Anderson C. W., \u201cNeuronlike adaptive elements that can solve difficult learning control problems,\u201d IEEE Transactions on Systems, Man, and Cybernetics, vol. SMC-13, no. 5, pp. 834\u2013846, 1983. [Online]. Available: https:\/\/incompleteideas.net\/papers\/barto-sutton-anderson-83.pdf."},{"key":"","unstructured":"Kumar A., Zhou A., Tucker G., and Levine S., \u201cConservative q-learning for offline reinforcement learning,\u201d in Advances in Neural Information Processing Systems, vol. 33, 2020. [Online]. Available: https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/0d2b2061826a5df3221116a5085a6052-Abstract.html."},{"key":"","unstructured":"Chen M. F., Hu M. Y., Lourie N., Cho K., and R\u00e9 C., \u201cAioli: A unified optimization framework for language model data mixing,\u201d in International Conference on Learning Representations (ICLR), 2025. [Online]. Available: https:\/\/openreview.net\/forum?id=sZGZJhaNSe."},{"key":"","unstructured":"Peng J., Zhuang X., Qiu J., Ma R., Yu J., Bai T., and He C., \u201cUnsupervised topic models are data mixers for pre-training language models,\u201d arXiv preprint arXiv: 2502.16802, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2502.16802."},{"key":"","unstructured":"MacQueen J. B., \u201cSome methods for classification and analysis of multivariate observations,\u201d in Proceedings of the Fifth Berkeley Symposium on Mathematical Statistics and Probability, Volume 1: Statistics, Le Cam L. M. and Neyman J., Eds. Berkeley, CA: University of California Press, 1967, pp. 281\u2013297. [Online]. Available: https:\/\/projecteuclid.org\/ebooks\/berkeley-symposium-on-mathematical-statistics-and-probability\/Proceedings-of-the-Fifth-Berkeley-Symposium-on-Mathematical-Statistics-and\/chapter\/Some-methods-for-classification-and-analysis-of-multivariate-observations\/bsmsp\/1200512992."},{"key":"","unstructured":"Arthur D. and Vassilvitskii S., \u201ck-means++: The advantages of careful seeding,\u201d in Proceedings of the Eighteenth Annual ACM-SIAM Symposium on Discrete Algorithms (SODA \u201907). New Orleans, Louisiana, USA: Society for Industrial and Applied Mathematics, 2007, pp. 1027\u20131035. [Online]. Available: https:\/\/dl.acm.org\/doi\/10.5555\/1283383.1283494."},{"key":"","unstructured":"Diao S., Yang Y., Fu Y., Dong X., Su D., Kliegl M., Chen Z., Belcak P., Suhara Y., Yin H., Patwary M., Lin Y., Kautz J., and Molchanov P., \u201cClimb: Clustering-based iterative data mixture bootstrapping for language model pre-training,\u201d arXiv preprint arXiv: 2504.13161, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2504.13161."},{"key":"","unstructured":"Zhang M., Tissue H., Wang L., and Qiu X., \u201cDomain2vec: Vectorizing datasets to find the optimal data mixture without training,\u201d arXiv preprint arXiv: 2506.10952, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2506.10952."},{"key":"","unstructured":"Shao Y., Li L., Fei Z., Yan H., Lin D., and Qiu X., \u201cBalanced data sampling for language model training with clustering,\u201d in Findings of the Association for Computational Linguistics: ACL 2024, Ku L.-W., Martins A., and Srikumar V., Eds. Bangkok, Thailand: Association for Computational Linguistics, Aug. 2024, pp. 14 012\u201314 023. [Online]. Available: https:\/\/aclanthology.org\/2024.findings-acl.833\/."},{"key":"","unstructured":"Fan S., Grangier D., and Ablin P., \u201cDynamic gradient alignment for online data mixing,\u201d OpenReview preprint, 2025, iCLR 2025 submission. [Online]. Available: https:\/\/openreview.net\/forum?id=O3SatrdL97."},{"key":"","unstructured":"Shi W., Zhang J., Wu Y., Fang J., Zhang S., Zhao Y., Chen H., Zhang R., Cui Y., Zhu J., Han S., Xu J., and Zhou X., \u201cDids: Domain impact-aware data sampling for large language model training,\u201d in Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, Christodoulopoulos C., Chakraborty T., Rose C., and Peng V., Eds. Suzhou, China: Association for Computational Linguistics, Nov. 2025, pp. 4330\u20134350. [Online]. Available: https:\/\/aclanthology.org\/2025.emnlp-main.215\/."},{"key":"","unstructured":"Zhu X., Gu Z., Zheng S., Wang T., Li T., Feng H., and Xiao Y., \u201cToremi: Topic-aware data reweighting for dynamic pre-training data selection,\u201d arXiv preprint arXiv: 2504.00695, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2504.00695."},{"key":"","unstructured":"Ge A., Huang T.-H., Cooper J., Trost A., Chu Z., Namburi GNVV S. S. S., Cai Z., Park K., Roberts N., and Sala F., \u201cR&b: Domain regrouping and data mixture balancing for efficient foundation model training,\u201d arXiv preprint arXiv: 2505.00358, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2505.00358."},{"key":"","unstructured":"Chen M. F., Roberts N., Bhatia K., Wang J., Zhang C., Sala F., and R\u00e9 C., \u201cSkill-it! a data-driven skills framework for understanding and training language models,\u201d in Advances in Neural Information Processing Systems, vol. 36, 2023. [Online]. Available: https:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/70b8505ac79e3e131756f793cd80eb8d-Abstract-Conference.html."},{"key":"","unstructured":"Ming C., Qu C., Cai M., Pei Q., Pan Z., Li Y., Duan X., Wu L., and He C., \u201cIdeal: Data equilibrium adaptation for multi-capability language model alignment,\u201d arXiv preprint arXiv: 2505.12762, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2505.12762."},{"key":"","unstructured":"Corrado N. E., Katz-Samuels J., Devraj A. M., Yun H., Zhang C., Xu Y., Pan Y., Yin B., and Chilimbi T., \u201cAutoMixAlign: Adaptive data mixing for multi-task preference optimization in LLMs,\u201d in Proceedings of the 63 rd  Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Che W., Nabende J., Shutova E., and Pilehvar M. T., Eds. Vienna, Austria: Association for Computational Linguistics, Jul. 2025, pp. 20234\u201320258. [Online]. Available: https:\/\/aclanthology.org\/2025.acl-long.990\/."},{"key":"","unstructured":"Xie W., Tonin F., and Cevher V., \u201cChameleon: A flexible data-mixing framework for language model pretraining and finetuning,\u201d arXiv preprint arXiv: 2505.24844, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2505.24844."},{"key":"","unstructured":"Bach F., \u201cSharp analysis of low-rank kernel matrix approximations,\u201d in Proceedings of the 26 th  Annual Conference on Learning Theory, ser. Proceedings of Machine Learning Research, Shalev-Shwartz S. and Steinwart I., Eds., vol. 30. Princeton, NJ, USA: PMLR, 12\u201314 Jun 2013, pp. 185\u2013209. [Online]. Available: https:\/\/proceedings.mlr.press\/v30\/Bach13.html."},{"key":"","unstructured":"Kaplan J., McCandlish S., Henighan T., Brown T. B., Chess B., Child R., Gray S., Radford A., Wu J., and Amodei D., \u201cScaling laws for neural language models,\u201d arXiv preprint arXiv: 2001.08361, 2020. [Online]. Available: https:\/\/arxiv.org\/abs\/2001.08361."},{"key":"","unstructured":"Xi X., Kong D., Yang J., Yang J., Chen Z., Wang W., Wang J., Cai X., Zhang S., and Ye W., \u201cSampleMix: A sample-wise pre-training data mixing strategy by coordinating data quality and diversity,\u201d in Findings of the Association for Computational Linguistics: EMNLP 2025, Christodoulopoulos C., Chakraborty T., Rose C., and Peng V., Eds. Suzhou, China: Association for Computational Linguistics, Nov. 2025, pp. 13736\u201313758. [Online]. Available: https:\/\/aclanthology.org\/2025.findings-emnlp.741\/."},{"key":"","unstructured":"Chen Z., Lau G. K. R., Foo C.-S., and Low B. K. H., \u201cDuet: Optimizing training data mixtures via feedback from unseen evaluation tasks,\u201d arXiv preprint arXiv: 2502.00270, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2502.00270."},{"key":"","unstructured":"Liu F., Zhou W., Liu B., Yu Z., Zhang Y., Lin H., Yu Y., Zhou X., Wang T., and Cao Y., \u201cQuadmix: Quality-diversity balanced data selection for efficient llm pretraining,\u201d arXiv preprint arXiv: 2504.16511, 2025. [Online]. Available: https:\/\/arxiv.org\/abs\/2504.16511."},{"key":"","unstructured":"Hayase J., Liu A., Choi Y., Oh S., and Smith N. A., \u201cData mixture inference attack: BPE tokenizers reveal training data compositions,\u201d in Advances in Neural Information Processing Systems 37 (NeurIPS 2024), 2024. [Online]. Available: https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/hash\/10e6dfea9a673bef4a7b1cb9234891bc-Abstract-Conference.html."},{"key":"","unstructured":"Sennrich R., Haddow B., and Birch A., \u201cNeural machine translation of rare words with subword units,\u201d in Proceedings of the 54 th  Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), Erk K. and Smith N. A., Eds. Berlin, Germany: Association for Computational Linguistics, Aug. 2016, pp. 1715\u20131725. [Online]. Available: https:\/\/aclanthology.org\/P16-1162\/."},{"key":"","unstructured":"Liang H., Zhao K., Yang Y., Cui B., Dong G., Zhou Z., and Zhang W., \u201cData proportion detection for optimized data management for large language models,\u201d arXiv preprint arXiv: 2409.17527, 2024. [Online]. Available: https:\/\/arxiv.org\/abs\/2409.17527."}],"container-title":["Data Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.sciengine.com\/doi\/pdf\/F635D24C18904B0B8F5F8D0CDAF39A79","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2026.0055","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/pdf\/F635D24C18904B0B8F5F8D0CDAF39A79","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T09:05:43Z","timestamp":1774429543000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2026.0055"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,1]]},"references-count":79,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,3,3]]},"published-print":{"date-parts":[[2026,3,1]]}},"URL":"https:\/\/doi.org\/10.3724\/2096-7004.di.2026.0055","relation":{},"ISSN":["2096-7004"],"issn-type":[{"value":"2096-7004","type":"print"}],"subject":[],"published":{"date-parts":[[2026,3,1]]}}}