{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T01:51:18Z","timestamp":1772934678066,"version":"3.50.1"},"reference-count":50,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T00:00:00Z","timestamp":1765152000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T00:00:00Z","timestamp":1765152000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100004731","name":"Zhejiang Provincial Natural Science Foundation of China","doi-asserted-by":"publisher","award":["LDT23F01013F01"],"award-info":[{"award-number":["LDT23F01013F01"]}],"id":[{"id":"10.13039\/501100004731","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,12,8]]},"DOI":"10.1109\/bigdata66926.2025.11401489","type":"proceedings-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T20:57:57Z","timestamp":1772830677000},"page":"5553-5562","source":"Crossref","is-referenced-by-count":0,"title":["Targeted Knowledge Enhancement: A Systematic Continual Pre-Training Approach for Effective Domain Adaptation"],"prefix":"10.1109","author":[{"given":"Yiqun","family":"Wang","sequence":"first","affiliation":[{"name":"College of Biomedical Engineering and Instrument Science, Zhejiang University,Hangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaoqun","family":"Wan","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, University of Science and Technology of China,Hefei,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiang","family":"Tian","sequence":"additional","affiliation":[{"name":"College of Biomedical Engineering and Instrument Science, Zhejiang University,Hangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuesong","family":"Liu","sequence":"additional","affiliation":[{"name":"College of Biomedical Engineering and Instrument Science, Zhejiang University,Hangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yaowu","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Biomedical Engineering and Instrument Science, Zhejiang University,Hangzhou,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref2","first-page":"arXiv-2407","article-title":"The llama 3 herd of models","author":"Dubey","year":"2024","journal-title":"arXiv e-prints"},{"key":"ref3","article-title":"Deepseek-v3 technical report","author":"Liu","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref4","article-title":"Qwen2. 5 technical report","author":"Yang","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref5","article-title":"Lawyer llama technical report","author":"Huang","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.606"},{"key":"ref7","first-page":"arXiv-2302","article-title":"Bbt-fin: Comprehensive construction of chinese financial domain pre-trained language model, corpus and benchmark","author":"Lu","year":"2023","journal-title":"arXiv eprints"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3735633"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3705725"},{"key":"ref10","article-title":"Llemma: An open language model for mathematics","volume-title":"The Twelfth International Conference on Learning Representations","author":"Azerbayev","year":"2024"},{"key":"ref11","article-title":"Saullm-7b: A pioneering large language model for law","author":"Colombo","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref12","article-title":"Sea-lion: Southeast asian languages in one network","author":"Ng","year":"2025","journal-title":"arXiv preprint arXiv"},{"key":"ref13","first-page":"32850","article-title":"Efficient domain continual pretraining by mitigating the stability gap","volume-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics","volume":"1","author":"Guo","year":"2025"},{"key":"ref14","first-page":"23746","article-title":"Capability salience vector: Finegrained alignment of loss and capabilities for downstream task scaling law","volume-title":"Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics","volume":"1","author":"Ge","year":"2025"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1207\/s15430421tip4104_2"},{"key":"ref16","article-title":"Synthetic continued pretraining","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Yang","year":"2025"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.757"},{"key":"ref18","article-title":"Continuous training and fine-tuning for domainspecific language models in medical question answering","author":"Guo","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.436"},{"key":"ref20","first-page":"22188","article-title":"Same pre-training loss, better downstream: Implicit bias matters for language models","volume-title":"International Conference on Machine Learning","author":"Liu","year":"2023"},{"key":"ref21","article-title":"Dataefficient pretraining with group-level data influence modeling","author":"Yu","year":"2025","journal-title":"arXiv preprint arXiv"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0455"},{"key":"ref23","first-page":"30 811","article-title":"The fineweb datasets: Decanting the web for the finest text data at scale","volume":"37","author":"Penedo","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref24","article-title":"Comprehensive exploration of synthetic data generation: A survey","author":"Bauer","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref25","article-title":"Best practices and lessons learned on synthetic data for language models","author":"Liu","year":"2024","journal-title":"CoRR"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.5250633"},{"key":"ref27","article-title":"Phi4 technical report","author":"Abdin","year":"2024","journal-title":"arXiv preprint arXiv"},{"issue":"3","key":"ref28","first-page":"3","article-title":"Phi-2: The surprising power of small language models","volume":"1","author":"Javaheripi","year":"2023","journal-title":"Microsoft Research Blog"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.2139\/ssrn.5250617"},{"key":"ref30","article-title":"Scaling synthetic data creation with 1,000,000,000 personas","author":"Ge","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref31","article-title":"Efficacy of synthetic data as a benchmark","author":"Maheshwari","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2025.3550410"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.526"},{"key":"ref34","article-title":"Measuring massive multitask language understanding","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR)","author":"Hendrycks","year":"2021"},{"key":"ref35","article-title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark","author":"Wang","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"ref37","first-page":"2717","article-title":"Self-prompted chain-of-thought on large language models for open-domain multi-hop reasoning","volume-title":"Findings of the Association for Computational Linguistics: EMNLP 2023","author":"Wang"},{"key":"ref38","article-title":"Creativity support in the age of large language models: An empirical study involving emerging writers","author":"Chakrabarty","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref39","doi-asserted-by":"crossref","first-page":"5155","DOI":"10.18653\/v1\/2024.emnlp-main.296","article-title":"SEACrowd: A multilingual multimodal data hub and benchmark suite for Southeast Asian languages","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","author":"Lovenia","year":"2024"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.39"},{"key":"ref41","doi-asserted-by":"crossref","first-page":"12308","DOI":"10.18653\/v1\/2025.findings-acl.636","article-title":"SEA-HELM: Southeast Asian holistic evaluation of language models","volume-title":"Findings of the Association for Computational Linguistics: ACL 2025","author":"Susanto","year":"2025"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00661"},{"key":"ref43","doi-asserted-by":"crossref","first-page":"6119","DOI":"10.18653\/v1\/2025.findings-naacl.341","article-title":"SeaExam and SeaBench: Benchmarking LLMs with local multilingual questions in Southeast Asia","volume-title":"Findings of the Association for Computational Linguistics: NAACL 2025","author":"Liu","year":"2025"},{"key":"ref44","first-page":"4076","article-title":"KMMLU: Measuring massive multitask language understanding in Korean","volume-title":"Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Son","year":"2025"},{"key":"ref45","article-title":"Mangosteen: An open thai corpus for language model pretraining","author":"Phatthiyaphaibun","year":"2025","journal-title":"arXiv preprint arXiv"},{"key":"ref46","volume-title":"Korean-webtext: A high-quality korean language corpus","year":"2024"},{"key":"ref47","doi-asserted-by":"crossref","first-page":"4297","DOI":"10.18653\/v1\/2024.findings-emnlp.249","article-title":"Revisiting catastrophic forgetting in large language model tuning","volume-title":"Findings of the Association for Computational Linguistics: EMNLP 2024","author":"Li","year":"2024"},{"key":"ref48","volume-title":"SlimPajama: A 627 B token cleaned and deduplicated version of RedPajama","author":"Soboleva","year":"2023"},{"key":"ref49","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.acl-demos.38","article-title":"Llamafactory: Unified efficient fine-tuning of 100 language models","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 3: System Demonstrations)","author":"Zheng"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657889"}],"event":{"name":"2025 IEEE International Conference on Big Data (BigData)","location":"Macau, China","start":{"date-parts":[[2025,12,8]]},"end":{"date-parts":[[2025,12,11]]}},"container-title":["2025 IEEE International Conference on Big Data (BigData)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11400704\/11400712\/11401489.pdf?arnumber=11401489","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T07:14:28Z","timestamp":1772867668000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11401489\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,8]]},"references-count":50,"URL":"https:\/\/doi.org\/10.1109\/bigdata66926.2025.11401489","relation":{},"subject":[],"published":{"date-parts":[[2025,12,8]]}}}