{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T15:27:25Z","timestamp":1759332445465,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":42,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819794331"},{"type":"electronic","value":"9789819794348"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-9434-8_18","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:03:04Z","timestamp":1730383384000},"page":"229-240","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Bread: A Hybrid Approach for\u00a0Instruction Data Mining Through Balanced Retrieval and\u00a0Dynamic Data Sampling"],"prefix":"10.1007","author":[{"given":"Xinlin","family":"Zhuang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Mao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan-Hao","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongyi","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shangqing","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Cai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxiang","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenghao","family":"Jia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuhao","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Man","family":"Lan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"18_CR1","doi-asserted-by":"crossref","unstructured":"Agrawal, S., Zhou, C., Lewis, M., Zettlemoyer, L., Ghazvininejad, M.: In-context examples selection for machine translation. In: Findings of the Association for Computational Linguistics: ACL 2023, pp. 8857\u20138873 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.564"},{"key":"18_CR2","unstructured":"Baichuan: Baichuan 2: Open large-scale language models. arXiv preprint arXiv:2309.10305 (2023). https:\/\/arxiv.org\/abs\/2309.10305"},{"key":"18_CR3","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Louradour, J., Collobert, R., Weston, J.: Curriculum learning. In: Proceedings of the 26th Annual International Conference on Machine Learning, pp. 41\u201348 (2009)","DOI":"10.1145\/1553374.1553380"},{"key":"18_CR4","unstructured":"Chang, Y., et\u00a0al.: A survey on evaluation of large language models. arXiv preprint arXiv:2307.03109 (2023)"},{"key":"18_CR5","unstructured":"Chen, L., et al.: Alpagasus: training a better alpaca with fewer data. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"18_CR6","doi-asserted-by":"crossref","unstructured":"Chu, X., Ilyas, I.F., Krishnan, S., Wang, J.: Data cleaning: overview and emerging challenges. In: Proceedings of the 2016 International Conference on Management of Data, pp. 2201\u20132206 (2016)","DOI":"10.1145\/2882903.2912574"},{"key":"18_CR7","unstructured":"Conover, M., et al.: Free dolly: Introducing the world\u2019s first truly open instruction-tuned llm (2023). https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Du, Z., et al.: Glm: general language model pretraining with autoregressive blank infilling. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 320\u2013335 (2022)","DOI":"10.18653\/v1\/2022.acl-long.26"},{"key":"18_CR9","unstructured":"Hendrycks, D., Burns, C., Basart, S., Zou, A., Mazeika, M., Song, D., Steinhardt, J.: Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300 (2020)"},{"key":"18_CR10","unstructured":"Hu, E.J., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., Chen, W., et\u00a0al.: Lora: low-rank adaptation of large language models. In: International Conference on Learning Representations (2021)"},{"key":"18_CR11","doi-asserted-by":"crossref","unstructured":"Huang, L., et\u00a0al.: A survey on hallucination in large language models: principles, taxonomy, challenges, and open questions. arXiv preprint arXiv:2311.05232 (2023)","DOI":"10.1145\/3703155"},{"key":"18_CR12","unstructured":"Iyer, R., Khargoankar, N., Bilmes, J., Asanani, H.: Submodular combinatorial information measures with applications in machine learning. In: Algorithmic Learning Theory, pp. 722\u2013754. PMLR (2021)"},{"issue":"8","key":"18_CR13","doi-asserted-by":"publisher","first-page":"651","DOI":"10.1016\/j.patrec.2009.09.011","volume":"31","author":"AK Jain","year":"2010","unstructured":"Jain, A.K.: Data clustering: 50 years beyond k-means. Pattern Recogn. Lett. 31(8), 651\u2013666 (2010)","journal-title":"Pattern Recogn. Lett."},{"key":"18_CR14","unstructured":"Jha, A., Havens, S., Dohmann, J., Trott, A., Portes, J.: LIMIT: less is more for instruction tuning across evaluation paradigms. In: NeurIPS 2023 Workshop on Instruction Tuning and Instruction Following (2023)"},{"issue":"12","key":"18_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3571730","volume":"55","author":"Z Ji","year":"2023","unstructured":"Ji, Z., et al.: Survey of hallucination in natural language generation. ACM Comput. Surv. 55(12), 1\u201338 (2023)","journal-title":"ACM Comput. Surv."},{"key":"18_CR16","unstructured":"K\u00f6ksal, A., Schick, T., Korhonen, A., Sch\u00fctze, H.: Longform: optimizing instruction tuning for long text generation with corpus extraction. arXiv preprint arXiv:2304.08460 (2023)"},{"key":"18_CR17","unstructured":"Li, M., et al.: From quantity to quality: Boosting llm performance with self-guided data selection for instruction tuning. arXiv preprint arXiv:2308.12032 (2023)"},{"key":"18_CR18","unstructured":"Lialin, V., Deshpande, V., Rumshisky, A.: Scaling down to scale up: a guide to parameter-efficient fine-tuning. arXiv preprint arXiv:2303.15647 (2023)"},{"key":"18_CR19","unstructured":"Liu, S., Zhao, S., Jia, C., Zhuang, X., Long, Z., Lan, M.: Bibench: benchmarking data analysis knowledge of large language models. arXiv preprint arXiv:2401.02982 (2024)"},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Mihaylov, T., Clark, P., Khot, T., Sabharwal, A.: Can a suit of armor conduct electricity? a new dataset for open book question answering. In: Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 2381\u20132391 (2018)","DOI":"10.18653\/v1\/D18-1260"},{"key":"18_CR21","unstructured":"Motamedi, M., Sakharnykh, N., Kaldewey, T.: A data-centric approach for training deep neural networks with less data. arXiv preprint arXiv:2110.03613 (2021)"},{"key":"18_CR22","unstructured":"Peng, B., Li, C., He, P., Galley, M., Gao, J.: Instruction tuning with gpt-4. arXiv preprint arXiv:2304.03277 (2023)"},{"key":"18_CR23","unstructured":"Rawte, V., Sheth, A., Das, A.: A survey of hallucination in large foundation models. arXiv preprint arXiv:2309.05922 (2023)"},{"key":"18_CR24","first-page":"80716","volume":"8","author":"KP Sinaga","year":"2020","unstructured":"Sinaga, K.P., Yang, M.S.: Unsupervised k-means clustering algorithm. IEEE Access 8, 80716\u201380727 (2020)","journal-title":"Unsupervised k-means clustering algorithm. IEEE Access"},{"key":"18_CR25","unstructured":"Song, P.: Lawgpt (2023). https:\/\/github.com\/pengxiao-song\/LaWGPT"},{"key":"18_CR26","unstructured":"Sun, Z., et al.: Principle-driven self-alignment of language models from scratch with minimal human supervision. arXiv preprint arXiv:2305.03047 (2023)"},{"key":"18_CR27","unstructured":"Taori, R., et al.: Stanford alpaca: An instruction-following llama model. https:\/\/github.com\/tatsu-lab\/stanford_alpaca (2023)"},{"key":"18_CR28","unstructured":"Touvron, H., et\u00a0al.: Llama: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"18_CR29","unstructured":"Vaswani, A., et al.: Attention is all you need. Advances in neural information processing systems 30 (2017)"},{"key":"18_CR30","unstructured":"Wang, H., et al.: Huatuo: tuning llama model with Chinese medical knowledge (2023)"},{"key":"18_CR31","unstructured":"Wang, Y., et\u00a0al.: Super-natural instructions: generalization via declarative instructions on 1600+ nlp tasks. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 5085\u20135109 (2022)"},{"key":"18_CR32","unstructured":"Xu, C., et al.: Wizardlm: empowering large language models to follow complex instructions. arXiv preprint arXiv:2304.12244 (2023)"},{"key":"18_CR33","doi-asserted-by":"crossref","unstructured":"Yang, Z., Zhang, Y., Sui, D., Liu, C., Zhao, J., Liu, K.: Representative demonstration selection for in-context learning with two-stage determinantal point process. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 5443\u20135456 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.331"},{"key":"18_CR34","doi-asserted-by":"crossref","unstructured":"Yin, D., Liu, X., Yin, F., Zhong, M., Bansal, H., Han, J., Chang, K.W.: Dynosaur: a dynamic growth paradigm for instruction-tuning data curation. arXiv preprint arXiv:2305.14327 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.245"},{"key":"18_CR35","doi-asserted-by":"crossref","unstructured":"Zellers, R., Holtzman, A., Bisk, Y., Farhadi, A., Choi, Y.: Hellaswag: can a machine really finish your sentence? In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 4791\u20134800 (2019)","DOI":"10.18653\/v1\/P19-1472"},{"key":"18_CR36","doi-asserted-by":"crossref","unstructured":"Zha, D., Bhat, Z.P., Lai, K.H., Yang, F., Hu, X.: Data-centric ai: perspectives and challenges. In: Proceedings of the 2023 SIAM International Conference on Data Mining (SDM), pp. 945\u2013948. SIAM (2023)","DOI":"10.1137\/1.9781611977653.ch106"},{"key":"18_CR37","unstructured":"Zha, D., et al.: Data-centric artificial intelligence: a survey. arXiv preprint arXiv:2303.10158 (2023)"},{"key":"18_CR38","unstructured":"Zhang, S., et\u00a0al.: Instruction tuning for large language models: a survey. arXiv preprint arXiv:2308.10792 (2023)"},{"key":"18_CR39","unstructured":"Zhang, Y., et\u00a0al.: Siren\u2019s song in the ai ocean: a survey on hallucination in large language models. arXiv preprint arXiv:2309.01219 (2023)"},{"key":"18_CR40","unstructured":"Zhao, W.X., et\u00a0al.: A survey of large language models. arXiv preprint arXiv:2303.18223 (2023)"},{"key":"18_CR41","unstructured":"Zhou, C., et al.: LIMA: less is more for alignment. In: Thirty-seventh Conference on Neural Information Processing Systems (2023)"},{"key":"18_CR42","unstructured":"Zhou, D., et al.: Dataset quantization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 17205\u201317216 (2023)"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Chinese Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-9434-8_18","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T15:55:59Z","timestamp":1732982159000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-9434-8_18"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9789819794331","9789819794348"],"references-count":42,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-9434-8_18","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLPCC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"CCF International Conference on Natural Language Processing and Chinese Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hangzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 November 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 November 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nlpcc2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/tcci.ccf.org.cn\/conference\/2024\/index.php","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}