{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T07:10:21Z","timestamp":1737443421777,"version":"3.33.0"},"reference-count":44,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,10,6]],"date-time":"2024-10-06T00:00:00Z","timestamp":1728172800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,10,6]],"date-time":"2024-10-06T00:00:00Z","timestamp":1728172800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,10,6]]},"DOI":"10.1109\/smc54092.2024.10831719","type":"proceedings-article","created":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T18:39:20Z","timestamp":1737398360000},"page":"2493-2500","source":"Crossref","is-referenced-by-count":0,"title":["AlpaCream: an Effective Method of Data Selection on Alpaca"],"prefix":"10.1109","author":[{"given":"Yijie","family":"Li","sequence":"first","affiliation":[{"name":"Minzu University of China,Beijing,100081"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan","family":"Sun","sequence":"additional","affiliation":[{"name":"Minzu University of China,Beijing,100081"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"Advances in neural information processing systems"},{"issue":"5","key":"ref2","article-title":"Gpt-4 technical report. arxiv 2303.08774","volume":"2","author":"OpenAI","year":"2023","journal-title":"View in Article"},{"journal-title":"Llama: Open and efficient foundation language models","year":"2023","author":"Touvron","key":"ref3"},{"journal-title":"Llama 2: Open foundation and fine-tuned chat models","year":"2023","author":"Touvron","key":"ref4"},{"journal-title":"Finetuned language models are zero-shot learners","year":"2021","author":"Wei","key":"ref5"},{"key":"ref6","first-page":"631","article-title":"The flan collection: Designing data and methods for effective instruction tuning","volume-title":"International Conference on Machine Learning","volume":"22","author":"Longpre","year":"2023"},{"key":"ref7","first-page":"27 730","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang","year":"2022","journal-title":"Advances in neural information processing systems"},{"journal-title":"In-structzero: Efficient instruction optimization for black-box large language models","year":"2023","author":"Chen","key":"ref8"},{"journal-title":"Stanford alpaca: An instruction-following llama model","year":"2023","author":"Taori","key":"ref9"},{"journal-title":"A general language assistant as a laboratory for alignment","year":"2021","author":"Askell","key":"ref10"},{"journal-title":"Self-alignment with instruction backtranslation","year":"2023","author":"Li","key":"ref11"},{"journal-title":"Instruction mining: High-quality instruction data selection for large language models","year":"2023","author":"Cao","key":"ref12"},{"journal-title":"What makes good data for alignment? a comprehensive study of automatic data selection in instruction tuning","year":"2023","author":"Liu","key":"ref13"},{"journal-title":"Alpagasus: Training a better alpaca with fewer data","year":"2023","author":"Chen","key":"ref14"},{"key":"ref15","article-title":"Lima: Less is more for alignment","volume":"36","author":"Zhou","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref16","article-title":"Judging llm-as-a-judge with mt-bench and chatbot arena","volume":"36","author":"Zheng","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"journal-title":"Cfgpt: Chinese financial assistant with large language model","year":"2023","author":"Li","key":"ref17"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3594536.3595163"},{"key":"ref19","article-title":"Openassistant conversations-democratizing large language model alignment","volume":"36","author":"K\u00f6pf","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"journal-title":"Introducing the world\u2019s first truly open instruction-tuned llm. databricks. com","year":"2023","author":"Dolly","key":"ref20"},{"issue":"3","key":"ref21","first-page":"6","volume-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","volume":"2","author":"Chiang","year":"2023"},{"journal-title":"Wizardlm: Empowering large language models to follow complex instructions","year":"2023","author":"Xu","key":"ref22"},{"journal-title":"Wizardcoder: Empowering code large language models with evol-instruct","year":"2023","author":"Luo","key":"ref23"},{"key":"ref24","first-page":"6","article-title":"Koala: A dialogue model for academic research","volume-title":"Blog post, April","volume":"1","author":"Geng","year":"2023"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.183"},{"journal-title":"Self-instruct: Aligning language models with self-generated instructions","year":"2022","author":"Wang","key":"ref26"},{"key":"ref27","article-title":"Large language model as attributed training data generator: A tale of diversity and bias","volume":"36","author":"Yu","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref28","article-title":"Principle-driven self-alignment of language models from scratch with minimal human supervision","volume":"36","author":"Sun","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"journal-title":"Mods: Model-oriented data selection for instruction tuning","year":"2023","author":"Du","key":"ref29"},{"key":"ref30","article-title":"How far can camels go? exploring the state of instruction tuning on open resources","volume":"36","author":"Wang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"journal-title":"Bertopic: Neural topic modeling with a class-based tf-idf procedure","year":"2022","author":"Grootendorst","key":"ref31"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICEEOT.2016.7754750"},{"journal-title":"Umap: Uniform manifold ap-proximation and projection for dimension reduction","year":"2018","author":"McInnes","key":"ref33"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.21105\/joss.00205"},{"journal-title":"Deberta: Decoding-enhanced bert with disentangled attention","author":"He","key":"ref35"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.84"},{"journal-title":"Large language models are not fair evaluators","year":"2023","author":"Wang","key":"ref37"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.28"},{"journal-title":"Instructeval: Towards holistic evaluation of instruction-tuned large language models","year":"2023","author":"Chia","key":"ref39"},{"journal-title":"Measuring massive multitask language under-standing","year":"2020","author":"Hendrycks","key":"ref40"},{"journal-title":"Drop: A reading comprehension benchmark requiring discrete reasoning over paragraphs","author":"Dua","key":"ref41"},{"journal-title":"Eval-uating large language models trained on code","year":"2021","author":"Chen","key":"ref42"},{"journal-title":"Challenging big-bench tasks and whether chain-of-thought can solve them","year":"2022","author":"Suzgun","key":"ref43"},{"journal-title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","year":"2024","author":"Reid","key":"ref44"}],"event":{"name":"2024 IEEE International Conference on Systems, Man, and Cybernetics (SMC)","start":{"date-parts":[[2024,10,6]]},"location":"Kuching, Malaysia","end":{"date-parts":[[2024,10,10]]}},"container-title":["2024 IEEE International Conference on Systems, Man, and Cybernetics (SMC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10830919\/10830920\/10831719.pdf?arnumber=10831719","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,21]],"date-time":"2025-01-21T06:32:57Z","timestamp":1737441177000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10831719\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,6]]},"references-count":44,"URL":"https:\/\/doi.org\/10.1109\/smc54092.2024.10831719","relation":{},"subject":[],"published":{"date-parts":[[2024,10,6]]}}}