{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T19:37:07Z","timestamp":1780774627667,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":57,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819681792","type":"print"},{"value":"9789819681808","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-8180-8_2","type":"book-chapter","created":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T13:15:24Z","timestamp":1750338924000},"page":"16-28","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Active Instruction Tuning for\u00a0Large Language Models with\u00a0Reference-Free Instruction Selection"],"prefix":"10.1007","author":[{"given":"Qiyuan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiehao","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chen","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"2_CR1","unstructured":"Zhou, C., et al.: LIMA: less is more for alignment. In: NeurIPS, vol. 36, pp. 55006\u201355021 (2023)"},{"key":"2_CR2","unstructured":"Xia, M., Malladi, S., Gururangan, S., Arora, S., Chen, D.: LESS: Selecting Influential Data for Targeted Instruction Tuning. arXiv (2024)"},{"key":"2_CR3","unstructured":"Liu, W., et al.: What makes good data for alignment? A comprehensive study of automatic data selection in instruction tuning. In: ICLR (2024)"},{"key":"2_CR4","unstructured":"Wei, J., et al.: Finetuned language models are zero-shot learners. In: ICLR (2022)"},{"key":"2_CR5","unstructured":"Longpre, S., Hou, L., Vu, T., et al.: The Flan Collection: Designing Data and Methods for Effective Instruction Tuning. arXiv (2023)"},{"key":"2_CR6","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Super-naturalinstructions: generalization via declarative instructions on 1600+ NLP tasks. In: EMNLP, pp. 5085\u20135109 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.340"},{"key":"2_CR7","unstructured":"K\u00f6pf, A., Kilcher, Y., R\u00fctte, et al.: Openassistant conversations - democratizing large language model alignment. In: NeurIPS (2023)"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Ding, N., Chen, Y., et al.: Enhancing chat language models by scaling high-quality instructional conversations. In: EMNLP (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.183"},{"key":"2_CR9","unstructured":"Li, M., Zhang, Y., et al.: From Quantity to Quality: Boosting LLM Performance with Self-Guided Data Selection for Instruction Tuning. arXiv (2023)"},{"key":"2_CR10","unstructured":"Du, Q., Zong, C., Zhang, J.: MoDS: Model-oriented Data Selection for Instruction Tuning. arXiv (2023)"},{"key":"2_CR11","unstructured":"Cao, Y., Kang, Y., Wang, C., Sun, L.: Instruction mining: instruction data selection for tuning large language models. In: ICLR (2023, under review)"},{"key":"2_CR12","unstructured":"Zheng, L., Chiang, W., et al.: LMSYS-chat-1M: a large-scale real-world LLM conversation dataset. In: ICLR (2024)"},{"key":"2_CR13","doi-asserted-by":"crossref","unstructured":"Kung, P., et al.: Active instruction tuning: improving cross-task generalization by training on prompt sensitive tasks. In: EMNLP (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.112"},{"key":"2_CR14","unstructured":"Settles, B.: Active learning literature survey. University of Wisconsin-Madison, Department of Computer Sciences (2009)"},{"key":"2_CR15","unstructured":"Wettig, A., Gupta, A., Malik, S., Chen, D.: QuRating: Selecting High-Quality Data for Training Language Models. arXiv (2024)"},{"key":"2_CR16","doi-asserted-by":"crossref","unstructured":"Chen, X., Bennett, P., Collins-Thompson, K., Horvitz, E.: Pairwise ranking aggregation in a crowdsourced setting. In: WSDM, pp. 193\u2013202 (2013)","DOI":"10.1145\/2433396.2433420"},{"key":"2_CR17","unstructured":"Chung, H., Hou, L., Longpre, S., et al.: Scaling Instruction-Finetuned Language Models. arXiv (2022)"},{"key":"2_CR18","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Self-instruct: aligning language models with self-generated instructions. In: ACL, pp. 13484\u201313508 (2023)","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"2_CR19","unstructured":"Conover, M., Hayes, M., Mathur, A., et al.: Free Dolly: Introducing the World\u2019s First Truly Open Instruction-Tuned LLM (2023)"},{"key":"2_CR20","unstructured":"Wang, Y., Ivison, H., Hajishirzi, H., et al.: How far can camels go? Exploring the state of instruction tuning on open resources. In: NeurIPS (2023)"},{"key":"2_CR21","unstructured":"Jiang, A., et al.: Mistral 7B. arXiv (2023)"},{"key":"2_CR22","unstructured":"Rafailov, R., Sharma, A., et al.: Direct Preference Optimization: Your Language Model is Secretly a Reward Model. arXiv (2023)"},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Jin, T., Xu, P., Gu, Q., Farnoud, F.: Rank Aggregation via Heterogeneous Thurstone Preference Models. arXiv (2019)","DOI":"10.1609\/aaai.v34i04.5860"},{"key":"2_CR24","unstructured":"Parkar, R., Kim, J., Park, J., Kang, D.: SelectLLM: Can LLMs Select Important Instructions to Annotate? arXiv (2024)"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Zhang, C., Chen, Y., et al.: DynaEval: unifying turn and dialogue level evaluation. In: ACL, pp. 5676\u20135689 (2021)","DOI":"10.18653\/v1\/2021.acl-long.441"},{"key":"2_CR26","doi-asserted-by":"crossref","unstructured":"Fu, J., Ng, S., Jiang, Z., Liu, P.: GPTScore: Evaluate as You Desire. arXiv (2023)","DOI":"10.18653\/v1\/2024.naacl-long.365"},{"key":"2_CR27","unstructured":"Celikyilmaz, A., Clark, E., Gao, J.: Evaluation of Text Generation: A Survey. arXiv (2021)"},{"key":"2_CR28","doi-asserted-by":"crossref","unstructured":"Zhong, M., Liu, Y., Yin, D., Mao, Y., et al.: Towards a unified multi-dimensional evaluator for text generation. In: EMNLP, pp. 2023\u20132038 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.131"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.: BLEU: a method for automatic evaluation of machine translation. In: ACL, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"2_CR30","doi-asserted-by":"crossref","unstructured":"Sellam, T., Das, D., Parikh, A.: BLEURT: learning robust metrics for text generation. In: ACL, pp. 7881\u20137892 (2020)","DOI":"10.18653\/v1\/2020.acl-main.704"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Mehri, S., Eskenazi, M.: Unsupervised evaluation of interactive dialog with DialoGPT. In: ACL, pp. 225\u2013235 (2020)","DOI":"10.18653\/v1\/2020.sigdial-1.28"},{"key":"2_CR32","doi-asserted-by":"crossref","unstructured":"Zha, Y., Yang, Y., Li, R., Hu, Z.: AlignScore: evaluating factual consistency with a unified alignment function. In: ACL, pp. 11328\u201311348 (2023)","DOI":"10.18653\/v1\/2023.acl-long.634"},{"key":"2_CR33","doi-asserted-by":"crossref","unstructured":"Xu, G., Liu, R., Harel-Canada, F., Chandra, N., Peng, N.: EnDex: evaluation of dialogue engagingness at scale. In: Findings of EMNLP, pp. 4884\u20134893 (2022)","DOI":"10.18653\/v1\/2022.findings-emnlp.359"},{"key":"2_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, C., et al.: FineD-Eval: fine-grained automatic dialogue-level evaluation. In: EMNLP, pp. 3336\u20133355 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.220"},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, C., et al.: MDD-Eval: self-training on augmented data for multi-domain dialogue evaluation. In: CAI, pp. 11657\u201311666 (2022)","DOI":"10.1609\/aaai.v36i10.21420"},{"key":"2_CR36","unstructured":"Beltagy, I., Peters, M., Cohan, A.: Longformer: The Long-Document Transformer. arXiv (2020)"},{"key":"2_CR37","unstructured":"Zheng, L., Chiang, W., et al.: Judging LLM-as-a-judge with MT-bench and chatbot arena. In: NeurIPS (2023)"},{"key":"2_CR38","unstructured":"Chen, L., Li, S., Jin, H., et al.: AlpaGasus: Training A Better Alpaca with Fewer Data. arXiv (2024)"},{"key":"2_CR39","unstructured":"Dubois, Y., Galambosi, B., Liang, P., Hashimoto, T.: Length-Controlled AlpacaEval: A Simple Way to Debias Automatic Evaluators. arXiv (2024)"},{"key":"2_CR40","unstructured":"AI@Meta Llama 3 Model Card (2024). https:\/\/github.com\/meta-llama\/llama3\/blob\/main\/MODEL_CARD.md"},{"key":"2_CR41","unstructured":"AI@Meta Llama 2 Model Card (2023). https:\/\/github.com\/meta-llama\/llama2\/blob\/main\/MODEL_CARD.md"},{"key":"2_CR42","unstructured":"Mistralai Mistral 7B v0.1 Model Card (2023). https:\/\/huggingface.co\/mistralai\/Mistral-7B-v0.1"},{"key":"2_CR43","unstructured":"Deepmind gemini Model Card (2024). https:\/\/deepmind.google\/technologies\/gemini\/"},{"key":"2_CR44","unstructured":"Achiam, J., et al.: GPT-4 Technical Report. arXiv (2024)"},{"key":"2_CR45","unstructured":"Tunstall, L., et al.: Zephyr: Direct Distillation of LM Alignment. arXiv (2023)"},{"key":"2_CR46","unstructured":"Ivison, H., Wang, Y., Pyatkin, V., et al.: Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2. arXiv (2023)"},{"key":"2_CR47","unstructured":"Mukherjee, S., Mitra, A., et al.: Orca: Progressive Learning from Complex Explanation Traces of GPT-4. arXiv (2023)"},{"key":"2_CR48","doi-asserted-by":"crossref","unstructured":"Fox, E., Shaw, J.: Combination of multiple searches. In: TREC, pp. 243\u2013252 (1993)","DOI":"10.6028\/NIST.SP.500-215.vt"},{"key":"2_CR49","doi-asserted-by":"crossref","unstructured":"Mallows, C.: Non-null ranking models. I. Biometrika 114\u2013130 (1957)","DOI":"10.1093\/biomet\/44.1-2.114"},{"key":"2_CR50","doi-asserted-by":"crossref","unstructured":"Bradley, R., Terry, M.: Rank analysis of incomplete block designs: the method of paired comparisons. Biometrika 324\u2013345 (1952)","DOI":"10.1093\/biomet\/39.3-4.324"},{"key":"2_CR51","doi-asserted-by":"crossref","unstructured":"Thurstone, L.: The method of paired comparisons for social values. In: JASP, p. 384 (1927)","DOI":"10.1037\/h0065439"},{"key":"2_CR52","doi-asserted-by":"crossref","unstructured":"Jin, T., Xu, P., Gu, Q., Farnoud, F.: Rank aggregation via heterogeneous thurstone preference models. In: CAI, pp. 4353\u20134360 (2020)","DOI":"10.1609\/aaai.v34i04.5860"},{"key":"2_CR53","doi-asserted-by":"crossref","unstructured":"Felzenszwalb, P., Girshick, R., McAllester, D., Ramanan, D.: Object detection with discriminatively trained part-based models. TPAMI 1627\u20131645 (2010)","DOI":"10.1109\/TPAMI.2009.167"},{"key":"2_CR54","doi-asserted-by":"crossref","unstructured":"Schr\u00f6der, C., et al.: Revisiting uncertainty-based query strategies for active learning with transformers. In: Findings of ACL, pp. 2194\u20132203 (2022)","DOI":"10.18653\/v1\/2022.findings-acl.172"},{"key":"2_CR55","doi-asserted-by":"crossref","unstructured":"Wang, Z., Bovik, A., Sheikh, H., Simoncelli, E.: Image quality assessment: from error visibility to structural similarity. TIP 600\u2013612 (2004)","DOI":"10.1109\/TIP.2003.819861"},{"key":"2_CR56","unstructured":"Meng, Y., Xia, M., Chen, D.: SimPO: Simple Preference Optimization with a Reference-Free Reward. arXiv (2024)"},{"key":"2_CR57","unstructured":"Muldrew, W., Hayes, P., Zhang, M., Barber, D.: Active Preference Learning for Large Language Models. arXiv (2024)"}],"container-title":["Lecture Notes in Computer Science","Advances in Knowledge Discovery and Data Mining"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-8180-8_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T13:15:40Z","timestamp":1750338940000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-8180-8_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819681792","9789819681808"],"references-count":57,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-8180-8_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"20 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PAKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific-Asia Conference on Knowledge Discovery and Data Mining","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Sydney, NSW","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 June 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 June 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pakdd2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/pakdd2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}