{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:05:11Z","timestamp":1784138711256,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809946","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"4361-4366","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["M-DaQ: Retrieving Samples with Multilingual Diversity and Quality for Instruction Fine-Tuning Datasets"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5501-2105","authenticated-orcid":false,"given":"Chunguang","family":"Zhao","sequence":"first","affiliation":[{"name":"Huawei Technologies Ltd., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2531-7728","authenticated-orcid":false,"given":"Yilun","family":"Liu","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6256-6315","authenticated-orcid":false,"given":"Pufan","family":"Zeng","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4411-462X","authenticated-orcid":false,"given":"Yuanchang","family":"Luo","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2795-6921","authenticated-orcid":false,"given":"Shimin","family":"Tao","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4719-6241","authenticated-orcid":false,"given":"Minggui","family":"He","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9384-9016","authenticated-orcid":false,"given":"Weibin","family":"Meng","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1917-1931","authenticated-orcid":false,"given":"Song","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Suzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2697-5195","authenticated-orcid":false,"given":"Chen","family":"Liu","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1046-4477","authenticated-orcid":false,"given":"Hongxia","family":"Ma","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6392-4589","authenticated-orcid":false,"given":"Li","family":"Zhang","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3170-4858","authenticated-orcid":false,"given":"Boxing","family":"Chen","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Montreal, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9606-5058","authenticated-orcid":false,"given":"Daimeng","family":"Wei","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.401"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/290941.291025"},{"key":"e_1_3_2_1_3_1","unstructured":"Lichang Chen Shiyang Li Jun Yan Hai Wang Kalpa Gunaratna Vikas Yadav Zheng Tang Vijay Srinivasan Tianyi Zhou Heng Huang and Hongxia Jin. 2024. AlpaGasus: Training A Better Alpaca with Fewer Data. arXiv:2307.08701 [cs.CL] https:\/\/arxiv.org\/abs\/2307.08701"},{"key":"e_1_3_2_1_4_1","unstructured":"Yijie Chen Yijin Liu Fandong Meng Yufeng Chen Jinan Xu and Jie Zhou. 2023. Improving Translation Faithfulness of Large Language Models via Augmenting Instructions. arXiv:2308.12674 [cs.CL] https:\/\/arxiv.org\/abs\/2308.12674"},{"key":"e_1_3_2_1_5_1","volume-title":"Xing","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang, Zhuohan Li, Zi Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph E. Gonzalez, Ion Stoica, and Eric P. Xing. 2023. Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"e_1_3_2_1_6_1","volume-title":"Free Dolly: Introducing the World's First Truly Open Instruction-Tuned LLM. https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm","author":"Conover Mike","year":"2023","unstructured":"Mike Conover, Matt Hayes, Ankit Mathur, Jianwei Xie, Jun Wan, Sam Shah, Ali Ghodsi, Patrick Wendell, Matei Zaharia, and Reynold Xin. 2023. Free Dolly: Introducing the World's First Truly Open Instruction-Tuned LLM. https:\/\/www.databricks.com\/blog\/2023\/04\/12\/dolly-first-open-commercially-viable-instruction-tuned-llm"},{"key":"e_1_3_2_1_7_1","unstructured":"Qianlong Du Chengqing Zong and Jiajun Zhang. 2023. MoDS: Model-oriented Data Selection for Instruction Tuning. arXiv:2311.15653 [cs.CL] https:\/\/arxiv.org\/abs\/2311.15653"},{"key":"e_1_3_2_1_8_1","unstructured":"Xiyan Fu and Wei Liu. 2025. How Reliable is Multilingual LLM-as-a-Judge? arXiv:2505.12201 [cs.CL] https:\/\/arxiv.org\/abs\/2505.12201"},{"key":"e_1_3_2_1_9_1","unstructured":"Yuan Ge Yilun Liu Chi Hu Weibin Meng Shimin Tao Xiaofeng Zhao Hongxia Ma Li Zhang Boxing Chen Hao Yang Bei Li Tong Xiao and Jingbo Zhu. 2024a. Clustering and Ranking: Diversity-preserved Instruction Selection through Expert-aligned Quality Estimation. arXiv:2402.18191 [cs.CL] https:\/\/arxiv.org\/abs\/2402.18191"},{"key":"e_1_3_2_1_10_1","unstructured":"Yuan Ge Yilun Liu Chi Hu Weibin Meng Shimin Tao Xiaofeng Zhao Hongxia Ma Li Zhang Boxing Chen Hao Yang Bei Li Tong Xiao and Jingbo Zhu. 2024b. Clustering and Ranking: Diversity-preserved Instruction Selection through Expert-aligned Quality Estimation. arXiv:2402.18191 [cs.CL] https:\/\/arxiv.org\/abs\/2402.18191"},{"key":"e_1_3_2_1_11_1","unstructured":"Hila Gonen Srini Iyer Terra Blevins Noah A. Smith and Luke Zettlemoyer. 2024. Demystifying Prompts in Language Models via Perplexity Estimation. arXiv:2212.04037 [cs.CL] https:\/\/arxiv.org\/abs\/2212.04037"},{"key":"e_1_3_2_1_12_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey and et al. 2024. The Llama 3 Herd of Models. arXiv:2407.21783 [cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_2_1_13_1","volume-title":"TACOS: Open Tagging and Comparative Scoring for Instruction Fine-Tuning Data Selection. arXiv:2507.03673 [cs.CL] https:\/\/arxiv.org\/abs\/2507.03673","author":"He Xixiang","year":"2025","unstructured":"Xixiang He, Hao Yu, Qiyao Sun, Ao Cheng, Tailai Zhang, Cong Liu, and Shuxuan Guo. 2025. TACOS: Open Tagging and Comparative Scoring for Instruction Fine-Tuning Data Selection. arXiv:2507.03673 [cs.CL] https:\/\/arxiv.org\/abs\/2507.03673"},{"key":"e_1_3_2_1_14_1","volume-title":"DIVERGE: Diversity-Enhanced RAG for Open-Ended Information Seeking. arXiv:2602.00238 [cs.CL] https:\/\/arxiv.org\/abs\/2602.00238","author":"Hu Tianyi","year":"2026","unstructured":"Tianyi Hu, Niket Tandon, and Akhil Arora. 2026. DIVERGE: Diversity-Enhanced RAG for Open-Ended Information Seeking. arXiv:2602.00238 [cs.CL] https:\/\/arxiv.org\/abs\/2602.00238"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Po-Nien Kung Fan Yin Di Wu Kai-Wei Chang and Nanyun Peng. 2023. Active Instruction Tuning: Improving Cross-Task Generalization by Training on Prompt Sensitive Tasks. arXiv:2311.00288 [cs.CL] https:\/\/arxiv.org\/abs\/2311.00288","DOI":"10.18653\/v1\/2023.emnlp-main.112"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Wen Lai Mohsen Mesgar and Alexander Fraser. 2024. LLMs Beyond English: Scaling the Multilingual Capability of LLMs with Cross-Lingual Feedback. arXiv:2406.01771 [cs.CL] https:\/\/arxiv.org\/abs\/2406.01771","DOI":"10.18653\/v1\/2024.findings-acl.488"},{"key":"e_1_3_2_1_17_1","unstructured":"Ming Li Yong Zhang Zhitao Li Jiuhai Chen Lichang Chen Ning Cheng Jianzong Wang Tianyi Zhou and Jing Xiao. 2024. From Quantity to Quality: Boosting LLM Performance with Self-Guided Data Selection for Instruction Tuning. arXiv:2308.12032 [cs.CL] https:\/\/arxiv.org\/abs\/2308.12032"},{"key":"e_1_3_2_1_18_1","volume-title":"Hashimoto","author":"Li Xuechen","year":"2023","unstructured":"Xuechen Li, Tianyi Zhang, Yann Dubois, Rohan Taori, Ishaan Gulrajani, Carlos Guestrin, Percy Liang, and Tatsunori B. Hashimoto. 2023. AlpacaEval: An Automatic Evaluator of Instruction-following Models. https:\/\/github.com\/tatsu-lab\/alpaca_eval."},{"key":"e_1_3_2_1_19_1","unstructured":"Wei Liu Weihao Zeng Keqing He Yong Jiang and Junxian He. 2024. What Makes Good Data for Alignment? A Comprehensive Study of Automatic Data Selection in Instruction Tuning. arXiv:2312.15685 [cs.CL] https:\/\/arxiv.org\/abs\/2312.15685"},{"key":"e_1_3_2_1_20_1","volume-title":"MIDB: Multilingual Instruction Data Booster for Enhancing Multilingual Instruction Synthesis. arXiv:2505.17671 [cs.CL] https:\/\/arxiv.org\/abs\/2505.17671","author":"Liu Yilun","year":"2025","unstructured":"Yilun Liu, Chunguang Zhao, Xinhua Yang, Hongyong Zeng, Shimin Tao, Weibin Meng, Minggui He, Chang Su, Yan Yu, Hongxia Ma, Li Zhang, Daimeng Wei, and Hao Yang. 2025. MIDB: Multilingual Instruction Data Booster for Enhancing Multilingual Instruction Synthesis. arXiv:2505.17671 [cs.CL] https:\/\/arxiv.org\/abs\/2505.17671"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","unstructured":"Akash Kumar Mohankumar Nikit Begwani and Amit Singh. 2021. Diversity driven Query Rewriting in Search Advertising. arXiv:2106.03816 [cs.CL] https:\/\/arxiv.org\/abs\/2106.03816","DOI":"10.1145\/3447548.3467202"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1108\/eb026647"},{"key":"e_1_3_2_1_23_1","volume-title":"Hashimoto","author":"Taori Rohan","year":"2023","unstructured":"Rohan Taori, Ishaan Gulrajani, Tianyi Zhang, Yann Dubois, Xuechen Li, Carlos Guestrin, Percy Liang, and Tatsunori B. Hashimoto. 2023. Stanford Alpaca: An Instruction-following LLaMA model. https:\/\/github.com\/tatsu-lab\/stanford_alpaca."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Yizhong Wang Swaroop Mishra Pegah Alipoormolabashi and et al. 2022. Super-NaturalInstructions: Generalization via Declarative Instructions on 1600 NLP Tasks. arXiv:2204.07705 [cs.CL] https:\/\/arxiv.org\/abs\/2204.07705","DOI":"10.18653\/v1\/2022.emnlp-main.340"},{"key":"e_1_3_2_1_25_1","unstructured":"Hao Zhao Maksym Andriushchenko Francesco Croce and Nicolas Flammarion. 2024a. Long Is More for Alignment: A Simple but Tough-to-Beat Baseline for Instruction Fine-Tuning. arXiv:2402.04833 [cs.CL] https:\/\/arxiv.org\/abs\/2402.04833"},{"key":"e_1_3_2_1_26_1","volume-title":"Zhang","author":"Zhao Yingxiu","year":"2024","unstructured":"Yingxiu Zhao, Bowen Yu, Binyuan Hui, Haiyang Yu, Fei Huang, Yongbin Li, and Nevin L. Zhang. 2024b. A Preliminary Study of the Intrinsic Relationship between Complexity and Alignment. arXiv:2308.05696 [cs.CL] https:\/\/arxiv.org\/abs\/2308.05696"},{"key":"e_1_3_2_1_27_1","volume-title":"LIMA: Less Is More for Alignment. arXiv:2305.11206 [cs.CL] https:\/\/arxiv.org\/abs\/2305.11206","author":"Zhou Chunting","year":"2023","unstructured":"Chunting Zhou, Pengfei Liu, Puxin Xu, Srini Iyer, Jiao Sun, Yuning Mao, Xuezhe Ma, Avia Efrat, Ping Yu, Lili Yu, Susan Zhang, Gargi Ghosh, Mike Lewis, Luke Zettlemoyer, and Omer Levy. 2023. LIMA: Less Is More for Alignment. arXiv:2305.11206 [cs.CL] https:\/\/arxiv.org\/abs\/2305.11206"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:15:52Z","timestamp":1784135752000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809946"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":27,"alternative-id":["10.1145\/3805712.3809946","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809946","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}