{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T10:40:09Z","timestamp":1755859209493,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","funder":[{"name":"Open Fund of National Key Laboratory of Parallel and Distributed Computing (PDL)","award":["NO.2024-KJWPDL-02"],"award-info":[{"award-number":["NO.2024-KJWPDL-02"]}]},{"name":"Hunan Provincial Innovation-driven Development Plan Project","award":["NO. 2023RC1005"],"award-info":[{"award-number":["NO. 2023RC1005"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,13]]},"DOI":"10.1145\/3726302.3730063","type":"proceedings-article","created":{"date-parts":[[2025,7,14]],"date-time":"2025-07-14T01:21:38Z","timestamp":1752456098000},"page":"191-200","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Preference-Strength-Aware Self-Improving Alignment with Generative Preference Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1385-0074","authenticated-orcid":false,"given":"Yuanzhao","family":"Zhai","sequence":"first","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1835-5411","authenticated-orcid":false,"given":"Zhuo","family":"Zhang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology (Shenzhen), Shenzhen, China and Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4782-1645","authenticated-orcid":false,"given":"Cheng","family":"Yang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, Hunan, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5997-5169","authenticated-orcid":false,"given":"Kele","family":"Xu","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9865-2212","authenticated-orcid":false,"given":"Yue","family":"Yu","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4197-5563","authenticated-orcid":false,"given":"Wei","family":"Li","sequence":"additional","affiliation":[{"name":"Independent Researcher, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0499-727X","authenticated-orcid":false,"given":"Hui","family":"Wang","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5550-6461","authenticated-orcid":false,"given":"Zenglin","family":"Xu","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China and Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7587-8905","authenticated-orcid":false,"given":"Dawei","family":"Feng","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1236-8318","authenticated-orcid":false,"given":"Bo","family":"Ding","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3245-1901","authenticated-orcid":false,"given":"Huaimin","family":"Wang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657848"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657689"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657782"},{"key":"e_1_3_2_1_4_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877--1901."},{"key":"e_1_3_2_1_5_1","unstructured":"Jiaming Ji Tianyi Qiu Boyuan Chen Borong Zhang Hantao Lou Kaile Wang Yawen Duan Zhonghao He Jiayi Zhou Zhaowei Zhang et al. 2023. Ai alignment: A comprehensive survey. arXiv preprint arXiv:2310.19852 (2023)."},{"key":"e_1_3_2_1_6_1","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al. 2022. Training language models to follow instructions with human feedback. Advances in Neural Information Processing Systems, Vol. 35 (2022), 27730--27744.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_7_1","volume-title":"Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems","author":"Rafailov Rafael","year":"2023","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Stefano Ermon, Christopher D Manning, and Chelsea Finn. 2023. Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems (2023)."},{"key":"e_1_3_2_1_8_1","volume-title":"A general theoretical paradigm to understand learning from human preferences. arXiv preprint arXiv:2310.12036","author":"Azar Mohammad Gheshlaghi","year":"2023","unstructured":"Mohammad Gheshlaghi Azar, Mark Rowland, Bilal Piot, Daniel Guo, Daniele Calandriello, Michal Valko, and R\u00e9mi Munos. 2023. A general theoretical paradigm to understand learning from human preferences. arXiv preprint arXiv:2310.12036 (2023)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3673791.3698411"},{"key":"e_1_3_2_1_10_1","volume-title":"Ultrafeedback: Boosting language models with high-quality feedback. arXiv preprint arXiv:2310.01377","author":"Cui Ganqu","year":"2023","unstructured":"Ganqu Cui, Lifan Yuan, Ning Ding, Guanming Yao, Wei Zhu, Yuan Ni, Guotong Xie, Zhiyuan Liu, and Maosong Sun. 2023. Ultrafeedback: Boosting language models with high-quality feedback. arXiv preprint arXiv:2310.01377 (2023)."},{"key":"e_1_3_2_1_11_1","volume-title":"Kyunghyun Cho, Sainbayar Sukhbaatar, Jing Xu, and Jason Weston.","author":"Yuan Weizhe","year":"2024","unstructured":"Weizhe Yuan, Richard Yuanzhe Pang, Kyunghyun Cho, Sainbayar Sukhbaatar, Jing Xu, and Jason Weston. 2024. Self-Rewarding Language Models. arXiv preprint arXiv:2401.10020 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"Meta-rewarding language models: Self-improving alignment with llm-as-a-meta-judge. arXiv preprint arXiv:2407.19594","author":"Wu Tianhao","year":"2024","unstructured":"Tianhao Wu, Weizhe Yuan, Olga Golovneva, Jing Xu, Yuandong Tian, Jiantao Jiao, Jason Weston, and Sainbayar Sukhbaatar. 2024. Meta-rewarding language models: Self-improving alignment with llm-as-a-meta-judge. arXiv preprint arXiv:2407.19594 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric Xing, et al. 2023. Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems, Vol. 36 (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods. arXiv preprint arXiv:2412.05579","author":"Li Haitao","year":"2024","unstructured":"Haitao Li, Qian Dong, Junjie Chen, Huixue Su, Yujia Zhou, Qingyao Ai, Ziyi Ye, and Yiqun Liu. 2024. LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods. arXiv preprint arXiv:2412.05579 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Kang Min Yoo, and Youngjae Yu","author":"Lee Sangkyu","year":"2024","unstructured":"Sangkyu Lee, Sungdong Kim, Ashkan Yousefpour, Minjoon Seo, Kang Min Yoo, and Youngjae Yu. 2024. Aligning Large Language Models by On-Policy Self-Judgment. arXiv preprint arXiv:2402.11253 (2024)."},{"key":"e_1_3_2_1_16_1","volume-title":"Iterative Data Smoothing: Mitigating Reward Overfitting and Overoptimization in RLHF. arXiv preprint arXiv:2401.16335","author":"Zhu Banghua","year":"2024","unstructured":"Banghua Zhu, Jordan Michael, I., and Jiantao Jiao. 2024. Iterative Data Smoothing: Mitigating Reward Overfitting and Overoptimization in RLHF. arXiv preprint arXiv:2401.16335 (2024)."},{"key":"e_1_3_2_1_17_1","volume-title":"Scaling laws for reward model overoptimization in direct alignment algorithms. arXiv preprint arXiv:2406.02900","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Yaswanth Chittepu, Ryan Park, Harshit Sikchi, Joey Hejna, Bradley Knox, Chelsea Finn, and Scott Niekum. 2024. Scaling laws for reward model overoptimization in direct alignment algorithms. arXiv preprint arXiv:2406.02900 (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"Length-controlled alpacaeval: A simple way to debias automatic evaluators. arXiv preprint arXiv:2404.04475","author":"Dubois Yann","year":"2024","unstructured":"Yann Dubois, Bal\u00e1zs Galambosi, Percy Liang, and Tatsunori B Hashimoto. 2024. Length-controlled alpacaeval: A simple way to debias automatic evaluators. arXiv preprint arXiv:2404.04475 (2024)."},{"key":"e_1_3_2_1_19_1","volume-title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline. arXiv preprint arXiv:2406.11939","author":"Li Tianle","year":"2024","unstructured":"Tianle Li, Wei-Lin Chiang, Evan Frick, Lisa Dunlap, Tianhao Wu, Banghua Zhu, Joseph E Gonzalez, and Ion Stoica. 2024. From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline. arXiv preprint arXiv:2406.11939 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.67"},{"volume-title":"SAIL: Self-improving Efficient Online Alignment of Large Language Models. In ICML 2024 Workshop on Theoretical Foundations of Foundation Models.","author":"Ding Mucong","key":"e_1_3_2_1_22_1","unstructured":"Mucong Ding, Souradip Chakraborty, Vibhu Agrawal, Zora Che, Alec Koppel, Mengdi Wang, Amrit Bedi, and Furong Huang. [n.,d.]. SAIL: Self-improving Efficient Online Alignment of Large Language Models. In ICML 2024 Workshop on Theoretical Foundations of Foundation Models."},{"key":"e_1_3_2_1_23_1","first-page":"18907","article-title":"Efficient first-order contextual bandits: Prediction, allocation, and triangular discrimination","volume":"34","author":"Foster Dylan J","year":"2021","unstructured":"Dylan J Foster and Akshay Krishnamurthy. 2021. Efficient first-order contextual bandits: Prediction, allocation, and triangular discrimination. Advances in Neural Information Processing Systems, Vol. 34 (2021), 18907--18919.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_24_1","volume-title":"Simpo: Simple preference optimization with a reference-free reward. arXiv preprint arXiv:2405.14734","author":"Meng Yu","year":"2024","unstructured":"Yu Meng, Mengzhou Xia, and Danqi Chen. 2024. Simpo: Simple preference optimization with a reference-free reward. arXiv preprint arXiv:2405.14734 (2024)."},{"key":"e_1_3_2_1_25_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"Rafael Rafailov, Chase Blagden, Nathan Lile, Louis Castricato, Jan-Philipp Fr\u00e4nken, Chelsea Finn, and Alon Albalak.","author":"Mahan Dakota","year":"2024","unstructured":"Dakota Mahan, Duy Van Phung, Rafael Rafailov, Chase Blagden, Nathan Lile, Louis Castricato, Jan-Philipp Fr\u00e4nken, Chelsea Finn, and Alon Albalak. 2024. Generative reward models. arXiv preprint arXiv:2410.12832 (2024)."},{"key":"e_1_3_2_1_27_1","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova DasSarma Dawn Drain Stanislav Fort Deep Ganguli Tom Henighan et al. 2022. Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862 (2022)."},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"6008","author":"Ethayarajh Kawin","year":"2022","unstructured":"Kawin Ethayarajh, Yejin Choi, and Swabha Swayamdipta. 2022. Understanding Dataset Difficulty with V-Usable Information. In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 5988--6008. https:\/\/proceedings.mlr.press\/v162\/ethayarajh22a.html"},{"key":"e_1_3_2_1_29_1","volume-title":"Data contamination: From memorization to exploitation. arXiv preprint arXiv:2203.08242","author":"Magar Inbal","year":"2022","unstructured":"Inbal Magar and Roy Schwartz. 2022. Data contamination: From memorization to exploitation. arXiv preprint arXiv:2203.08242 (2022)."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"6008","author":"Ethayarajh Kawin","year":"2022","unstructured":"Kawin Ethayarajh, Yejin Choi, and Swabha Swayamdipta. 2022. Understanding Dataset Difficulty with V-Usable Information. In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 5988--6008. https:\/\/proceedings.mlr.press\/v162\/ethayarajh22a.html"},{"key":"e_1_3_2_1_31_1","volume-title":"International Conference on Machine Learning. PMLR, 10835--10866","author":"Gao Leo","year":"2023","unstructured":"Leo Gao, John Schulman, and Jacob Hilton. 2023. Scaling laws for reward model overoptimization. In International Conference on Machine Learning. PMLR, 10835--10866."}],"event":{"name":"SIGIR '25: The 48th International ACM SIGIR Conference on Research and Development in Information Retrieval","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"],"location":"Padua Italy","acronym":"SIGIR '25"},"container-title":["Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3726302.3730063","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T09:59:24Z","timestamp":1755856764000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3726302.3730063"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,13]]},"references-count":31,"alternative-id":["10.1145\/3726302.3730063","10.1145\/3726302"],"URL":"https:\/\/doi.org\/10.1145\/3726302.3730063","relation":{},"subject":[],"published":{"date-parts":[[2025,7,13]]},"assertion":[{"value":"2025-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}