{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:09:04Z","timestamp":1784138944699,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":78,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Research Grants Council of Hong Kong","award":["PolyU&#x5c;&#x2f;15213323"],"award-info":[{"award-number":["PolyU&#x5c;&#x2f;15213323"]}]},{"name":"Research Grants Council of Hong Kong","award":["PolyU&#x5c;&#x2f;15207122"],"award-info":[{"award-number":["PolyU&#x5c;&#x2f;15207122"]}]},{"name":"Research Grants Council of Hong Kong","award":["PolyU&#x5c;&#x2f;15209724"],"award-info":[{"award-number":["PolyU&#x5c;&#x2f;15209724"]}]},{"name":"Research Grants Council of Hong Kong","award":["PolyU&#x5c;&#x2f;15205325"],"award-info":[{"award-number":["PolyU&#x5c;&#x2f;15205325"]}]},{"name":"PolyU Internal Grants","award":["BDWP"],"award-info":[{"award-number":["BDWP"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809689","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"110-121","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["One Adapts to Any: Meta Reward Modeling for Personalized LLM Alignment"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-9857-6639","authenticated-orcid":false,"given":"Hongru","family":"Cai","sequence":"first","affiliation":[{"name":"The Hong Kong Polytechnic University, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6932-4228","authenticated-orcid":false,"given":"Yongqi","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong Polytechnic University, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5396-950X","authenticated-orcid":false,"given":"Tiezheng","family":"Yu","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6776-2040","authenticated-orcid":false,"given":"Fengbin","family":"Zhu","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5199-1428","authenticated-orcid":false,"given":"Wenjie","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5828-9842","authenticated-orcid":false,"given":"Fuli","family":"Feng","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7360-8864","authenticated-orcid":false,"given":"Wenjie","family":"Li","sequence":"additional","affiliation":[{"name":"The Hong Kong Polytechnic University, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova DasSarma Dawn Drain Stanislav Fort Deep Ganguli Tom Henighan et al. 2022. Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback. arXiv:2204.05862"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00933"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412010"},{"key":"e_1_3_2_1_4_1","volume-title":"Lin Xiao, and Maryam Fazel.","author":"Bose Avinandan","year":"2025","unstructured":"Avinandan Bose, Zhihan Xiong, Yuejie Chi, Simon Shaolei Du, Lin Xiao, and Maryam Fazel. 2025. LoRe: Personalizing LLMs via Low-Rank Reward Modeling. arXiv:2504.14439"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.2307\/2334029"},{"key":"e_1_3_2_1_6_1","unstructured":"Zheng Cai Maosong Cao Haojiong Chen Kai Chen Keyu Chen Xin Chen Xun Chen Zehui Chen Zhi Chen Pei Chu Xiaoyi Dong et al. 2024. InternLM2 Technical Report. arXiv:2403.17297"},{"key":"e_1_3_2_1_7_1","unstructured":"Maosong Cao Alexander Lam Haodong Duan Hongwei Liu Songyang Zhang and Kai Chen. 2024. CompassJudger-1: All-in-one Judge Model Helps Model Evaluation and Evolution. arXiv:2410.16256"},{"key":"e_1_3_2_1_8_1","volume-title":"PAL: Sample-Efficient Personalized Reward Modeling for Pluralistic Alignment. In The Thirteenth International Conference on Learning Representations.","author":"Chen Daiwei","year":"2025","unstructured":"Daiwei Chen, Yi Chen, Aniket Rege, Zhi Wang, and Ramya Korlakai Vinayak. 2025. PAL: Sample-Efficient Personalized Reward Modeling for Pluralistic Alignment. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.53"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295184"},{"key":"e_1_3_2_1_11_1","unstructured":"Yan Duan John Schulman Xi Chen Peter L. Bartlett Ilya Sutskever and Pieter Abbeel. 2016. RL$^2$: Fast Reinforcement Learning via Slow Reinforcement Learning. arXiv:1611.02779"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.5555\/3692070.3692574"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 34th International Conference on Machine Learning -","volume":"70","author":"Finn Chelsea","year":"2017","unstructured":"Chelsea Finn, Pieter Abbeel, and Sergey Levine. 2017. Model-agnostic meta-learning for fast adaptation of deep networks. In Proceedings of the 34th International Conference on Machine Learning - Volume 70 (ICML'17)."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of The 27th International Conference on Artificial Intelligence and Statistics.","author":"Azar Mohammad Gheshlaghi","year":"2024","unstructured":"Mohammad Gheshlaghi Azar, Zhaohan Daniel Guo, Bilal Piot, Remi Munos, Mark Rowland, Michal Valko, and Daniele Calandriello. 2024. A General Theoretical Paradigm to Understand Learning from Human Preferences. In Proceedings of The 27th International Conference on Artificial Intelligence and Statistics."},{"key":"e_1_3_2_1_15_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman et al. 2024. The Llama 3 Herd of Models. arXiv:2407.21783"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.277"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the ACM Web Conference 2025 (WWW'25)","author":"Fengbin Zhu Xiaoyu Shen Wenjie Wang","year":"2025","unstructured":"Wenjie Wang Fengbin Zhu Xiaoyu Shen Wenjie Li Tat-Seng Chua Hongru Cai, Yongqi Li. 2025. Large Language Models Empowered Personalized Web Agents. In Proceedings of the ACM Web Conference 2025 (WWW'25)."},{"key":"e_1_3_2_1_18_1","volume-title":"Yizhong Wang, Jack Hessel, Luke Zettlemoyer, Hannaneh Hajishirzi, Yejin Choi, and Prithviraj Ammanabrolu.","author":"Jang Joel","year":"2023","unstructured":"Joel Jang, Seungone Kim, Bill Yuchen Lin, Yizhong Wang, Jack Hessel, Luke Zettlemoyer, Hannaneh Hajishirzi, Yejin Choi, and Prithviraj Ammanabrolu. 2023. Personalized Soups: Personalized Large Language Model Alignment via Post-hoc Parameter Merging. arXiv:2310.11564"},{"key":"e_1_3_2_1_19_1","volume-title":"A Survey of Reinforcement Learning from Human Feedback. Transactions on Machine Learning Research","author":"Kaufmann Timo","year":"2025","unstructured":"Timo Kaufmann, Paul Weng, Viktor Bengs, and Eyke H\u00fcllermeier. 2025. A Survey of Reinforcement Learning from Human Feedback. Transactions on Machine Learning Research (2025)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614965"},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the Twelfth Language Resources and Evaluation Conference.","author":"King Milton","year":"2020","unstructured":"Milton King and Paul Cook. 2020. Evaluating Approaches to Personalizing Language Models. In Proceedings of the Twelfth Language Resources and Evaluation Conference."},{"key":"e_1_3_2_1_22_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2015","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization."},{"key":"e_1_3_2_1_23_1","volume-title":"Katerina Margatina, Rafael Mosquera, Juan Manuel Ciro, Max Bartolo, Adina Williams, He He, Bertie Vidgen, and Scott A. Hale.","author":"Kirk Hannah Rose","year":"2024","unstructured":"Hannah Rose Kirk, Alexander Whitefield, Paul R\u00f6ttger, Andrew Michael Bean, Katerina Margatina, Rafael Mosquera, Juan Manuel Ciro, Max Bartolo, Adina Williams, He He, Bertie Vidgen, and Scott A. Hale. 2024. The PRISM Alignment Dataset: What Participatory, Representative and Individualised Human Feedback Reveals About the Subjective and Multicultural Alignment of Large Language Models. In The Thirty-eight Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330859"},{"key":"e_1_3_2_1_25_1","volume-title":"Generative Judge for Evaluating Alignment. In The Twelfth International Conference on Learning Representations.","author":"Li Junlong","year":"2024","unstructured":"Junlong Li, Shichao Sun, Weizhe Yuan, Run-Ze Fan, hai zhao, and Pengfei Liu. 2024. Generative Judge for Evaluating Alignment. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.291"},{"key":"e_1_3_2_1_27_1","volume-title":"The Twelfth International Conference on Learning Representations.","author":"Lightman Hunter","year":"2024","unstructured":"Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe. 2024. Let's Verify Step by Step. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_28_1","unstructured":"Chris Yuhao Liu Liang Zeng Jiacai Liu Rui Yan Jujie He Chaojie Wang Shuicheng Yan Yang Liu and Yahui Zhou. 2024. Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs. arXiv:2410.18451"},{"key":"e_1_3_2_1_29_1","unstructured":"Chris Yuhao Liu Liang Zeng Yuzhen Xiao Jujie He Jiacai Liu Chaojie Wang Rui Yan Wei Shen Fuxiang Zhang Jiacheng Xu Yang Liu and Yahui Zhou. 2025. Skywork-Reward-V2: Scaling Preference Data Curation via Human-AI Synergy. arXiv:2507.01352"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1542"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1298"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.201"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.252"},{"key":"e_1_3_2_1_34_1","volume-title":"International Conference on Learning Representations.","author":"Mishra Nikhil","year":"2018","unstructured":"Nikhil Mishra, Mostafa Rohaninejad, Xi Chen, and Pieter Abbeel. 2018. A Simple Neural Attentive Meta-Learner. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems (2022).","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1664"},{"key":"e_1_3_2_1_37_1","volume-title":"First Conference on Language Modeling.","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Joey Hejna, Ryan Park, and Chelsea Finn. 2024. From $r$ to $Qtextasciicircum*$: Your Language Model is Secretly a Q-Function. In First Conference on Language Modeling."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2338"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3114"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.397"},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of the 33rd International Conference on International Conference on Machine Learning -","volume":"48","author":"Santoro Adam","year":"2016","unstructured":"Adam Santoro, Sergey Bartunov, Matthew Botvinick, Daan Wierstra, and Timothy Lillicrap. 2016. Meta-learning with memory-augmented neural networks. In Proceedings of the 33rd International Conference on International Conference on Machine Learning - Volume 48 (ICML'16)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.bea-1.8"},{"key":"e_1_3_2_1_43_1","volume-title":"Second Conference on Language Modeling.","author":"Shenfeld Idan","year":"2025","unstructured":"Idan Shenfeld, Felix Faltings, Pulkit Agrawal, and Aldo Pacchiano. 2025. Language Model Personalization via Reward Factorization. In Second Conference on Language Modeling."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2017.41"},{"key":"e_1_3_2_1_45_1","volume-title":"2nd Workshop on Models of Human Feedback for AI Alignment.","author":"Singh Anikait","year":"2025","unstructured":"Anikait Singh, Sheryl Hsu, Kyle Hsu, Eric Mitchell, Stefano Ermon, Tatsunori Hashimoto, Archit Sharma, and Chelsea Finn. 2025. FSPO: Few-Shot Preference Optimization of Synthetic Preference Data Elicits LLM Personalization to Real Users. In 2nd Workshop on Models of Human Feedback for AI Alignment."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295163"},{"key":"e_1_3_2_1_47_1","volume-title":"Human Language Modeling. In Findings of the Association for Computational Linguistics: ACL","author":"Soni Nikita","year":"2022","unstructured":"Nikita Soni, Matthew Matero, Niranjan Balasubramanian, and H. Andrew Schwartz. 2022. Human Language Modeling. In Findings of the Association for Computational Linguistics: ACL 2022."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning (ICML'24)","author":"Sorensen Taylor","year":"2024","unstructured":"Taylor Sorensen, Jared Moore, Jillian Fisher, Mitchell Gordon, Niloofar Mireshghallah, Christopher Michael Rytting, Andre Ye, Liwei Jiang, Ximing Lu, Nouha Dziri, Tim Althoff, and Yejin Choi. 2024. Position: a roadmap to pluralistic alignment. In Proceedings of the 41st International Conference on Machine Learning (ICML'24)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3495977"},{"key":"e_1_3_2_1_50_1","volume-title":"Two Tales of Persona in LLMs: A Survey of Role-Playing and Personalization. In Findings of the Association for Computational Linguistics: EMNLP","author":"Tseng Yu-Min","year":"2024","unstructured":"Yu-Min Tseng, Yu-Chao Huang, Teng-Yun Hsiao, Wei-Lin Chen, Chao-Wei Huang, Yu Meng, and Yun-Nung Chen. 2024. Two Tales of Persona in LLMs: A Survey of Role-Playing and Personalization. In Findings of the Association for Computational Linguistics: EMNLP 2024."},{"key":"e_1_3_2_1_51_1","volume-title":"Lisa Wang, Antonia Creswell, Geoffrey Irving, and Irina Higgins.","author":"Uesato Jonathan","year":"2023","unstructured":"Jonathan Uesato, Nate Kushman, Ramana Kumar, H. Francis Song, Noah Yamamoto Siegel, Lisa Wang, Antonia Creswell, Geoffrey Irving, and Irina Higgins. 2023. Solving Math Word Problems with Process-based and Outcome-based Feedback."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295434"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.620"},{"key":"e_1_3_2_1_54_1","unstructured":"Jane X Wang Zeb Kurth-Nelson Dhruva Tirumala Hubert Soyer Joel Z Leibo Remi Munos Charles Blundell Dharshan Kumaran and Matt Botvinick. 2017. Learning to reinforcement learn. arXiv:1611.05763"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.510"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3637528.3671599"},{"key":"e_1_3_2_1_57_1","unstructured":"Yufei Wang Wanjun Zhong Liangyou Li Fei Mi Xingshan Zeng Wenyong Huang Lifeng Shang Xin Jiang and Qun Liu. 2023. Aligning Large Language Models with Human: A Survey. arXiv:2307.12966"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.429"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.122"},{"key":"e_1_3_2_1_60_1","volume-title":"TidyBot: personalized robot assistance with large language models. Auton. Robots","author":"Wu Jimmy","year":"2023","unstructured":"Jimmy Wu, Rika Antonova, Adam Kan, Marion Lepert, Andy Zeng, Shuran Song, Jeannette Bohg, Szymon Rusinkiewicz, and Thomas Funkhouser. 2023. TidyBot: personalized robot assistance with large language models. Auton. Robots (2023)."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591719"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"crossref","unstructured":"Jianfei Xiao Xiang Yu Chengbing Wang Wuqiang Zheng Xinyu Lin Kaining Liu Hongxun Ding Yang Zhang Wenjie Wang Fuli Feng and Xiangnan He. 2026. AlpsBench: An LLM Personalization Benchmark for Real-Dialogue Memorization and Preference Alignment. arXiv:2603.26680","DOI":"10.1145\/3805712.3808634"},{"key":"e_1_3_2_1_63_1","volume-title":"Regularizing Hidden States Enables Learning Generalizable Reward Model for LLMs. In The Thirty-eighth Annual Conference on Neural Information Processing Systems.","author":"Yang Rui","year":"2024","unstructured":"Rui Yang, Ruomeng Ding, Yong Lin, Huan Zhang, and Tong Zhang. 2024. Regularizing Hidden States Enables Learning Generalizable Reward Model for LLMs. In The Thirty-eighth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_64_1","volume-title":"Learning LLM-as-a-Judge for Preference Alignment. In The Thirteenth International Conference on Learning Representations.","author":"Ye Ziyi","year":"2025","unstructured":"Ziyi Ye, Xiangsheng Li, Qiuchi Li, Qingyao Ai, Yujia Zhou, Wei Shen, Dong Yan, and Yiqun LIU. 2025. Learning LLM-as-a-Judge for Preference Alignment. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_65_1","unstructured":"Runyang You Yongqi Li Xinyu Lin Xin Zhang Wenjie Wang Wenjie Li and Liqiang Nie. 2025. R$^2$ec: Towards Large Recommender Models with Reasoning. arXiv:2505.16994"},{"key":"e_1_3_2_1_66_1","unstructured":"Qinkai Yu Mingyu Jin Dong Shu Chong Zhang Lizhou Fan Wenyue Hua Suiyuan Zhu Yanda Meng Zhenting Wang Mengnan Du and Yongfeng Zhang. 2025. Health-LLM: Personalized Retrieval-Augmented Disease Prediction System. arXiv:2402.00746"},{"key":"e_1_3_2_1_67_1","volume-title":"Advancing LLM Reasoning Generalists with Preference Trees. In The Thirteenth International Conference on Learning Representations.","author":"Yuan Lifan","year":"2025","unstructured":"Lifan Yuan, Ganqu Cui, Hanbin Wang, Ning Ding, Xingyao Wang, Boji Shan, Zeyuan Liu, Jia Deng, Huimin Chen, Ruobing Xie, Yankai Lin, Zhenghao Liu, Bowen Zhou, Hao Peng, Zhiyuan Liu, and Maosong Sun. 2025. Advancing LLM Reasoning Generalists with Preference Trees. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_68_1","volume-title":"Guided Profile Generation Improves Personalization with Large Language Models. In Findings of the Association for Computational Linguistics: EMNLP","author":"Zhang Jiarui","year":"2024","unstructured":"Jiarui Zhang. 2024. Guided Profile Generation Improves Personalization with Large Language Models. In Findings of the Association for Computational Linguistics: EMNLP 2024."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1205"},{"key":"e_1_3_2_1_70_1","unstructured":"Zhehao Zhang Ryan A. Rossi Branislav Kveton Yijia Shao Diyi Yang Hamed Zamani Franck Dernoncourt Joe Barrow Tong Yu Sungchul Kim et al. 2025. Personalization of Large Language Models: A Survey. Transactions on Machine Learning Research (2025). Survey Certification."},{"key":"e_1_3_2_1_71_1","volume-title":"Group Preference Optimization: Few-Shot Alignment of Large Language Models. In The Twelfth International Conference on Learning Representations.","author":"Zhao Siyan","year":"2024","unstructured":"Siyan Zhao, John Dang, and Aditya Grover. 2024. Group Preference Optimization: Few-Shot Alignment of Large Language Models. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_72_1","volume-title":"Liu","author":"Zhao Yao","year":"2023","unstructured":"Yao Zhao, Rishabh Joshi, Tianqi Liu, Misha Khalman, Mohammad Saleh, and Peter J. Liu. 2023. SLiC-HF: Sequence Likelihood Calibration with Human Feedback. arXiv:2305.10425"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"crossref","unstructured":"Yushang Zhao Huijie Shen Dannier Li Lu Chang Chengrui Zhou and Yinuo Yang. 2025. Meta-Learning for Cold-Start Personalization in Prompt-Tuned LLMs. arXiv:2507.16672","DOI":"10.1109\/CBASE67452.2025.11335518"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2020"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.426"},{"key":"e_1_3_2_1_76_1","unstructured":"Jialun Zhong Wei Shen Yanzeng Li Songyang Gao Hua Lu Yicheng Chen Yang Zhang Wei Zhou Jinjie Gu and Lei Zou. 2025. A Comprehensive Survey of Reward Models: Taxonomy Applications Challenges and Future. arXiv:2504.12328"},{"key":"e_1_3_2_1_77_1","volume-title":"First Conference on Language Modeling.","author":"Zhu Banghua","year":"2024","unstructured":"Banghua Zhu, Evan Frick, Tianhao Wu, Hanlin Zhu, Karthik Ganesan, Wei-Lin Chiang, Jian Zhang, and Jiantao Jiao. 2024. Starling-7B: Improving Helpfulness and Harmlessness with RLAIF. In First Conference on Language Modeling."},{"key":"e_1_3_2_1_78_1","volume-title":"The Thirteenth International Conference on Learning Representations.","author":"Zollo Thomas P","year":"2025","unstructured":"Thomas P Zollo, Andrew Wei Tung Siah, Naimeng Ye, Ang Li, and Hongseok Namkoong. 2025. PersonalLLM: Tailoring LLMs to Individual Preferences. In The Thirteenth International Conference on Learning Representations."}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:30:22Z","timestamp":1784136622000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809689"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":78,"alternative-id":["10.1145\/3805712.3809689","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809689","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}