{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T20:40:56Z","timestamp":1782938456396,"version":"3.54.5"},"reference-count":111,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"7","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Artif. Intell."],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1109\/tai.2026.3654921","type":"journal-article","created":{"date-parts":[[2026,1,23]],"date-time":"2026-01-23T21:02:09Z","timestamp":1769202129000},"page":"3688-3702","source":"Crossref","is-referenced-by-count":0,"title":["A Survey of Progress in LLM Alignment From the Perspective of Reward Design"],"prefix":"10.1109","volume":"7","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1767-9166","authenticated-orcid":false,"given":"Miaomiao","family":"Ji","sequence":"first","affiliation":[{"name":"Business School, Sichuan University, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5689-1797","authenticated-orcid":false,"given":"Yanqiu","family":"Wu","sequence":"additional","affiliation":[{"name":"School of Computing, Macquarie University, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9372-0992","authenticated-orcid":false,"given":"Zhibin","family":"Wu","sequence":"additional","affiliation":[{"name":"Business School, Sichuan University, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1133-9379","authenticated-orcid":false,"given":"Shoujin","family":"Wang","sequence":"additional","affiliation":[{"name":"Data Science Institute, University of Technology Sydney, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4408-1952","authenticated-orcid":false,"given":"Jian","family":"Yang","sequence":"additional","affiliation":[{"name":"School of Computing, Macquarie University, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9908-7182","authenticated-orcid":false,"given":"Mark","family":"Dras","sequence":"additional","affiliation":[{"name":"School of Computing, Macquarie University, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0191-7171","authenticated-orcid":false,"given":"Usman","family":"Naseem","sequence":"additional","affiliation":[{"name":"School of Computing, Macquarie University, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"GPT-4 technical report","year":"2023"},{"key":"ref2","article-title":"Introducing Claude"},{"key":"ref3","article-title":"Gemini: A family of highly capable multimodal models","author":"Team","year":"2024"},{"key":"ref4","article-title":"We think, therefore we align LLMs to helpful, harmless and honest before they go wrong","author":"Kashyap","year":"2025"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.3115\/1073445.1073465"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"ref7","article-title":"Deep reinforcement learning from human preferences","volume":"30","author":"Christiano","year":"2017","journal-title":"Advances in Neural Information Processing System"},{"key":"ref8","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref9","article-title":"Concrete problems in AI safety","author":"Amodei","year":"2016"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0687"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.451"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.217"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.1007"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0095"},{"key":"ref15","article-title":"Improving video generation with human feedback","author":"Liu","year":"2025"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i8.28777"},{"key":"ref17","article-title":"Steering over-refusals towards safety in retrieval augmented generation","author":"Maskey","year":"2025"},{"key":"ref18","article-title":"On targeted manipulation and deception when optimizing LLMs for user feedback","author":"Williams","year":"2024"},{"key":"ref19","article-title":"SRLM: Human-in-loop interactive social robot navigation with large language model and deep reinforcement learning","author":"Wang","year":"2024"},{"key":"ref20","article-title":"Constitutional AI: Harmlessness from AI feedback","author":"Bai","year":"2022"},{"key":"ref21","article-title":"Safe RLHF: Safe reinforcement learning from human feedback","author":"Dai","year":"2023"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.641"},{"key":"ref23","article-title":"Let\u2019s verify step by step","volume-title":"Proc. 12th Int. Conf. Learn. Representat.","author":"Lightman","year":"2023"},{"key":"ref24","article-title":"Token-level direct preference optimization","author":"Zeng","year":"2024"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.775"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.468"},{"key":"ref27","article-title":"AutoRule: Reasoning chain-of-thought extracted rule-based rewards improve preference learning","author":"Wang","year":"2025"},{"key":"ref28","first-page":"200","article-title":"Multimodal few-shot learning with frozen language models","volume":"34","author":"Tsimpoukelli","year":"2021","journal-title":"Advances in Neural Information Processing System"},{"key":"ref29","first-page":"1","article-title":"ReAct: Synergizing reasoning and acting in language models","volume-title":"Proc. 12th Int. Conf. Learn. Represent.","author":"Yao","year":"2022"},{"key":"ref30","article-title":"Large language model alignment: A survey","author":"Shen","year":"2023"},{"key":"ref31","article-title":"Aligning large language models with human: A survey","author":"Wang","year":"2023"},{"key":"ref32","article-title":"A comprehensive survey of LLM alignment techniques: RLHF, RLAIF, PPO, DPO and more","author":"Wang","year":"2024"},{"key":"ref33","article-title":"A survey on human preference learning for large language models","author":"Jiang","year":"2024"},{"key":"ref34","article-title":"A comprehensive survey of reward models: Taxonomy, applications, challenges, and future","author":"Zhong","year":"2025"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.525"},{"key":"ref36","article-title":"TLDR: Token-level detective reward model for large vision language models","author":"Fu","year":"2024"},{"key":"ref37","article-title":"Discriminative policy optimization for token-level reward models","author":"Chen","year":"2025"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.2307\/2334029"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.2307\/2346567"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2011"},{"key":"ref41","article-title":"Training a helpful and harmless assistant with reinforcement learning from human feedback","author":"Bai","year":"2022"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2019"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.889"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-short.62"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2338"},{"key":"ref46","article-title":"Reward model ensembles help mitigate overoptimization","author":"Coste","year":"2023"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.754"},{"key":"ref48","article-title":"UltraFeedback: Boosting language models with scaled AI feedback","author":"Cui","year":"2023"},{"key":"ref49","article-title":"Transforming and combining rewards for aligning large language models","author":"Wang","year":"2024"},{"key":"ref50","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3114"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.630"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.620"},{"key":"ref54","first-page":"57905","article-title":"Self-rewarding language models","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Yuan","year":"2024"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.427"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0377"},{"key":"ref57","article-title":"Shepherd: A critic for language model generation","author":"Wang","year":"2023"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.sigdial-1.27"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3457"},{"key":"ref60","article-title":"RuleAlign: Making large language models better physicians with diagnostic rule alignment","author":"Wang","year":"2024"},{"key":"ref61","article-title":"BLEUBERI: BLEU is a surprisingly effective reward for instruction following","author":"Chang","year":"2025"},{"key":"ref62","article-title":"Inverse-RLignment: Large language model alignment from demonstrations through inverse reinforcement learning","author":"Sun","year":"2024"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i22.34519"},{"key":"ref64","article-title":"Inverse reinforcement learning with dynamic reward scaling for LLM alignment","author":"Cheng","year":"2025"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.775"},{"key":"ref66","article-title":"MOSLIM: Align with diverse preferences in prompts through reward classification","author":"Zhang","year":"2025"},{"key":"ref67","first-page":"18874","article-title":"HAFRM: A hybrid alignment framework for reward model training","volume-title":"Proc. 63rd Annu. Meeting Assoc. Comput. Linguist. (ACL)","author":"Liu","year":"2025"},{"key":"ref68","article-title":"Reasoning through Execution: Unifying process and outcome rewards for code generation","author":"Yu","year":"2024"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.462"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.465"},{"key":"ref71","article-title":"Sentence-level reward model can generalize better for aligning LLM from human preference","author":"Qiu","year":"2024"},{"key":"ref72","article-title":"RLHF workflow: From reward modeling to online RLHF","author":"Dong","year":"2024"},{"key":"ref73","article-title":"On the weaknesses of reinforcement learning for neural machine translation","author":"Choshen","year":"2019"},{"key":"ref74","first-page":"1","article-title":"Implementation matters in deep RL: A case study on PPO and TRPO","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Engstrom","year":"2019"},{"key":"ref75","article-title":"DPO Meets PPO: Reinforced token optimization for RLHF","author":"Zhong","year":"2024"},{"key":"ref76","first-page":"1","article-title":"CPPO: Continual learning for reinforcement learning with human feedback","volume-title":"Proc. 12th Int. Conf. Learn. Representat.","author":"Zhang","year":"2024"},{"key":"ref77","article-title":"Pairwise proximal policy optimization: Harnessing relative feedback for LLM alignment","author":"Wu","year":"2023"},{"key":"ref78","article-title":"Efficient RLHF: Reducing the memory usage of PPO","author":"Santacroce","year":"2023"},{"key":"ref79","article-title":"Improving reinforcement learning from human feedback using contrastive rewards","author":"Shen","year":"2024"},{"key":"ref80","article-title":"Secrets of RLHF in large language models Part I: PPO","author":"Rafailov","year":"2023"},{"key":"ref81","article-title":"ReMax: A simple, effective, and efficient reinforcement learning method for aligning large language models","author":"Li","year":"2023"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.662"},{"key":"ref83","article-title":"REINFORCE++: An efficient RLHF algorithm with robustness to both prompt and reward models","author":"Hu","year":"2025"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-70362-1_7"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1990"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.64"},{"key":"ref87","first-page":"7546","article-title":"DORA: Dynamic optimization prompt for continuous reflection of LLM-based agent","volume-title":"Proc. 31st Int. Conf. Comput. Linguist.","author":"Li","year":"2025"},{"key":"ref88","article-title":"Learning to select in-context examples from reward","author":"Li"},{"key":"ref89","article-title":"Visual prompt selection for in-context learning segmentation","author":"Suo","year":"2024"},{"key":"ref90","first-page":"1","article-title":"Large language models are human-level prompt engineers","volume-title":"Proc. 11th Int. Conf. Learn. Represent.","author":"Zhou","year":"2022"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.3233\/FAIA240720"},{"key":"ref92","article-title":"Rewards-in-context: Multi-objective alignment of foundation models with dynamic preference adjustment","author":"Yang","year":"2024"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.92"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2400"},{"key":"ref96","article-title":"On the essence and prospect: An investigation of alignment approaches for big models","author":"Xu","year":"2024"},{"key":"ref97","article-title":"A comprehensive survey of direct preference optimization: Datasets, theories, variants, and applications","author":"Xiao","year":"2024"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.52202\/079017-4128"},{"key":"ref99","article-title":"StepDPO: Stepwise preference optimization for longchain reasoning of LLMs","author":"Lai","year":"2024"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.775"},{"key":"ref101","article-title":"A comprehensive survey of direct preference optimization: Datasets, theories, variants, and applications","author":"Xu","year":"2024"},{"key":"ref102","article-title":"How to evaluate reward models for RLHF","author":"Frick","year":"2024"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-naacl.96"},{"key":"ref104","article-title":"RM-Bench: Benchmarking reward models of language models with subtlety and style","author":"Liu","year":"2024"},{"key":"ref105","article-title":"RMB: Comprehensively benchmarking reward models in LLM alignment","author":"Zhou","year":"2024"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1230"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.02296"},{"key":"ref108","first-page":"1","article-title":"MJ-Bench: Is your multimodal reward model really a good judge?","volume-title":"Proc. ICML Workshop Found. Models in Wild","author":"Chen","year":"2024"},{"key":"ref109","article-title":"Multimodal RewardBench: Holistic evaluation of reward models for vision language models","author":"Yasunaga","year":"2025"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.3"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.877"}],"container-title":["IEEE Transactions on Artificial Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/9078688\/11589479\/11361384.pdf?arnumber=11361384","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T19:39:10Z","timestamp":1782934750000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11361384\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":111,"journal-issue":{"issue":"7"},"URL":"https:\/\/doi.org\/10.1109\/tai.2026.3654921","relation":{},"ISSN":["2691-4581"],"issn-type":[{"value":"2691-4581","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]}}}