{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T21:10:30Z","timestamp":1779311430418,"version":"3.51.4"},"reference-count":63,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"am","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CCF-2232907"],"award-info":[{"award-number":["CCF-2232907"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Signal Process."],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/tsp.2025.3607114","type":"journal-article","created":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T17:49:18Z","timestamp":1757353758000},"page":"3886-3901","source":"Crossref","is-referenced-by-count":1,"title":["Human Feedback Attack on Online RLHF: Attack and Robust Defense"],"prefix":"10.1109","volume":"73","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9273-3889","authenticated-orcid":false,"given":"Chenye","family":"Yang","sequence":"first","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of California, Davis, Davis, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3905-3142","authenticated-orcid":false,"given":"Mo","family":"Lyu","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of California, Davis, Davis, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0595-9398","authenticated-orcid":false,"given":"Guanlin","family":"Liu","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of California, Davis, Davis, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9493-8248","authenticated-orcid":false,"given":"Lifeng","family":"Lai","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering, University of California, Davis, Davis, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"4302","article-title":"Deep reinforcement learning from human preferences","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Christiano","year":"2017"},{"key":"ref2","article-title":"Open problems and fundamental limitations of reinforcement learning from human feedback","author":"Casper","year":"2023"},{"key":"ref3","article-title":"Training a helpful and harmless assistant with reinforcement learning from human feedback","author":"Bai","year":"2022"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/sp61157.2025.00094"},{"key":"ref5","article-title":"Universal jailbreak backdoors from poisoned human feedback","author":"Rando","year":"2023"},{"key":"ref6","article-title":"Best-of-venom: Attacking RLHF by injecting poisoned preference data","author":"Baumg\u00e4rtner","year":"2024"},{"key":"ref7","first-page":"1117","article-title":"Policy teaching via data poisoning in learning from human preferences","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Nika","year":"2025"},{"key":"ref8","article-title":"Is poisoning a real threat to LLM alignment? Maybe more so than you think","author":"Pathmanathan","year":"2024"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.2307\/2334029"},{"key":"ref10","article-title":"Fine-tuning language models from human preferences","author":"Ziegler","year":"2019"},{"key":"ref11","first-page":"53728","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Rafailov","year":"2023"},{"key":"ref12","first-page":"6263","article-title":"Dueling RL: Reinforcement learning with trajectory preferences","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","volume":"206","author":"Saha","year":"2023"},{"key":"ref13","first-page":"76006","article-title":"Is RLHF more difficult than standard RL? A theoretical perspective","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Wang","year":"2023"},{"key":"ref14","first-page":"43037","article-title":"Principled reinforcement learning with human feedback from pairwise or k-wise comparisons","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"202","author":"Zhu","year":"2023"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2019.2943232"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2023.3328111"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1201\/chmonstaapp"},{"key":"ref18","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref19","first-page":"3640","article-title":"Adversarial attacks on stochastic bandits","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"31","author":"Jun","year":"2018"},{"key":"ref20","first-page":"4042","article-title":"Data poisoning attacks on stochastic bandits","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"97","author":"Liu","year":"2019"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2020.3021525"},{"key":"ref22","article-title":"Adversarial attacks on adversarial bandits","author":"Ma","year":"2023"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/IEEECONF59524.2023.10476992"},{"key":"ref24","article-title":"Efficient action poisoning attacks on linear contextual bandits","author":"Liu","year":"2021"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-62416-7_19"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-32430-8_14"},{"key":"ref27","first-page":"14570","article-title":"Policy poisoning in batch reinforcement learning and control","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Ma","year":"2019"},{"key":"ref28","first-page":"11225","article-title":"Adaptive reward-poisoning attacks against reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"119","author":"Zhang","year":"2020"},{"key":"ref29","first-page":"12400","article-title":"Provably efficient black-box action poisoning attacks against reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Liu","year":"2021"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2023.3340028"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2021.3115943"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2022.3153135"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/3188745.3188918"},{"key":"ref34","first-page":"1562","article-title":"Better algorithms for stochastic bandits with adversarial corruptions","volume-title":"Proc. Conf. Learn. Theory","volume":"99","author":"Gupta","year":"2019"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5821"},{"key":"ref36","first-page":"991","article-title":"Stochastic linear bandits robust to adversarial attacks","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","volume":"130","author":"Bogunovic","year":"2021"},{"key":"ref37","first-page":"19943","article-title":"Adversarial bandits with corruptions: Regret lower bound and no-regret algorithm","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Yang","year":"2020"},{"key":"ref38","first-page":"7111","article-title":"Robust stochastic linear contextual bandits under adversarial attacks","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","author":"Ding","year":"2022"},{"key":"ref39","first-page":"3092","article-title":"The intrinsic robustness of stochastic bandits to strategic manipulation","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"119","author":"Feng","year":"2020"},{"key":"ref40","article-title":"Robust Lipschitz bandits to adversarial corruptions","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Kang","year":"2023"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2024.3486240"},{"key":"ref42","first-page":"6215","article-title":"Action robust reinforcement learning and applications in continuous control","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"97","author":"Tessler","year":"2019"},{"key":"ref43","first-page":"5757","article-title":"Corruption-robust offline reinforcement learning","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","volume":"151","author":"Zhang","year":"2022"},{"key":"ref44","first-page":"36208","article-title":"Corruption-robust offline reinforcement learning with general function approximation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Ye","year":"2023"},{"key":"ref45","article-title":"Rime: Robust preference-based reinforcement learning with noisy preferences","author":"Cheng","year":"2024"},{"key":"ref46","article-title":"Corruption robust offline reinforcement learning with human feedback","author":"Mandal","year":"2024"},{"key":"ref47","article-title":"Reward-robust RLHF in LLMs","author":"Yan","year":"2024"},{"key":"ref48","article-title":"Provably robust DPO: Aligning language models with noisy feedback","author":"Chowdhury","year":"2024"},{"key":"ref49","article-title":"Towards robust alignment of language models: distributionally robustifying direct preference optimization","author":"Wu","year":"2024"},{"key":"ref50","article-title":"Iterative preference learning from human feedback: Bridging theory and practice for RLHF under KL-constraint","author":"Xiong","year":"2023"},{"key":"ref51","article-title":"Statistical rejection sampling improves preference optimization","author":"Liu","year":"2023"},{"key":"ref52","first-page":"4843","article-title":"Differentially private reward estimation with preference feedback","volume-title":"Proc. Int. Conf. Artif. Intell. Statist.","volume":"238","author":"Chowdhury","year":"2024"},{"key":"ref53","article-title":"Making RL with preference-based feedback efficient via randomization","author":"Wu","year":"2023"},{"key":"ref54","article-title":"BPR: Bayesian personalized ranking from implicit feedback","author":"Rendle","year":"2012"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i13.29346"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/471"},{"key":"ref57","first-page":"24401","article-title":"Efficient adversarial attacks on online multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Liu","year":"2024"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1145\/3450267.3450537"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.2307\/1403865"},{"key":"ref60","first-page":"1141","article-title":"Control regularization for reduced variance reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"97","author":"Cheng","year":"2019"},{"key":"ref61","first-page":"22","article-title":"Constrained policy optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"70","author":"Achiam","year":"2017"},{"key":"ref62","first-page":"263","article-title":"Minimax regret bounds for reinforcement learning","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"70","author":"Azar","year":"2017"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1145\/3733592"}],"container-title":["IEEE Transactions on Signal Processing"],"original-title":[],"link":[{"URL":"https:\/\/ieeexplore.ieee.org\/ielam\/78\/10807692\/11153038-aam.pdf","content-type":"application\/pdf","content-version":"am","intended-application":"syndication"},{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/78\/10807692\/11153038.pdf?arnumber=11153038","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T18:12:48Z","timestamp":1762539168000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11153038\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":63,"URL":"https:\/\/doi.org\/10.1109\/tsp.2025.3607114","relation":{},"ISSN":["1053-587X","1941-0476"],"issn-type":[{"value":"1053-587X","type":"print"},{"value":"1941-0476","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]}}}