{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T23:10:17Z","timestamp":1780355417424,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T00:00:00Z","timestamp":1746662400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,8]]},"DOI":"10.1145\/3701716.3717570","type":"proceedings-article","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T16:12:56Z","timestamp":1748016776000},"page":"2120-2128","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Adaptive Confidence-aware Preference-based Reinforcement Learning with Noisy Feedback"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-1146-9715","authenticated-orcid":false,"given":"Yuhao","family":"Gong","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0918-7524","authenticated-orcid":false,"given":"Zhenbo","family":"Lu","sequence":"additional","affiliation":[{"name":"Hefei Comprehensive National Science Center, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4612-508X","authenticated-orcid":false,"given":"Wanxuan","family":"Lu","sequence":"additional","affiliation":[{"name":"Aerospace Information Research Institute, CAS, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1690-9836","authenticated-orcid":false,"given":"Wengang","family":"Zhou","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2188-3028","authenticated-orcid":false,"given":"Houqiang","family":"Li","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,5,23]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645610"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.2307\/2334029"},{"key":"e_1_3_2_2_3_1","volume-title":"RIME: Robust Preference-based Reinforcement Learning with Noisy Preferences. International Conference on Machine Learning","author":"Cheng Jie","year":"2024","unstructured":"Jie Cheng, Gang Xiong, Xingyuan Dai, Qinghai Miao, Yisheng Lv, and Fei-Yue Wang. 2024. RIME: Robust Preference-based Reinforcement Learning with Noisy Preferences. International Conference on Machine Learning (2024)."},{"key":"e_1_3_2_2_4_1","volume-title":"Deep reinforcement learning from human preferences. Advances in neural information processing systems","author":"Christiano Paul F","year":"2017","unstructured":"Paul F Christiano, Jan Leike, Tom Brown, Miljan Martic, Shane Legg, and Dario Amodei. 2017. Deep reinforcement learning from human preferences. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583388"},{"key":"e_1_3_2_2_6_1","volume-title":"International Conference on Learning Representations","author":"Hejna Joey","year":"2024","unstructured":"Joey Hejna, Rafael Rafailov, Harshit Sikchi, Chelsea Finn, Scott Niekum, W Bradley Knox, and Dorsa Sadigh. 2024. Contrastive Prefence learning: Learning from Human feedback without RL. International Conference on Learning Representations (2024)."},{"key":"e_1_3_2_2_7_1","volume-title":"Jaehoon Kim, Min Gu Kwak, Young Joon Park, and Seoung Bum Kim.","author":"Heo Jongkook","year":"2024","unstructured":"Jongkook Heo, Young Jae Lee, Jaehoon Kim, Min Gu Kwak, Young Joon Park, and Seoung Bum Kim. 2024. Mixing Corrupted Preferences for Robust and Feedback-Efficient Preference-Based Reinforcement Learning. Knowledge-Based Systems (2024), 112824."},{"key":"e_1_3_2_2_8_1","volume-title":"International Conference on Learning Representations.","author":"Kim Changyeon","year":"2023","unstructured":"Changyeon Kim, Jongjin Park, Jinwoo Shin, Honglak Lee, Pieter Abbeel, and Kimin Lee. 2023. Preference Transformer: Modeling Human Preferences using Transformers for RL. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19806-9_41"},{"key":"e_1_3_2_2_10_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Lee Kimin","year":"2021","unstructured":"Kimin Lee, Laura Smith, and Pieter Abbeel. 2021a. Pebble: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training. Proceedings of the 38th International Conference on Machine Learning (2021)."},{"key":"e_1_3_2_2_11_1","volume-title":"B-pref: Benchmarking preference-based reinforcement learning. Neural Information Processing Systems","author":"Lee Kimin","year":"2021","unstructured":"Kimin Lee, Laura Smith, Anca Dragan, and Pieter Abbeel. 2021b. B-pref: Benchmarking preference-based reinforcement learning. Neural Information Processing Systems (2021)."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645534"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645386"},{"key":"e_1_3_2_2_14_1","volume-title":"International Conference on Learning Representations","author":"Liang Xinran","year":"2022","unstructured":"Xinran Liang, Katherine Shu, Kimin Lee, and Pieter Abbeel. 2022. Reward uncertainty for exploration in preference-based reinforcement learning. International Conference on Learning Representations (2022)."},{"key":"e_1_3_2_2_15_1","first-page":"22270","article-title":"Meta-reward-net: Implicitly differentiable reward learning for preference-based reinforcement learning","volume":"35","author":"Liu Runze","year":"2022","unstructured":"Runze Liu, Fengshuo Bai, Yali Du, and Yaodong Yang. 2022. Meta-reward-net: Implicitly differentiable reward learning for preference-based reinforcement learning. Advances in Neural Information Processing Systems, Vol. 35 (2022), 22270--22284.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_16_1","volume-title":"International conference on machine learning. PMLR, 6226--6236","author":"Liu Yang","year":"2020","unstructured":"Yang Liu and Hongyi Guo. 2020. Peer loss functions: Learning from noisy labels without knowing noise rates. In International conference on machine learning. PMLR, 6226--6236."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583467"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3647982"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Volodymyr Mnih Koray Kavukcuoglu David Silver Andrei A Rusu Joel Veness Marc G Bellemare Alex Graves Martin Riedmiller Andreas K Fidjeland Georg Ostrovski et al. 2015. Human-level control through deep reinforcement learning. nature Vol. 518 7540 (2015) 529--533.","DOI":"10.1038\/nature14236"},{"key":"e_1_3_2_2_20_1","volume-title":"International Conference on Learning Representations","author":"Nguyen Duc Tam","year":"2020","unstructured":"Duc Tam Nguyen, Chaithanya Kumar Mummadi, Thi Phuong Nhung Ngo, Thi Hoai Phuong Nguyen, Laura Beggel, and Thomas Brox. 2020. Self: Learning to filter noisy labels with self-ensembling. International Conference on Learning Representations (2020)."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.12125"},{"key":"e_1_3_2_2_22_1","volume-title":"International Conference on Learning Representations","author":"Park Jongjin","year":"2022","unstructured":"Jongjin Park, Younggyo Seo, Jinwoo Shin, Honglak Lee, Pieter Abbeel, and Kimin Lee. 2022. SURF: Semi-supervised reward learning with data augmentation for feedback-efficient preference-based reinforcement learning. International Conference on Learning Representations (2022)."},{"key":"e_1_3_2_2_23_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2024. Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583298"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583523"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645365"},{"key":"e_1_3_2_2_27_1","first-page":"4409","article-title":"Instance-dependent label-noise learning under a structural causal model","volume":"34","author":"Yao Yu","year":"2021","unstructured":"Yu Yao, Tongliang Liu, Mingming Gong, Bo Han, Gang Niu, and Kun Zhang. 2021. Instance-dependent label-noise learning under a structural causal model. Advances in Neural Information Processing Systems, Vol. 34 (2021), 4409--4420.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_28_1","volume-title":"Conference on robot learning. PMLR, 1094--1100","author":"Yu Tianhe","year":"2020","unstructured":"Tianhe Yu, Deirdre Quillen, Zhanpeng He, Ryan Julian, Karol Hausman, Chelsea Finn, and Sergey Levine. 2020. Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning. In Conference on robot learning. PMLR, 1094--1100."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645615"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583313"},{"key":"e_1_3_2_2_31_1","volume-title":"International Conference on Machine Learning. PMLR, 27412--27427","author":"Zhu Zhaowei","year":"2022","unstructured":"Zhaowei Zhu, Zihao Dong, and Yang Liu. 2022. Detecting corrupted labels without training a model to predict. In International Conference on Machine Learning. PMLR, 27412--27427."},{"key":"e_1_3_2_2_32_1","volume-title":"Modeling purposeful adaptive behavior with the principle of maximum causal entropy","author":"Ziebart Brian D","unstructured":"Brian D Ziebart. 2010. Modeling purposeful adaptive behavior with the principle of maximum causal entropy. Carnegie Mellon University."}],"event":{"name":"WWW '25: The ACM Web Conference 2025","location":"Sydney NSW Australia","acronym":"WWW '25","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Companion Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3717570","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3701716.3717570","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,8]],"date-time":"2025-10-08T03:05:56Z","timestamp":1759892756000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3701716.3717570"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,8]]},"references-count":32,"alternative-id":["10.1145\/3701716.3717570","10.1145\/3701716"],"URL":"https:\/\/doi.org\/10.1145\/3701716.3717570","relation":{},"subject":[],"published":{"date-parts":[[2025,5,8]]},"assertion":[{"value":"2025-05-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}