{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T05:35:49Z","timestamp":1783575349524,"version":"3.55.0"},"reference-count":69,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,15]],"date-time":"2024-12-15T00:00:00Z","timestamp":1734220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,15]],"date-time":"2024-12-15T00:00:00Z","timestamp":1734220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,15]]},"DOI":"10.1109\/bigdata62323.2024.10825538","type":"proceedings-article","created":{"date-parts":[[2025,1,16]],"date-time":"2025-01-16T18:31:23Z","timestamp":1737052283000},"page":"1664-1671","source":"Crossref","is-referenced-by-count":4,"title":["Mitigating Sycophancy in Large Language Models via Direct Preference Optimization"],"prefix":"10.1109","author":[{"given":"Azal Ahmad","family":"Khan","sequence":"first","affiliation":[{"name":"University of Minnesota-Twin Cities,Department of Computer Science and Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sayan","family":"Alam","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Guwahati"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinran","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Minnesota-Twin Cities,Department of Computer Science and Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ahmad Faraz","family":"Khan","sequence":"additional","affiliation":[{"name":"Virginia Tech,Department of Computer Science"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Debanga Raj","family":"Neog","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Guwahati"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ali","family":"Anwar","sequence":"additional","affiliation":[{"name":"University of Minnesota-Twin Cities,Department of Computer Science and Engineering"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"issue":"8","key":"ref1","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume-title":"OpenAI blog","volume":"1","author":"Radford","year":"2019"},{"key":"ref2","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Advances in neural information processing systems","volume":"33","author":"Brown","year":"2020"},{"key":"ref3","article-title":"Gpt-4 technical report","author":"Achiam","year":"2023"},{"key":"ref4","article-title":"Palm: Scaling language modeling with pathways","author":"Chowdhery","year":"2022"},{"key":"ref5","article-title":"Palm 2 technical report","author":"Anil","year":"2023"},{"key":"ref6","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref7","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023"},{"key":"ref8","article-title":"Deep reinforcement learning from human preferences","volume-title":"Advances in neural information processing systems","volume":"30","author":"Christiano"},{"key":"ref9","article-title":"Scaling instruction-finetuned language models","author":"Chung","year":"2022"},{"key":"ref10","article-title":"Evaluating large language models trained on code","author":"Chen","year":"2021"},{"key":"ref11","article-title":"Discovering language model behaviors with model-written evaluations","author":"Perez","year":"2022"},{"key":"ref12","article-title":"Simple synthetic data reduces sycophancy in large language models","author":"Wei","year":"2023"},{"key":"ref13","article-title":"When large language models contradict humans? large language models\u2019 sycophantic behaviour","author":"Ranaldi","year":"2023"},{"key":"ref14","article-title":"Towards understanding sycophancy in language models","author":"Sharma","year":"2023"},{"key":"ref15","article-title":"Trustllm: Trustworthiness in large language models","author":"Sun","year":"2024"},{"key":"ref16","article-title":"A survey of large language models","author":"Zhao","year":"2023"},{"key":"ref17","article-title":"Aligning large multimodal models with factually augmented rlhf","author":"Sun","year":"2023"},{"key":"ref18","article-title":"Direct preference optimization: Your language model is secretly a reward model","author":"Rafailov","year":"2023"},{"key":"ref19","article-title":"Preference ranking optimization for human alignment","author":"Song","year":"2023"},{"key":"ref20","article-title":"Rrhf: Rank responses to align language models with human feedback without tears","author":"Yuan","year":"2023"},{"key":"ref21","article-title":"Making large language models better reasoners with alignment","author":"Wang","year":"2023"},{"key":"ref22","article-title":"Pangu-coder2: Boosting large language models for code with ranking feedback","author":"Shen","year":"2023"},{"key":"ref23","article-title":"Fine-tuning language models for factuality","author":"Tian","year":"2023"},{"key":"ref24","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1031"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W19-4302"},{"key":"ref27","article-title":"Scaling laws for neural language models","author":"Kaplan","year":"2020"},{"key":"ref28","first-page":"2790","article-title":"Parameter-efficient transfer learning for nlp","volume-title":"International Conference on machine learning","author":"Houlsby"},{"key":"ref29","article-title":"Lora: Low-rank adaptation of large language models","author":"Hu","year":"2021"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"ref31","article-title":"Qlora: Efficient finetuning of quantized llms","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Dettmers"},{"key":"ref32","article-title":"Parameter-efficient fine-tuning methods for pretrained language models: A critical review and assessment","author":"Xu","year":"2023"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.740"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1371"},{"key":"ref35","article-title":"Bert post-training for review reading comprehension and aspect-based sentiment analysis","author":"Xu","year":"2019"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.244"},{"key":"ref37","article-title":"Fine-tuning language models from human preferences","author":"Ziegler","year":"2019"},{"key":"ref38","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Advances in neural information processing systems","volume":"35","author":"Ouyang","year":"2022"},{"key":"ref39","first-page":"3008","article-title":"Learning to summarize with human feedback","volume":"33","author":"Stiennon","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref40","article-title":"Constitutional ai: Harmlessness from ai feedback","author":"Bai","year":"2022"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.225"},{"key":"ref42","article-title":"A general language assistant as a laboratory for alignment","author":"Askell","year":"2021"},{"key":"ref43","article-title":"On the opportunities and risks of foundation models","author":"Bommasani","year":"2021"},{"key":"ref44","first-page":"10 835","article-title":"Scaling laws for reward model overoptimization","volume-title":"International Conference on Machine Learning","author":"Gao"},{"key":"ref45","article-title":"The curse of recursion: Training on generated data makes models forget","author":"Shumailov","year":"2023"},{"key":"ref46","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Rafailov"},{"key":"ref47","article-title":"Are you sure? challenging llms leads to performance drops in the flipflop experiment","author":"Laban","year":"2023"},{"key":"ref48","article-title":"How to catch an ai liar: Lie detection in black-box llms by asking unrelated questions","author":"Pacchiardi","year":"2023"},{"key":"ref49","article-title":"Secrets of rlhf in large language models part ii: Reward modeling","author":"Wang","year":"2024"},{"key":"ref50","article-title":"Instruction tuning with gpt-4","author":"Peng","year":"2023"},{"key":"ref51","article-title":"Statistical rejection sampling improves preference optimization","author":"Liu","year":"2023"},{"key":"ref52","article-title":"A general theoretical paradigm to understand learning from human preferences","author":"Azar","year":"2023"},{"key":"ref53","article-title":"Kto: Model alignment as prospect theoretic optimization","author":"Ethayarajh","year":"2024"},{"key":"ref54","article-title":"Large language model alignment: A survey","author":"Shen","year":"2023"},{"key":"ref55","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2023.findings-emnlp.892","article-title":"Ethical reasoning over moral alignment: A case and framework for in-context ethical policies in llms","author":"Rao","year":"2023"},{"key":"ref56","article-title":"Knowledgeable preference alignment for llms in domain-specific question answering","author":"Zhang","year":"2023"},{"key":"ref57","article-title":"Beavertails: Towards improved safety alignment of llm via a human-preference dataset","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Ji"},{"key":"ref58","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.emnlp-main.585","article-title":"Inferaligner: Inference-time alignment for harmlessness through cross-model guidance","author":"Wang","year":"2024"},{"key":"ref59","article-title":"Aligning large language models for clinical tasks","author":"Manathunga","year":"2023"},{"key":"ref60","article-title":"Towards better humanagent alignment: Assessing task utility in llm-powered applications","author":"Arabzadeh","year":"2024"},{"key":"ref61","first-page":"2096","article-title":"Human preference score: Better aligning text-to-image models with human preference","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"Wu"},{"key":"ref62","article-title":"Pick-a-pic: An open dataset of user preferences for text-to-image generation","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Kirstain"},{"issue":"3\/4","key":"ref63","first-page":"324","article-title":"Rank analysis of incomplete block designs: I. the method of paired comparisons","volume-title":"Biometrika","volume":"39","author":"Bradley","year":"1952"},{"key":"ref64","article-title":"Mistral 7b","author":"Jiang","year":"2023"},{"key":"ref65","article-title":"Stability ai launches the first of its stablelm suite of language models","author":"AI","year":"2024"},{"key":"ref66","article-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","author":"Chiang","year":"2023"},{"key":"ref67","article-title":"Qlora: Efficient finetuning of quantized llms","author":"Dettmers","year":"2023"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1145\/3649506"},{"key":"ref69","article-title":"Large language models still can\u2019t plan (a benchmark for llms on planning and reasoning about change)","author":"Valmeekam","year":"2022"}],"event":{"name":"2024 IEEE International Conference on Big Data (BigData)","location":"Washington, DC, USA","start":{"date-parts":[[2024,12,15]]},"end":{"date-parts":[[2024,12,18]]}},"container-title":["2024 IEEE International Conference on Big Data (BigData)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10824975\/10824942\/10825538.pdf?arnumber=10825538","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T07:47:09Z","timestamp":1737100029000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10825538\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,15]]},"references-count":69,"URL":"https:\/\/doi.org\/10.1109\/bigdata62323.2024.10825538","relation":{},"subject":[],"published":{"date-parts":[[2024,12,15]]}}}