{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T18:17:50Z","timestamp":1783102670246,"version":"3.54.6"},"reference-count":53,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2023YFB3308004"],"award-info":[{"award-number":["2023YFB3308004"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["92267107"],"award-info":[{"award-number":["92267107"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2025ZD0123703"],"award-info":[{"award-number":["2025ZD0123703"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.neucom.2026.133996","type":"journal-article","created":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T06:43:55Z","timestamp":1779259435000},"page":"133996","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Mixed-policy preference optimization with self-generated non-preferred responses and off-policy preference distillation"],"prefix":"10.1016","volume":"695","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9851-5528","authenticated-orcid":false,"given":"Binrui","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zikai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6867-2063","authenticated-orcid":false,"given":"Yongping","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-4753-428X","authenticated-orcid":false,"given":"Mingyang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.133996_bib0005","author":"DeepSeek-AI"},{"key":"10.1016\/j.neucom.2026.133996_bib0010","author":"Achiam"},{"key":"10.1016\/j.neucom.2026.133996_bib0015","series-title":"The Twelfth International Conference on Learning Representations (ICLR), OpenReview","article-title":"Gaining wisdom from setbacks: Aligning large language models via mistake analysis","author":"Chen","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0020","series-title":"The Twelfth International Conference on Learning Representations (ICLR)","article-title":"Can large language models infer causation from correlation?","author":"Jin","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0025","author":"Kang"},{"key":"10.1016\/j.neucom.2026.133996_bib0030","series-title":"Forty-first International Conference on Machine Learning (ICML), PMLR","article-title":"Self-alignment of large language models via monopolylogue-based social scene simulation","author":"Pang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0035","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (ACL)","article-title":"Self-alignment for factuality: Mitigating hallucinations in llms via self-evaluation","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0040","author":"Gao"},{"key":"10.1016\/j.neucom.2026.133996_bib0045","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (ACL)","article-title":"Whose preferences? differences in fairness preferences and their impact on the fairness of ai utilizing human feedback","author":"Lerner","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0050","author":"Thakkar"},{"key":"10.1016\/j.neucom.2026.133996_bib0055","series-title":"Forty-First International Conference on Machine Learning (ICML)","first-page":"6621","article-title":"Self-play fine-tuning converts weak language models to strong language models","author":"Chen","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0060","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume":"vol. 35","author":"Ouyang","year":"2022"},{"key":"10.1016\/j.neucom.2026.133996_bib0065","author":"Schulman"},{"key":"10.1016\/j.neucom.2026.133996_bib0070","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (ACL)","first-page":"13484","article-title":"Self-instruct: Aligning language models with self-generated instructions","author":"Wang","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0075","author":"Wu"},{"key":"10.1016\/j.neucom.2026.133996_bib0080","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"53728","article-title":"Direct preference optimization: Your language model is secretly a reward model","author":"Rafailov","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0085","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (EMNLP), Association for Computational Linguistics","first-page":"11170","article-title":"Orpo: Monolithic preference optimization without reference model","author":"Hong","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0090","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"124198","article-title":"Simpo: Simple preference optimization with a reference-free reward","volume":"vol. 37","author":"Meng","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0095","article-title":"A comprehensive survey on pretrained foundation models: a history from bert to chatgpt","author":"Zhou","year":"2024","journal-title":"Int. J. Mach. Learn. Cybern."},{"key":"10.1016\/j.neucom.2026.133996_bib0100","series-title":"Forty-first International Conference on Machine Learning (ICML), OpenReview","article-title":"Offline actor-critic reinforcement learning scales to large models","author":"Springenberg","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0105","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"97","article-title":"Natural actor-critic for robust reinforcement learning with function approximation","volume":"vol. 36","author":"Zhou","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0110","series-title":"Forty-first International Conference on Machine Learning (ICML)","article-title":"Rlaif vs. rlhf: Scaling reinforcement learning from human feedback with ai feedback","author":"Lee","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0115","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"4299","article-title":"Deep reinforcement learning from human preferences","volume":"vol. 30","author":"Christiano","year":"2017"},{"key":"10.1016\/j.neucom.2026.133996_bib0120","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","first-page":"2511","article-title":"Principle-driven self-alignment of language models from scratch with minimal human supervision","volume":"vol. 36","author":"Sun","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0125","article-title":"Open problems and fundamental limitations of reinforcement learning from human feedback","volume":"2023","author":"Casper","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.133996_bib0130","series-title":"International Conference on Artificial Intelligence and Statistics, PMLR","first-page":"4447","article-title":"A general theoretical paradigm to understand learning from human preferences","author":"Azar","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0135","series-title":"The Twelfth International Conference on Learning Representations (ICLR)","article-title":"Beyond reverse kl: Generalizing direct preference optimization with diverse divergence constraints","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0140","series-title":"Forty-first International Conference on Machine Learning (ICML)","article-title":"Towards efficient exact optimization of language model alignment","author":"Ji","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0145","author":"Pal"},{"key":"10.1016\/j.neucom.2026.133996_bib0150","series-title":"Forty-first International Conference on Machine Learning (ICML)","article-title":"Token-level direct preference optimization","author":"Zeng","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0155","doi-asserted-by":"crossref","first-page":"10935","DOI":"10.52202\/075280-0482","article-title":"Rrhf: Rank responses to align language models with human feedback","volume":"36","author":"Yuan","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst. (neurips)"},{"key":"10.1016\/j.neucom.2026.133996_bib0160","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"Aligner: Efficient alignment by learning to correct","author":"Ji","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0165","series-title":"Findings of the Association for Computational Linguistics: ACL 2023","first-page":"8003","article-title":"Distilling Step-by-Step! Outperforming Larger Language Models with Less Training Data and Smaller Model Sizes","author":"Hsieh","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0170","author":"Stanton"},{"key":"10.1016\/j.neucom.2026.133996_bib0175","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"10900","article-title":"Revisiting Knowledge Distillation for Autoregressive Language Models","author":"Zhong","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0180","series-title":"Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics","first-page":"6934","article-title":"Uncertainty-Aware Curriculum Learning for Neural Machine Translation","author":"Zhou","year":"2020"},{"key":"10.1016\/j.neucom.2026.133996_bib0185","series-title":"The Twelfth International Conference on Learning Representations","article-title":"Can LLMs Express Their Uncertainty? An Empirical Evaluation of Confidence Elicitation in LLMs","author":"Xiong","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0190","series-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","first-page":"21635","article-title":"Can LLMs Learn Uncertainty on Their Own? Expressing Uncertainty Effectively in A Self-Training Manner","author":"Liu","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0195","series-title":"Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 2: Short Papers)","first-page":"841","article-title":"Token-Level Self-Evolution Training for Sequence-to-Sequence Learning","author":"Peng","year":"2023"},{"key":"10.1016\/j.neucom.2026.133996_bib0200","series-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)","first-page":"11087","article-title":"Uncertainty Aware Learning for Language Model Alignment","author":"Wang","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0205","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TBDATA.2024.3524104","article-title":"Aligning Crowd-Sourced Human Feedback for Reinforcement Learning on Code Generation by Large Language Models","author":"Wong","year":"2024","journal-title":"IEEE Trans. Big Data"},{"key":"10.1016\/j.neucom.2026.133996_bib0210","author":"Kveton"},{"key":"10.1016\/j.neucom.2026.133996_bib0215","series-title":"Proceedings of the 41st International Conference on Machine Learning","first-page":"36577","article-title":"Active Preference Learning for Large Language Models","author":"Muldrew","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0220","author":"Yang"},{"key":"10.1016\/j.neucom.2026.133996_bib0225","author":"Touvron"},{"key":"10.1016\/j.neucom.2026.133996_bib0230","author":"Grattafiori"},{"key":"10.1016\/j.neucom.2026.133996_bib0235","author":"Jiang"},{"key":"10.1016\/j.neucom.2026.133996_bib0240","author":"Chu"},{"key":"10.1016\/j.neucom.2026.133996_bib0245","author":"Bai"},{"key":"10.1016\/j.neucom.2026.133996_bib0250","series-title":"Forty-first International Conference on Machine Learning (ICML)","article-title":"Ultrafeedback: Boosting language models with scaled ai feedback","author":"Cui","year":"2024"},{"key":"10.1016\/j.neucom.2026.133996_bib0255","series-title":"The Tenth International Conference on Learning Representations (ICLR)","article-title":"Lora: Low-rank adaptation of large language models","author":"Hu","year":"2022"},{"key":"10.1016\/j.neucom.2026.133996_bib0260","series-title":"SC20: International Conference for High Performance Computing, Networking, Storage and Analysis","first-page":"1","article-title":"Zero: Memory optimizations toward training trillion parameter models","author":"Rajbhandari","year":"2020"},{"key":"10.1016\/j.neucom.2026.133996_bib0265","author":"Shao"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226013949?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226013949?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T17:40:14Z","timestamp":1783100414000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226013949"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":53,"alternative-id":["S0925231226013949"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.133996","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Mixed-policy preference optimization with self-generated non-preferred responses and off-policy preference distillation","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.133996","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133996"}}