{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T11:04:59Z","timestamp":1774868699578,"version":"3.50.1"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2026,1]]},"DOI":"10.1007\/s10489-025-07079-9","type":"journal-article","created":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T01:34:34Z","timestamp":1769650474000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SIPA: a self-iterative preference alignment method for generative language models"],"prefix":"10.1007","volume":"56","author":[{"given":"Yongping","family":"Du","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9851-5528","authenticated-orcid":false,"given":"Binrui","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yin","family":"Hou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Honggui","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,29]]},"reference":[{"key":"7079_CR1","unstructured":"Azar MG, Guo ZD, Piot B, et al (2024) A General Theoretical Paradigm to Understand Learning from Human Preferences. In: Proceedings of the 27th international conference on artificial intelligence and statistics. PMLR, pp 4447\u20134455"},{"key":"7079_CR2","doi-asserted-by":"publisher","unstructured":"Bai Y, Jones A, Ndousse K, et al (2022) Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback. https:\/\/doi.org\/10.48550\/arXiv.2204.05862","DOI":"10.48550\/arXiv.2204.05862"},{"key":"7079_CR3","unstructured":"Brown TB, Mann B, Ryder N, et al (2020) Language models are few-shot learners. In: Proceedings of the 34th international conference on neural information processing systems. Curran Associates Inc., Red Hook, NY, USA, NIPS \u201920, pp 1877\u20131901"},{"key":"7079_CR4","unstructured":"Chen K, Wang C, Yang K, et al (2023) Gaining Wisdom from Setbacks: Aligning Large Language Models via Mistake Analysis. In: The twelfth international conference on learning representations"},{"key":"7079_CR5","unstructured":"Chen Z, Deng Y, Yuan H, et al (2024) Self-Play Fine-Tuning Converts Weak Language Models to Strong Language Models. In: Forty-first international conference on machine learning"},{"key":"7079_CR6","unstructured":"Christiano PF, Leike J, Brown T, et al (2017) Deep Reinforcement Learning from Human Preferences. In: Advances in neural information processing systems, vol 30. Curran Associates, Inc"},{"key":"7079_CR7","doi-asserted-by":"publisher","unstructured":"Cui G, Yuan L, Ding N, et al (2023) UltraFeedback: boosting language models with high-quality feedback. https:\/\/doi.org\/10.48550\/arXiv.2310.01377","DOI":"10.48550\/arXiv.2310.01377"},{"key":"7079_CR8","doi-asserted-by":"publisher","unstructured":"DeepSeek-AI, Guo D, Yang D, et al (2025) DeepSeek-R1: incentivizing reasoning capability in LLMs via reinforcement learning. https:\/\/doi.org\/10.48550\/arXiv.2501.12948","DOI":"10.48550\/arXiv.2501.12948"},{"key":"7079_CR9","unstructured":"Dong H, Xiong W, Goyal D, et al (2023) RAFT: Reward rAnked FineTuning for generative foundation model alignment. Transactions on Machine Learning Research"},{"key":"7079_CR10","doi-asserted-by":"publisher","unstructured":"Dou ZY, Yang CF, Wu X, et al (2024) Re-ReST: Reflection-Reinforced Self-Training for Language Agents. In: Al-Onaizan Y, Bansal M, Chen YN (eds) Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics, Miami, Florida, USA, pp 15394\u201315411 https:\/\/doi.org\/10.18653\/v1\/2024.emnlp-main.861","DOI":"10.18653\/v1\/2024.emnlp-main.861"},{"key":"7079_CR11","unstructured":"Dubois Y, Li CX, Taori R, et al (2023) AlpacaFarm: A simulation framework for methods that learn from human feedback. Advances in Neural Information Processing Systems 36"},{"key":"7079_CR12","unstructured":"Guo S, Zhang B, Liu T, et al (2024) Direct language model alignment from online AI feedback. https:\/\/arxiv.org\/abs\/2402.04792v2"},{"key":"7079_CR13","unstructured":"Hejna J, Rafailov R, Sikchi H, et al (2023) Contrastive preference learning: learning from human feedback without reinforcement learning. In: The twelfth international conference on learning representations"},{"key":"7079_CR14","doi-asserted-by":"publisher","unstructured":"Hong J, Lee N, Thorne J (2024) ORPO: monolithic preference optimization without reference model. In: Al-Onaizan Y, Bansal M, Chen YN (eds) Proceedings of the 2024 conference on empirical methods in natural language processing. Association for Computational Linguistics, Miami, Florida, USA, pp 11170\u201311189 https:\/\/doi.org\/10.18653\/v1\/2024.emnlp-main.626","DOI":"10.18653\/v1\/2024.emnlp-main.626"},{"key":"7079_CR15","unstructured":"Hu EJ, Shen Y, Wallis P, et al (2021) LoRA: low-rank adaptation of large language models. In: International conference on learning representations"},{"key":"7079_CR16","unstructured":"Ji H, Lu C, Niu Y, et al (2024) Towards efficient exact optimization of language model alignment. In: Forty-first international conference on machine learning. https:\/\/openreview.net\/forum?id=66k81s33p3"},{"issue":"4","key":"7079_CR17","doi-asserted-by":"publisher","first-page":"383","DOI":"10.1038\/s42256-024-00820-y","volume":"6","author":"HR Kirk","year":"2024","unstructured":"Kirk HR, Vidgen B, R\u00f6ttger P et al (2024) The benefits, risks and bounds of personalizing the alignment of large language models to individuals. Nat Mach Intell 6(4):383\u2013392. https:\/\/doi.org\/10.1038\/s42256-024-00820-y","journal-title":"Nat Mach Intell"},{"key":"7079_CR18","first-page":"47669","volume":"36","author":"A K\u00f6pf","year":"2023","unstructured":"K\u00f6pf A, Kilcher Y, von R\u00fctte D et al (2023) OpenAssistant Conversations - Democratizing Large Language Model Alignment. Adv Neural Inf Process Syst 36:47669\u201347681","journal-title":"Adv Neural Inf Process Syst"},{"key":"7079_CR19","unstructured":"Lee H, Phatale S, Mansoor H, et al (2023) RLAIF: scaling reinforcement learning from human feedback with AI Feedback. https:\/\/arxiv.org\/abs\/2309.00267v2"},{"key":"7079_CR20","unstructured":"Li X, Zhang T, Dubois Y, et al (2023) Alpacaeval: An automatic evaluator of instruction-following models. https:\/\/github.com\/tatsu-lab\/alpaca_eval"},{"key":"7079_CR21","unstructured":"Liu H, Sferrazza C, Abbeel P (2023a) Chain of hindsight aligns language models with feedback. In: The Twelfth international conference on learning representations"},{"key":"7079_CR22","unstructured":"Liu T, Zhao Y, Joshi R, et al (2023b) Statistical rejection sampling improves preference optimization. In: The Twelfth international conference on learning representations"},{"key":"7079_CR23","doi-asserted-by":"crossref","unstructured":"Meng Y, Xia M, Chen D (2024) SimPO: simple preference optimization with a reference-free reward. Adv Neural Inf Process Syst 37:124198\u2013124235. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/hash\/e099c1c9699814af0be873a175361713-Abstract-Conference.html","DOI":"10.52202\/079017-3946"},{"key":"7079_CR24","unstructured":"Meta AI (2024) Llama 3.2: Revolutionizing edge AI and vision with open, customizable models. https:\/\/ai.meta.com\/blog\/llama-3-2-connect-2024-vision-edge-mobile-devices\/"},{"key":"7079_CR25","doi-asserted-by":"publisher","unstructured":"Navigli R, Conia S, Ross B (2023) Biases in large language models: origins, inventory, and discussion. J Data and Information Quality 15(2):10:1\u201310:21. https:\/\/doi.org\/10.1145\/3597307","DOI":"10.1145\/3597307"},{"key":"7079_CR26","first-page":"27730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang L, Wu J, Jiang X et al (2022) Training language models to follow instructions with human feedback. Adv Neural Inf Process Syst 35:27730\u201327744","journal-title":"Adv Neural Inf Process Syst"},{"key":"7079_CR27","doi-asserted-by":"publisher","unstructured":"Pal A, Karkhanis D, Dooley S, et al (2024) Smaug: fixing failure modes of preference Optimisation with DPO-Positive. https:\/\/doi.org\/10.48550\/arXiv.2402.13228","DOI":"10.48550\/arXiv.2402.13228"},{"key":"7079_CR28","unstructured":"Pang RY, Yuan W, Cho K, et al (2024) Iterative reasoning preference optimization. Adv Neural Inf Process Syst 37:116617\u2013116637. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/hash\/d37c9ad425fe5b65304d500c6edcba00-Abstract-Conference.html"},{"key":"7079_CR29","first-page":"53728","volume":"36","author":"R Rafailov","year":"2023","unstructured":"Rafailov R, Sharma A, Mitchell E et al (2023) Direct preference optimization: your language model is secretly a reward model. Adv Neural Inf Process Syst 36:53728\u201353741","journal-title":"Adv Neural Inf Process Syst"},{"key":"7079_CR30","doi-asserted-by":"publisher","unstructured":"Rajbhandari S, Rasley J, Ruwase O, et al (2020) ZeRO: memory optimizations toward training trillion parameter models. In: SC20: International conference for high performance computing, networking, storage and analysis, pp 1\u201316 https:\/\/doi.org\/10.1109\/SC41405.2020.00024","DOI":"10.1109\/SC41405.2020.00024"},{"key":"7079_CR31","doi-asserted-by":"publisher","unstructured":"Shao Z, Wang P, Zhu Q, et al (2024) DeepSeekMath: pushing the limits of mathematical reasoning in open language models. https:\/\/doi.org\/10.48550\/arXiv.2402.03300","DOI":"10.48550\/arXiv.2402.03300"},{"key":"7079_CR32","unstructured":"Stiennon N, Ouyang L, Wu J, et al (2020) Learning to summarize with human feedback. In: Advances in Neural information processing systems, vol 33. Curran Associates, Inc., pp 3008\u20133021"},{"key":"7079_CR33","first-page":"2511","volume":"36","author":"Z Sun","year":"2023","unstructured":"Sun Z, Shen Y, Zhou Q et al (2023) Principle-Driven self-alignment of language models from scratch with minimal human supervision. Adv Neural Inf Process Syst 36:2511\u20132565","journal-title":"Adv Neural Inf Process Syst"},{"key":"7079_CR34","unstructured":"Tan W, Zhang W, Liu S, et al (2023) True knowledge comes from practice: aligning large language models with embodied environments via reinforcement learning. In: The twelfth international conference on learning representations"},{"key":"7079_CR35","doi-asserted-by":"publisher","unstructured":"Tunstall L, Beeching E, Lambert N, et al (2023) Zephyr: direct distillation of LM alignment. https:\/\/doi.org\/10.48550\/arXiv.2310.16944","DOI":"10.48550\/arXiv.2310.16944"},{"key":"7079_CR36","unstructured":"Wang C, Jiang Y, Yang C, et al (2023a) Beyond reverse KL: generalizing direct preference optimization with diverse divergence constraints. In: The twelfth international conference on learning representations. https:\/\/openreview.net\/forum?id=2cRzmWXK9N"},{"key":"7079_CR37","doi-asserted-by":"publisher","unstructured":"Wang T, Li S, Lu W (2024) Self-training with direct preference optimization improves chain-of-thought reasoning. In: Ku LW, Martins A, Srikumar V (eds) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). Association for Computational Linguistics, Bangkok, Thailand, pp 11917\u201311928 https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.643","DOI":"10.18653\/v1\/2024.acl-long.643"},{"key":"7079_CR38","doi-asserted-by":"publisher","unstructured":"Wang Y, Zhong W, Li L, et al (2023b) Aligning large language models with human: a survey. https:\/\/doi.org\/10.48550\/arXiv.2307.12966","DOI":"10.48550\/arXiv.2307.12966"},{"issue":"5","key":"7079_CR39","doi-asserted-by":"publisher","first-page":"1122","DOI":"10.1109\/JAS.2023.123618","volume":"10","author":"T Wu","year":"2023","unstructured":"Wu T, He S, Liu J et al (2023) A Brief overview of ChatGPT: the history, status quo and potential future development. IEEE\/CAA Journal Of Automatica Sinica 10(5):1122\u20131136. https:\/\/doi.org\/10.1109\/JAS.2023.123618","journal-title":"IEEE\/CAA Journal Of Automatica Sinica"},{"key":"7079_CR40","doi-asserted-by":"publisher","unstructured":"Yang A, Li A, Yang B, et al (2025) Qwen3 Technical Report. https:\/\/doi.org\/10.48550\/arXiv.2505.09388","DOI":"10.48550\/arXiv.2505.09388"},{"key":"7079_CR41","first-page":"55734","volume":"36","author":"Y Yu","year":"2023","unstructured":"Yu Y, Zhuang Y, Zhang J et al (2023) Large language model as attributed training data generator: a tale of diversity and bias. Adv Neural Inf Process Syst 36:55734\u201355784","journal-title":"Adv Neural Inf Process Syst"},{"key":"7079_CR42","doi-asserted-by":"crossref","unstructured":"Yuan H, Yuan Z, Tan C, et al (2023) RRHF: Rank responses to align language models with human feedback. Adv Neural Inf Process Syst 36:10935\u201310950. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/hash\/23e6f78bdec844a9f7b6c957de2aae91-Abstract-Conference.html","DOI":"10.52202\/075280-0482"},{"key":"7079_CR43","unstructured":"Yuan W, Pang RY, Cho K, et al (2024) Self-rewarding language models. In: Forty-first international conference on machine learning"},{"key":"7079_CR44","doi-asserted-by":"crossref","unstructured":"Zelikman E, Wu Y, Mu J, et al (2022) STaR: Bootstrapping Reasoning With Reasoning. Advances in Neural Information Processing Systems 35:15476\u201315488. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/hash\/639a9a172c044fbb64175b5fad42e9a5-Abstract-Conference.html","DOI":"10.52202\/068431-1126"},{"key":"7079_CR45","unstructured":"Zeng Y, Liu G, Ma W, et al (2024) Token-level Direct Preference Optimization. In: Forty-first international conference on machine learning. https:\/\/openreview.net\/forum?id=1RZKuvqYCR"},{"key":"7079_CR46","doi-asserted-by":"crossref","unstructured":"Zhang D, Zhoubian S, Hu Z, et al (2024a) ReST-MCTS*: LLM Self-Training via Process Reward Guided Tree Search. Advances in Neural Information Processing Systems 37:64735\u201364772. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2024\/hash\/76ec4dc30e9faaf0e4b6093eaa377218-Abstract-Conference.html","DOI":"10.52202\/079017-2066"},{"key":"7079_CR47","doi-asserted-by":"publisher","unstructured":"Zhang X, Peng B, Tian Y, et al (2024b) Self-Alignment for Factuality: Mitigating Hallucinations in LLMs via Self-Evaluation. In: Ku LW, Martins A, Srikumar V (eds) Proceedings of the 62nd annual meeting of the association for computational linguistics (Volume 1: Long Papers). Association for Computational Linguistics, Bangkok, Thailand, pp 1946\u20131965 https:\/\/doi.org\/10.18653\/v1\/2024.acl-long.107","DOI":"10.18653\/v1\/2024.acl-long.107"},{"key":"7079_CR48","doi-asserted-by":"crossref","unstructured":"Zheng L, Chiang WL, Sheng Y, et al (2024) Judging LLM-as-a-judge with MT-bench and Chatbot Arena. In: Proceedings of the 37th international conference on neural information processing systems. Curran Associates Inc., Red Hook, NY, USA, NIPS \u201923, pp 46595\u201346623","DOI":"10.52202\/075280-2020"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-07079-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-07079-9","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-07079-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T10:15:59Z","timestamp":1774865759000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-07079-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1]]},"references-count":48,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,1]]}},"alternative-id":["7079"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-07079-9","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1]]},"assertion":[{"value":"4 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"64"}}