{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T15:05:57Z","timestamp":1782313557216,"version":"3.54.5"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032060952","type":"print"},{"value":"9783032060969","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,9,27]],"date-time":"2025-09-27T00:00:00Z","timestamp":1758931200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,27]],"date-time":"2025-09-27T00:00:00Z","timestamp":1758931200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-06096-9_6","type":"book-chapter","created":{"date-parts":[[2025,9,26]],"date-time":"2025-09-26T09:54:34Z","timestamp":1758880474000},"page":"96-112","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Active Preference Optimization for\u00a0Sample Efficient RLHF"],"prefix":"10.1007","author":[{"given":"Nirjhar","family":"Das","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Souradip","family":"Chakraborty","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aldo","family":"Pacchiano","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sayak Ray","family":"Chowdhury","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,27]]},"reference":[{"key":"6_CR1","unstructured":"Abeille, M., Faury, L., Calauz\u00e8nes, C.: Instance-wise minimax-optimal algorithms for logistic bandits. In: International Conference on Artificial Intelligence and Statistics, pp. 3691\u20133699. PMLR (2021)"},{"key":"6_CR2","unstructured":"Bai, Y., et\u00a0al.: Training a helpful and harmless assistant with reinforcement learning from human feedback (2022)"},{"key":"6_CR3","doi-asserted-by":"crossref","unstructured":"Bradley, R.A., Terry, M.E.: Rank analysis of incomplete block designs: I. The method of paired comparisons. Biometrika 39(3\/4), 324\u2013345 (1952)","DOI":"10.1093\/biomet\/39.3-4.324"},{"key":"6_CR4","first-page":"118052","volume":"37","author":"L Carvalho Melo","year":"2024","unstructured":"Carvalho Melo, L., Tigas, P., Abate, A., Gal, Y.: Deep Bayesian active learning for preference modeling in large language models. Adv. Neural. Inf. Process. Syst. 37, 118052\u2013118085 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"6_CR5","unstructured":"Chen, X., Zhong, H., Yang, Z., Wang, Z., Wang, L.: Human-in-the-loop: provably efficient preference-based reinforcement learning with general function approximation. In: International Conference on Machine Learning, pp. 3773\u20133793 (2022)"},{"key":"6_CR6","unstructured":"Christiano, P.F., Leike, J., Brown, T., Martic, M., Legg, S., Amodei, D.: Deep reinforcement learning from human preferences. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"issue":"39","key":"6_CR7","first-page":"1079","volume":"7","author":"E Even-Dar","year":"2006","unstructured":"Even-Dar, E., Mannor, S., Mansour, Y.: Action elimination and stopping conditions for the multi-armed bandit and reinforcement learning problems. J. Mach. Learn. Res. 7(39), 1079\u20131105 (2006)","journal-title":"J. Mach. Learn. Res."},{"key":"6_CR8","unstructured":"Glaese, A., et\u00a0al.: Improving alignment of dialogue agents via targeted human judgements. arXiv preprint arXiv:2209.14375 (2022)"},{"key":"6_CR9","doi-asserted-by":"crossref","unstructured":"Hazan, E., et\u00a0al.: Introduction to online convex optimization. Found. Trends\u00ae Optim. 2(3-4), 157\u2013325 (2016)","DOI":"10.1561\/2400000013"},{"key":"6_CR10","unstructured":"Ji, K., He, J., Gu, Q.: Reinforcement learning from human feedback with active queries. arXiv preprint arXiv:2402.09401 (2024)"},{"key":"6_CR11","unstructured":"Kingma, D., Ba, J.: Adam: a method for stochastic optimization. In: International Conference on Learning Representations (ICLR). San Diego, CA, USA (2015)"},{"key":"6_CR12","doi-asserted-by":"crossref","unstructured":"Lattimore, T., Szepesv\u00e1ri, C.: Bandit Algorithms. Cambridge University Press (2020)","DOI":"10.1017\/9781108571401"},{"key":"6_CR13","unstructured":"Lee, J., Yun, S.Y., Jun, K.S.: Improved regret bounds of (multinomial) logistic bandits via regret-to-confidence-set conversion. In: International Conference on Artificial Intelligence and Statistics, pp. 4474\u20134482. PMLR (2024)"},{"key":"6_CR14","unstructured":"Lee, K., Smith, L.M., Abbeel, P.: Pebble: Feedback-efficient interactive reinforcement learning via relabeling experience and unsupervised pre-training. In: Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 6152\u20136163. PMLR (2021)"},{"key":"6_CR15","unstructured":"Luce, R.D.: Individual Choice Behavior. Wiley (1959)"},{"key":"6_CR16","unstructured":"Maas, A.L., Daly, R.E., Pham, P.T., Huang, D., Ng, A.Y., Potts, C.: Learning word vectors for sentiment analysis. In: Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: human Language Technologies, pp. 142\u2013150. Association for Computational Linguistics (2011)"},{"key":"6_CR17","unstructured":"Maiti, A., Boczar, R., Jamieson, K., Ratliff, L.: Near-optimal pure exploration in matrix games: a generalization of stochastic bandits and dueling bandits. In: Proceedings of The 27th International Conference on Artificial Intelligence and Statistics, pp. 2602\u20132610 (2024)"},{"key":"6_CR18","unstructured":"Mehta, V., et al.: Sample efficient reinforcement learning from human feedback via active exploration. arXiv preprint arXiv:2312.00267 (2023)"},{"key":"6_CR19","unstructured":"Muldrew, W., Hayes, P., Zhang, M., Barber, D.: Active preference learning for large language models. In: Proceedings of the 41st International Conference on Machine Learning, pp. 36577\u201336590. PMLR (2024)"},{"key":"6_CR20","unstructured":"Ouyang, L., et\u00a0al.: Training language models to follow instructions with human feedback. In: Advances in Neural Information Processing Systems (2022)"},{"key":"6_CR21","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I.: Language models are unsupervised multitask learners (2019)"},{"key":"6_CR22","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Manning, C.D., Ermon, S., Finn, C.: Direct preference optimization: your language model is secretly a reward model. In: Thirty-seventh Conference on Neural Information Processing Systems (2023)"},{"key":"6_CR23","unstructured":"Ray\u00a0Chowdhury, S., Kini, A., Natarajan, N.: Provably robust DPO: aligning language models with noisy feedback. In: Proceedings of the 41st International Conference on Machine Learning, pp. 42258\u201342274. PMLR (2024)"},{"key":"6_CR24","doi-asserted-by":"crossref","unstructured":"Sadigh, D., Dragan, A.D., Sastry, S.S., Seshia, S.A.: Active preference-based learning of reward functions. In: Robotics: Science and Systems (2017)","DOI":"10.15607\/RSS.2017.XIII.053"},{"key":"6_CR25","first-page":"30050","volume":"34","author":"A Saha","year":"2021","unstructured":"Saha, A.: Optimal algorithms for stochastic contextual preference bandits. Adv. Neural. Inf. Process. Syst. 34, 30050\u201330062 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"6_CR26","unstructured":"Saha, A., Pacchiano, A., Lee, J.: Dueling RL: reinforcement learning with trajectory preferences. In: Proceedings of The 26th International Conference on Artificial Intelligence and Statistics, pp. 6263\u20136289. PMLR (2023)"},{"key":"6_CR27","unstructured":"Sawarni, A., Das, N., Barman, S., Sinha, G.: Generalized linear bandits with limited adaptivity. In: Advances in Neural Information Processing Systems (2024)"},{"key":"6_CR28","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"6_CR29","first-page":"3008","volume":"33","author":"N Stiennon","year":"2020","unstructured":"Stiennon, N., et al.: Learning to summarize with human feedback. Adv. Neural. Inf. Process. Syst. 33, 3008\u20133021 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"6_CR30","unstructured":"Team, G., et\u00a0al.: Gemma: open models based on Gemini research and technology. arXiv preprint arXiv:2403.08295 (2024)"},{"key":"6_CR31","unstructured":"Zhan, W., Uehara, M., Sun, W., Lee, J.D.: How to query human feedback efficiently in RL? ArXiv abs\/2305.18505 (2023)"},{"key":"6_CR32","unstructured":"Zhu, B., Jordan, M., Jiao, J.: Principled reinforcement learning with human feedback from pairwise or k-wise comparisons. In: Proceedings of the 40th International Conference on Machine Learning, pp. 43037\u201343067. PMLR (2023)"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases. Research Track"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-06096-9_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T10:24:29Z","timestamp":1777631069000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-06096-9_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,27]]},"ISBN":["9783032060952","9783032060969"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-06096-9_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,27]]},"assertion":[{"value":"27 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECML PKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Porto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Portugal","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecml2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecmlpkdd.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}