{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,20]],"date-time":"2026-03-20T15:40:28Z","timestamp":1774021228821,"version":"3.50.1"},"publisher-location":"Cham","reference-count":64,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031746260","type":"print"},{"value":"9783031746277","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-74627-7_21","type":"book-chapter","created":{"date-parts":[[2024,12,31]],"date-time":"2024-12-31T14:00:23Z","timestamp":1735653623000},"page":"276-294","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["On the\u00a0Challenges and\u00a0Practices of\u00a0Reinforcement Learning from\u00a0Real Human Feedback"],"prefix":"10.1007","author":[{"given":"Timo","family":"Kaufmann","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sarah","family":"Ball","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jacob","family":"Beck","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eyke","family":"H\u00fcllermeier","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Frauke","family":"Kreuter","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,1,1]]},"reference":[{"key":"21_CR1","doi-asserted-by":"publisher","unstructured":"Aizpurua, E., Heiden, E.O., Park, K.H., Wittrock, J., Losch, M.E.: Investigating Respondent Multitasking and Distraction Using Self-reports and Interviewers\u2019 Observations in a Dual-frame Telephone Survey. Survey Methods: Insights from the Field (SMIF), November 2018. https:\/\/doi.org\/10.13094\/SMIF-2018-00006","DOI":"10.13094\/SMIF-2018-00006"},{"key":"21_CR2","unstructured":"Amodei, D., Christiano, P., Ray, A.: Learning from human preferences, June 2017. https:\/\/openai.com\/research\/learning-from-human-preferences. Accessed 25 May 2023"},{"issue":"2","key":"21_CR3","doi-asserted-by":"publisher","first-page":"216","DOI":"10.1093\/jssam\/smv003","volume":"3","author":"S Ansolabehere","year":"2015","unstructured":"Ansolabehere, S., Schaffner, B.F.: Distractions: the incidence and consequences of interruptions for survey respondents. J. Surv. Stat. Methodol. 3(2), 216\u2013239 (2015). https:\/\/doi.org\/10.1093\/jssam\/smv003","journal-title":"J. Surv. Stat. Methodol."},{"key":"21_CR4","doi-asserted-by":"publisher","unstructured":"Argall, B., Browning, B., Veloso, M.: Learning by demonstration with critique from a human teacher. In: Proceedings of the ACM\/IEEE International Conference on Human-robot Interaction, pp. 57\u201364. Association for Computing Machinery, March 2007. https:\/\/doi.org\/10.1145\/1228716.1228725","DOI":"10.1145\/1228716.1228725"},{"key":"21_CR5","doi-asserted-by":"publisher","unstructured":"Arzate\u00a0Cruz, C., Igarashi, T.: A survey on interactive reinforcement learning: design principles and open challenges. In: Proceedings of the 2020 ACM Designing Interactive Systems Conference, pp. 1195\u20131209. Association for Computing Machinery, July 2020. https:\/\/doi.org\/10.1145\/3357236.3395525","DOI":"10.1145\/3357236.3395525"},{"key":"21_CR6","unstructured":"Barnett, P., Freedman, R., Svegliato, J., Russell, S.: Active reward learning from multiple teachers. In: The AAAI Workshop on Artificial Intelligence Safety, February 2023"},{"key":"21_CR7","doi-asserted-by":"publisher","unstructured":"Basu, C., Singhal, M., Dragan, A.D.: Learning from richer human guidance: augmenting comparison-based learning with feature queries. In: Proceedings of the 2018 ACM\/IEEE International Conference on Human-Robot Interaction, pp. 132\u2013140. Association for Computing Machinery, February 2018. https:\/\/doi.org\/10.1145\/3171221.3171284","DOI":"10.1145\/3171221.3171284"},{"key":"21_CR8","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1007\/978-3-031-21707-4_19","volume-title":"HCII 2022","author":"J Beck","year":"2022","unstructured":"Beck, J., Eckman, S., Chew, R., Kreuter, F.: Improving labeling through social science insights: results and research agenda. In: Chen, J.Y.C., Fragomeni, G., Degen, H., Ntoa, S. (eds.) HCII 2022. LNCS, vol. 13518, pp. 245\u2013261. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-21707-4_19"},{"key":"21_CR9","doi-asserted-by":"publisher","DOI":"10.1007\/s00521-021-06850-6","author":"A Bignold","year":"2022","unstructured":"Bignold, A., Cruz, F., Dazeley, R., Vamplew, P., Foale, C.: Human engagement providing evaluative and informative advice for interactive reinforcement learning. Neural Comput. Appl. (2022). https:\/\/doi.org\/10.1007\/s00521-021-06850-6","journal-title":"Neural Comput. Appl."},{"key":"21_CR10","unstructured":"B\u0131y\u0131k, E., Palan, M., Landolfi, N.C., Losey, D.P., Sadigh, D.: Asking easy questions: a user-friendly approach to active reward learning. In: Proceedings of the Conference on Robot Learning, pp. 1177\u20131190. PMLR, May 2020. https:\/\/proceedings.mlr.press\/v100\/b-iy-ik20a.html"},{"key":"21_CR11","doi-asserted-by":"publisher","unstructured":"Bless, H., Schwarz, N.: Chapter 6 - mental construal and the emergence of assimilation and contrast effects: the inclusion\/exclusion model. In: Advances in Experimental Social Psychology, vol.\u00a042, pp. 319\u2013373. Academic Press, January 2010. https:\/\/doi.org\/10.1016\/S0065-2601(10)42006-7","DOI":"10.1016\/S0065-2601(10)42006-7"},{"issue":"2","key":"21_CR12","doi-asserted-by":"publisher","first-page":"106","DOI":"10.1177\/1077727X04269573","volume":"33","author":"SC Broussard","year":"2004","unstructured":"Broussard, S.C., Garrison, M.E.B.: The relationship between classroom motivation and academic achievement in elementary-school-aged children. Fam. Consum. Sci. Res. J. 33(2), 106\u2013120 (2004). https:\/\/doi.org\/10.1177\/1077727X04269573","journal-title":"Fam. Consum. Sci. Res. J."},{"issue":"1","key":"21_CR13","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1177\/1745691610393980","volume":"6","author":"M Buhrmester","year":"2011","unstructured":"Buhrmester, M., Kwang, T., Gosling, S.D.: Amazon\u2019s mechanical turk: a new source of inexpensive, yet high-quality, data? Perspect. Psychol. Sci. 6(1), 3\u20135 (2011). https:\/\/doi.org\/10.1177\/1745691610393980","journal-title":"Perspect. Psychol. Sci."},{"issue":"4","key":"21_CR14","doi-asserted-by":"publisher","first-page":"980","DOI":"10.1037\/a0035661","volume":"140","author":"CP Cerasoli","year":"2014","unstructured":"Cerasoli, C.P., Nicklin, J.M., Ford, M.T.: Intrinsic motivation and extrinsic incentives jointly predict performance: a 40-year meta-analysis. Psychol. Bull. 140(4), 980\u20131008 (2014). https:\/\/doi.org\/10.1037\/a0035661","journal-title":"Psychol. Bull."},{"key":"21_CR15","unstructured":"Christiano, P.F., Leike, J., Brown, T., Martic, M., Legg, S., Amodei, D.: Deep reinforcement learning from human preferences. In: Advances in Neural Information Processing Systems, vol.\u00a030. Curran Associates, Inc. (2017). https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/d5e2c0adad503c91f91df240d0cd4e49-Abstract.html"},{"key":"21_CR16","doi-asserted-by":"publisher","first-page":"87","DOI":"10.1080\/00224545.1977.9923987","volume":"101","author":"WD Crano","year":"1977","unstructured":"Crano, W.D.: Primacy versus recency in retention of information and opinion change. J. Soc. Psychol. 101, 87\u201396 (1977). https:\/\/doi.org\/10.1080\/00224545.1977.9923987","journal-title":"J. Soc. Psychol."},{"key":"21_CR17","unstructured":"Cui, Y., Zhang, Q., Knox, B., Allievi, A., Stone, P., Niekum, S.: The EMPATHIC framework for task learning from implicit human feedback. In: Proceedings of the 2020 Conference on Robot Learning, pp. 604\u2013626. PMLR, October 2021. https:\/\/proceedings.mlr.press\/v155\/cui21a.html"},{"key":"21_CR18","unstructured":"Early, J., Bewley, T., Evers, C., Ramchurn, S.: Non-Markovian reward modelling from trajectory labels via interpretable multiple instance learning. Adv. Neural Inf. Process. Syst. 35, 27652\u201327663 (2022). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/hash\/b157cfde6794e93b2353b9712bbd45a5-Abstract-Conference.html"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Fort, K., Ehrmann, M., Nazarenko, A.: Towards a methodology for named entities annotation. In: Proceedings of the Third Linguistic Annotation Workshop, pp. 142\u2013145. Association for Computational Linguistics, August 2009. ISBN 978-1-932432-52-7","DOI":"10.3115\/1698381.1698406"},{"key":"21_CR20","doi-asserted-by":"publisher","unstructured":"F\u00fcrnkranz, J., H\u00fcllermeier, E.: Preference learning and ranking by pairwise comparison. In: F\u00fcrnkranz, J., H\u00fcllermeier, E. (eds.) Preference Learning, pp. 65\u201382. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-14125-6_4","DOI":"10.1007\/978-3-642-14125-6_4"},{"key":"21_CR21","unstructured":"Groves, R.M., Fowler\u00a0Jr, F.J., Couper, M.P., Lepkowski, J.M., Singer, E., Tourangeau, R.: Survey Methodology, 2 edn. Wiley, Hoboken (2009). ISBN 978-0-470-46546-2"},{"key":"21_CR22","unstructured":"Guan, L., Verma, M., Guo, S., Zhang, R., Kambhampati, S.: Widening the pipeline in human-guided reinforcement learning with explanation and context-aware data augmentation. In: Advances in Neural Information Processing Systems, October 2021. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/b6f8dc086b2d60c5856e4ff517060392-Abstract.html"},{"issue":"3","key":"21_CR23","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1007\/s10940-005-4275-4","volume":"21","author":"TC Hart","year":"2005","unstructured":"Hart, T.C., Rennison, C.M., Gibson, C.: Revisiting respondent \u201cfatigue bias\u2019\u2019 in the national crime victimization survey. J. Quant. Criminol. 21(3), 345\u2013363 (2005). https:\/\/doi.org\/10.1007\/s10940-005-4275-4","journal-title":"J. Quant. Criminol."},{"key":"21_CR24","unstructured":"Hejna, D.J., Sadigh, D.: Few-shot preference learning for human-in-the-loop RL. In: Proceedings of the 6th Conference on Robot Learning, pp. 2014\u20132025. PMLR, March 2023. https:\/\/proceedings.mlr.press\/v205\/iii23a.html. ISSN 2640-3498"},{"issue":"4","key":"21_CR25","doi-asserted-by":"publisher","first-page":"549","DOI":"10.1086\/268687","volume":"45","author":"AR Herzog","year":"1981","unstructured":"Herzog, A.R., Bachman, J.G.: Effects of questionnaire length on response quality. Public Opin. Q. 45(4), 549\u2013559 (1981). https:\/\/doi.org\/10.1086\/268687","journal-title":"Public Opin. Q."},{"key":"21_CR26","unstructured":"Holladay, R., Javdani, S., Dragan, A., Srinivasa, S.: Active comparison based learning incorporating user uncertainty and noise. In: RSS Workshop on Model Learning for Human-Robot Communication (2016)"},{"key":"21_CR27","unstructured":"Ibarz, B., Leike, J., Pohlen, T., Irving, G., Legg, S., Amodei, D.: Reward learning from human preferences and demonstrations in Atari. In: Advances in Neural Information Processing Systems, vol.\u00a031. Curran Associates, Inc. (2018). https:\/\/proceedings.neurips.cc\/paper\/2018\/hash\/8cbe9ce23f42628c98f80fa0fac8b19a-Abstract.html"},{"key":"21_CR28","unstructured":"Interactive Agents\u00a0Team, D., et al.: Improving Multimodal Interactive Agents with Reinforcement Learning from Human Feedback, November 2022. http:\/\/arxiv.org\/abs\/2211.11602"},{"key":"21_CR29","unstructured":"Jain, A., Wojcik, B., Joachims, T., Saxena, A.: Learning trajectory preferences for manipulators via iterative improvement. In: Advances in Neural Information Processing Systems, vol.\u00a026. Curran Associates, Inc. (2013). https:\/\/proceedings.neurips.cc\/paper\/2013\/hash\/c058f544c737782deacefa532d9add4c-Abstract.html"},{"key":"21_CR30","unstructured":"Jeon, H.J., Milli, S., Dragan, A.: Reward-rational (implicit) choice: a unifying formalism for reward learning. In: Advances in Neural Information Processing Systems, vol.\u00a033, pp. 4415\u20134426. Curran Associates, Inc. (2020). https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/2f10c1578a0706e06b6d7db6f0b4a6af-Abstract.html"},{"key":"21_CR31","doi-asserted-by":"publisher","DOI":"10.1016\/j.jdeveco.2022.102992","volume":"161","author":"D Jeong","year":"2023","unstructured":"Jeong, D., Aggarwal, S., Robinson, J., Kumar, N., Spearot, A., Park, D.S.: Exhaustive or exhausting? Evidence on respondent fatigue in long surveys. J. Dev. Econ. 161, 102992 (2023). https:\/\/doi.org\/10.1016\/j.jdeveco.2022.102992","journal-title":"J. Dev. Econ."},{"key":"21_CR32","doi-asserted-by":"publisher","unstructured":"Judah, K., Roy, S., Fern, A., Dietterich, T.: Reinforcement learning via practice and critique advice. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a024, pp. 481\u2013486, July 2010. https:\/\/doi.org\/10.1609\/aaai.v24i1.7690","DOI":"10.1609\/aaai.v24i1.7690"},{"key":"21_CR33","doi-asserted-by":"publisher","unstructured":"Koyama, Y., Sato, I., Sakamoto, D., Igarashi, T.: Sequential line search for efficient visual design optimization by crowds. ACM Trans. Graph. 36(4), 48:1\u201348:11 (2017). https:\/\/doi.org\/10.1145\/3072959.3073598","DOI":"10.1145\/3072959.3073598"},{"issue":"2","key":"21_CR34","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1086\/269029","volume":"51","author":"JA Krosnick","year":"1987","unstructured":"Krosnick, J.A., Alwin, D.F.: An evaluation of a cognitive theory of response-order effects in survey measurement. Public Opin. Q. 51(2), 201\u2013219 (1987). https:\/\/doi.org\/10.1086\/269029","journal-title":"Public Opin. Q."},{"key":"21_CR35","doi-asserted-by":"publisher","first-page":"244","DOI":"10.1016\/j.joep.2017.05.004","volume":"61","author":"B Kuvaas","year":"2017","unstructured":"Kuvaas, B., Buch, R., Weibel, A., Dysvik, A., Nerstad, C.G.L.: Do intrinsic and extrinsic motivation relate differently to employee outcomes? J. Econ. Psychol. 61, 244\u2013258 (2017). https:\/\/doi.org\/10.1016\/j.joep.2017.05.004","journal-title":"J. Econ. Psychol."},{"key":"21_CR36","unstructured":"Lawler, E.E.: Motivation in Work Organizations. Brooks\/Cole Publishing Co (1973). ISBN 0-8185-0088-3"},{"key":"21_CR37","doi-asserted-by":"publisher","unstructured":"Li, K., et al.: ROIAL: region of interest active learning for characterizing exoskeleton gait preference landscapes. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 3212\u20133218, May 2021. https:\/\/doi.org\/10.1109\/ICRA48506.2021.9560840","DOI":"10.1109\/ICRA48506.2021.9560840"},{"key":"21_CR38","doi-asserted-by":"publisher","unstructured":"Li, Z., Shi, L., Cristea, A.I., Zhou, Y.: A survey of collaborative reinforcement learning: interactive methods and design patterns. In: Designing Interactive Systems Conference 2021, pp. 1579\u20131590. Association for Computing Machinery, June 2021. https:\/\/doi.org\/10.1145\/3461778.3462135","DOI":"10.1145\/3461778.3462135"},{"issue":"2","key":"21_CR39","doi-asserted-by":"publisher","first-page":"519","DOI":"10.3758\/s13428-014-0483-x","volume":"47","author":"L Litman","year":"2015","unstructured":"Litman, L., Robinson, J., Rosenzweig, C.: The relationship between motivation, monetary compensation, and data quality among US- and India-based workers on Mechanical Turk. Behav. Res. Methods 47(2), 519\u2013528 (2015). https:\/\/doi.org\/10.3758\/s13428-014-0483-x","journal-title":"Behav. Res. Methods"},{"key":"21_CR40","doi-asserted-by":"publisher","unstructured":"Martin, D., Hanrahan, B.V., O\u2019Neill, J., Gupta, N.: Being a turker. In: Proceedings of the 17th ACM Conference on Computer Supported Cooperative Work & Social Computing, pp. 224\u2013235. Association for Computing Machinery, February 2014. https:\/\/doi.org\/10.1145\/2531602.2531663","DOI":"10.1145\/2531602.2531663"},{"key":"21_CR41","unstructured":"Metz, Y., Lindner, D., Baur, R., Keim, D.A., El-Assady, M.: RLHF-blender: a configurable interactive interface for learning from diverse human feedback. In: ICML 2023 Workshop Interactive Learning with Implicit Human Feedback, June 2023. https:\/\/openreview.net\/forum?id=JvkZtzJBFQ"},{"key":"21_CR42","unstructured":"Mitchell, J.V.: Interrelationships and predictive efficacy for indices of intrinsic, extrinsic, and self-assessed motivation for learning. J. Res. Dev. Educ. 25, 149\u2013155 (1992). ISSN 0022-426X"},{"issue":"7540","key":"21_CR43","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015). https:\/\/doi.org\/10.1038\/nature14236","journal-title":"Nature"},{"issue":"3","key":"21_CR44","doi-asserted-by":"publisher","first-page":"134","DOI":"10.4314\/ijah.v2i3","volume":"2","author":"US Muogbo","year":"2013","unstructured":"Muogbo, U.S.: The influence of motivation on employees\u2019 performance: a study of some selected firms in Anambra State. AFRREV IJAH Int. J. Arts Humanit. 2(3), 134\u2013151 (2013). https:\/\/doi.org\/10.4314\/ijah.v2i3","journal-title":"AFRREV IJAH Int. J. Arts Humanit."},{"issue":"2","key":"21_CR45","doi-asserted-by":"publisher","first-page":"522","DOI":"10.1111\/j.1083-6101.2006.00025.x","volume":"11","author":"J Murphy","year":"2006","unstructured":"Murphy, J., Hofacker, C., Mizerski, R.: Primacy and recency effects on clicking behavior. J. Comput.-Mediat. Commun. 11(2), 522\u2013535 (2006). https:\/\/doi.org\/10.1111\/j.1083-6101.2006.00025.x","journal-title":"J. Comput.-Mediat. Commun."},{"key":"21_CR46","unstructured":"Myers, V., B\u0131y\u0131k, E., Anari, N., Sadigh, D.: Learning multimodal rewards from rankings. In: Proceedings of the 5th Conference on Robot Learning, pp. 342\u2013352. PMLR, January 2022. https:\/\/proceedings.mlr.press\/v164\/myers22a.html"},{"key":"21_CR47","unstructured":"N\u00e9dellec, C., Bessieres, P., Bossy, R.R., Kotoujansky, A., Manine, A.P.: Annotation guidelines for machine learning-based named entity recognition in microbiology. In: Proceeding of Data and Text Mining for Integrative Biology Workshop 17. European Conference on Machine Learning 10. European Conference on Principles and Practice of Knowledge Discovery in Databases. Springer (2006)"},{"key":"21_CR48","unstructured":"OpenAI: ChatGPT: Optimizing Language Models for Dialogue (2022). https:\/\/openai.com\/blog\/chatgpt\/. Accessed 02 Feb 2023"},{"key":"21_CR49","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback. In: Advances in Neural Information Processing Systems, vol.\u00a035, pp. 27730\u201327744, December 2022. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/hash\/b1efde53be364a73914f58805a001731-Abstract-Conference.html"},{"key":"21_CR50","unstructured":"Porter, L.W., Lawler, E.E.: Managerial Attitudes and Performance. R.D. Irwin, Homewood (1968)"},{"key":"21_CR51","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., Moritz, P.: Trust region policy optimization. In: Proceedings of the 32nd International Conference on Machine Learning, pp. 1889\u20131897. PMLR, June 2015. ISSN 1938-7228. https:\/\/proceedings.mlr.press\/v37\/schulman15.html"},{"key":"21_CR52","doi-asserted-by":"publisher","unstructured":"Sendelbah, A., Vehovar, V., Slavec, A., Petrov\u010di\u010d, A.: Investigating respondent multitasking in web surveys using paradata. Comput. Hum. Behav. 55, 777\u2013787 (2016). https:\/\/doi.org\/10.1016\/j.chb.2015.10.028","DOI":"10.1016\/j.chb.2015.10.028"},{"key":"21_CR53","doi-asserted-by":"publisher","unstructured":"Shaw, A.D., Horton, J.J., Chen, D.L.: Designing incentives for inexpert human raters. In: Proceedings of the ACM 2011 Conference on Computer Supported Cooperative Work, pp. 275\u2013284. Association for Computing Machinery, March 2011. https:\/\/doi.org\/10.1145\/1958824.1958865","DOI":"10.1145\/1958824.1958865"},{"issue":"7587","key":"21_CR54","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., et al.: Mastering the game of Go with deep neural networks and tree search. Nature 529(7587), 484\u2013489 (2016). https:\/\/doi.org\/10.1038\/nature16961","journal-title":"Nature"},{"key":"21_CR55","unstructured":"Stiennon, N., et al.: Learning to summarize from human feedback, February 2022. http:\/\/arxiv.org\/abs\/2009.01325"},{"key":"21_CR56","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction, 2nd edn. The MIT Press, Cambridge (2018). ISBN 978-0-262-03924-6"},{"issue":"7782","key":"21_CR57","doi-asserted-by":"publisher","first-page":"350","DOI":"10.1038\/s41586-019-1724-z","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals, O., et al.: Grandmaster level in StarCraft II using multi-agent reinforcement learning. Nature 575(7782), 350\u2013354 (2019). https:\/\/doi.org\/10.1038\/s41586-019-1724-z","journal-title":"Nature"},{"issue":"1","key":"21_CR58","doi-asserted-by":"publisher","first-page":"148","DOI":"10.1177\/0894439319851503","volume":"39","author":"A Wenz","year":"2021","unstructured":"Wenz, A.: Do distractions during web survey completion affect data quality? Findings from a laboratory experiment. Soc. Sci. Comput. Rev. 39(1), 148\u2013161 (2021). https:\/\/doi.org\/10.1177\/0894439319851503","journal-title":"Soc. Sci. Comput. Rev."},{"key":"21_CR59","unstructured":"Wilde, N., B\u0131y\u0131k, E., Sadigh, D., Smith, S.L.: Learning reward functions from scale feedback. In: Proceedings of the 5th Conference on Robot Learning, pp. 353\u2013362, PMLR, January 2022. https:\/\/proceedings.mlr.press\/v164\/wilde22a.html"},{"key":"21_CR60","doi-asserted-by":"publisher","unstructured":"Yin, M., Chen, Y., Sun, Y.A.: The effects of performance-contingent financial incentives in online labor markets. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a027, pp. 1191\u20131197, June 2013. https:\/\/doi.org\/10.1609\/aaai.v27i1.8461","DOI":"10.1609\/aaai.v27i1.8461"},{"key":"21_CR61","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/978-3-030-46133-1_1","volume-title":"Machine Learning and Knowledge Discovery in Databases","author":"A Zap","year":"2020","unstructured":"Zap, A., Joppen, T., F\u00fcrnkranz, J.: Deep ordinal reinforcement learning. In: Brefeld, U., Fromont, E., Hotho, A., Knobbe, A., Maathuis, M., Robardet, C. (eds.) ECML PKDD 2019. LNCS (LNAI), vol. 11908, pp. 3\u201318. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-46133-1_1"},{"key":"21_CR62","unstructured":"Zhang, D., Carroll, M., Bobu, A., Dragan, A.: Time-efficient reward learning via visually assisted cluster ranking. In: NeurIPS Workshop on Human in the Loop Learning, December 2022"},{"key":"21_CR63","unstructured":"Ziegler, D.M., et al.: Fine-Tuning Language Models from Human Preferences, January 2020. http:\/\/arxiv.org\/abs\/1909.08593"},{"key":"21_CR64","doi-asserted-by":"publisher","unstructured":"Zwarun, L., Hall, A.: What\u2019s going on? Age, distraction, and multitasking during online survey taking. Comput. Hum. Behav. 41, 236\u2013244 (2014). https:\/\/doi.org\/10.1016\/j.chb.2014.09.041","DOI":"10.1016\/j.chb.2014.09.041"}],"container-title":["Communications in Computer and Information Science","Machine Learning and Principles and Practice of Knowledge Discovery in Databases"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-74627-7_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,31]],"date-time":"2024-12-31T14:14:43Z","timestamp":1735654483000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-74627-7_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031746260","9783031746277"],"references-count":64,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-74627-7_21","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"1 January 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECML PKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Turin","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 September 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 September 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecml2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2023.ecmlpkdd.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}