{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:42:40Z","timestamp":1783194160220,"version":"3.54.6"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2022,1]]},"DOI":"10.1007\/s10994-021-06103-6","type":"journal-article","created":{"date-parts":[[2022,1,4]],"date-time":"2022-01-04T20:02:47Z","timestamp":1641326567000},"page":"173-203","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["SAMBA: safe model-based\u00a0&amp; active reinforcement learning"],"prefix":"10.1007","volume":"111","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2669-9513","authenticated-orcid":false,"given":"Alexander I.","family":"Cowen-Rivers","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daniel","family":"Palenicek","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vincent","family":"Moens","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mohammed Amin","family":"Abdullah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aivar","family":"Sootla","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haitham","family":"Bou-Ammar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,1,4]]},"reference":[{"key":"6103_CR1","unstructured":"Achiam, J., Held, D., Tamar, A., & Abbeel, P. (2017). Constrained policy optimization. In: International conference on machine learning, pp 22\u201331."},{"key":"6103_CR2","doi-asserted-by":"crossref","unstructured":"Akametalu, A.K., Fisac, J.F., Gillula, J.H., Kaynama, S., Zeilinger, M.N., & Tomlin, C.J. (2014). Reachability-based safe learning with gaussian processes. In: IEEE conference on decision and control, pp 1424\u20131431.","DOI":"10.1109\/CDC.2014.7039601"},{"key":"6103_CR3","volume-title":"Constrained markov decision processes","author":"E Altman","year":"1999","unstructured":"Altman, E. (1999). Constrained markov decision processes. CRC Press."},{"key":"6103_CR5","unstructured":"Ammar, H.B., Eaton, E., Ruvolo, P., & Taylor, M. (2014). Online multi-task learning for policy gradient methods. In: International conference on machine learning, pp 1206\u20131214."},{"issue":"5","key":"6103_CR6","doi-asserted-by":"publisher","first-page":"1216","DOI":"10.1016\/j.automatica.2013.02.003","volume":"49","author":"A Aswani","year":"2013","unstructured":"Aswani, A., Gonzalez, H., Sastry, S. S., & Tomlin, C. (2013). Provably safe and robust learning-based model predictive control. Automatica, 49(5), 1216\u20131226.","journal-title":"Automatica"},{"key":"6103_CR7","unstructured":"Ball, P., Parker-Holder, J., Pacchiano, A., Choromanski, K., & Roberts, S. (2020). Ready policy one: World building through active learning. arXiv preprint arXiv:200202693"},{"key":"6103_CR8","doi-asserted-by":"crossref","unstructured":"Berkenkamp, F., Moriconi, R., Schoellig, A.P., & Krause, A. (2016). Safe learning of regions of attraction for uncertain, nonlinear systems with gaussian processes. In: 2016 IEEE 55th conference on decision and control (CDC), IEEE, pp 4661\u20134666.","DOI":"10.1109\/CDC.2016.7798979"},{"key":"6103_CR9","unstructured":"Berkenkamp, F., Turchetta, M., Schoellig, A., & Krause, A. (2017). Safe model-based reinforcement learning with stability guarantees. In: Advances in neural information processing systems, pp 908\u2013918."},{"issue":"3","key":"6103_CR10","doi-asserted-by":"publisher","first-page":"334","DOI":"10.1057\/palgrave.jors.2600425","volume":"48","author":"DP Bertsekas","year":"1997","unstructured":"Bertsekas, D. P. (1997). Nonlinear programming. Journal of the Operational Research Society, 48(3), 334.","journal-title":"Journal of the Operational Research Society"},{"key":"6103_CR11","unstructured":"Brockman, G., Cheung, V., Pettersson, L., Schneider, J., Schulman, J., Tang, J., & Zaremba, W. (2016). Openai gym. arXiv preprint arXiv:160601540"},{"key":"6103_CR12","unstructured":"Buisson-Fenet, M., Solowjow, F., & Trimpe, S. (2019). Actively learning gaussian process dynamics. arXiv preprint arXiv:191109946"},{"key":"6103_CR13","unstructured":"Camacho, E.F., & Alba, C.B. (2013). Model predictive control. Springer Science & Business Media."},{"key":"6103_CR14","unstructured":"Chow, Y., & Ghavamzadeh, M. (2014). Algorithms for cvar optimization in mdps. In: Advances in neural information processing systems, pp 3509\u20133517"},{"key":"6103_CR15","unstructured":"Chow, Y., Tamar, A., Mannor, S., & Pavone, M. (2015). Risk-sensitive and robust decision-making: a cvar optimization approach. In: Advances in neural information processing systems, pp 1522\u20131530."},{"issue":"1","key":"6103_CR16","first-page":"6070","volume":"18","author":"Y Chow","year":"2017","unstructured":"Chow, Y., Ghavamzadeh, M., Janson, L., & Pavone, M. (2017). Risk-constrained reinforcement learning with percentile risk criteria. The Journal of Machine Learning Research, 18(1), 6070\u20136120.","journal-title":"The Journal of Machine Learning Research"},{"key":"6103_CR17","unstructured":"Chow, Y., Nachum, O., Duenez-Guzman, E., & Ghavamzadeh, M. (2018). A lyapunov-based approach to safe reinforcement learning. In: Advances in neural information processing systems, pp 8092\u20138101."},{"key":"6103_CR18","unstructured":"Chow, Y., Nachum, O., Faust, A., Ghavamzadeh, M., & Duenez-Guzman, E. (2019). Lyapunov-based safe policy optimization for continuous control. arXiv preprint arXiv:190110031"},{"key":"6103_CR19","unstructured":"Dalal, G., Dvijotham, K., Vecerik, M., Hester, T., Paduraru, C., & Tassa, Y. (2018). Safe exploration in continuous action spaces. arXiv preprint arXiv:180108757"},{"key":"6103_CR20","unstructured":"Damianou, A., Titsias, M.K., Lawrence, N.D. (2011). Variational gaussian process dynamical systems. In: Advances in neural information processing systems, pp 2510\u20132518."},{"key":"6103_CR60","doi-asserted-by":"crossref","unstructured":"de\u00a0Wolff, T., Cuevas, A., & Tobar, F. (2020). Mogptk: The multi-output gaussian process toolkit. arXiv preprint arXiv:200203471","DOI":"10.1016\/j.neucom.2020.09.085"},{"key":"6103_CR21","unstructured":"Deisenroth, M., & Rasmussen, C.E. (2011). Pilco: A model-based and data-efficient approach to policy search. In: International conference on machine learning, pp 465\u2013472."},{"key":"6103_CR22","volume-title":"Theory of optimal experiments","author":"VV Fedorov","year":"2013","unstructured":"Fedorov, V. V. (2013). Theory of optimal experiments. Elsevier."},{"key":"6103_CR23","unstructured":"Gal, Y., Islam, R., & Ghahramani, Z. (2017). Deep bayesian active learning with image data. arXiv preprint arXiv:170302910"},{"key":"6103_CR24","unstructured":"Gardner, J., Pleiss, G., Weinberger, K.Q., Bindel, D., & Wilson, A.G. (2018). Gpytorch: Blackbox matrix-matrix gaussian process inference with gpu acceleration. In: Advances in neural information processing systems, pp 7576\u20137586"},{"issue":"1","key":"6103_CR25","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1023\/A:1017513905271","volume":"109","author":"C Goh","year":"2001","unstructured":"Goh, C., & Yang, X. (2001). Nonlinear lagrangian theory for nonconvex optimization. Journal of Optimization Theory and Applications, 109(1), 99\u2013121.","journal-title":"Journal of Optimization Theory and Applications"},{"key":"6103_CR26","unstructured":"Ha, D., & Schmidhuber, J. (2018). World models. arXiv preprint arXiv:180310122"},{"key":"6103_CR27","unstructured":"Hafner, D., Lillicrap, T., Fischer, I., Villegas, R., Ha, D., Lee, H., & Davidson, J. (2019). Learning latent dynamics for planning from pixels. In: International conference on machine learning."},{"key":"6103_CR28","unstructured":"Hafner, D., Lillicrap, T., Ba, J., & Norouzi, M. (2020). Dream to control: Learning behaviors by latent imagination. In: International conference on learning representations."},{"key":"6103_CR29","doi-asserted-by":"crossref","unstructured":"Hessel, M., Modayil, J., Van\u00a0Hasselt, H., Schaul, T., Ostrovski, G., Dabney, W., Horgan, D., Piot, B., Azar, M., & Silver, D. (2018). Rainbow: Combining improvements in deep reinforcement learning. In: AAAI conference on artificial intelligence.","DOI":"10.1609\/aaai.v32i1.11796"},{"key":"6103_CR30","unstructured":"Hjmshi. (2018). hjmshi\/pytorch-lbfgs. https:\/\/githubcom\/hjmshi\/PyTorch-LBFGS"},{"key":"6103_CR31","doi-asserted-by":"crossref","unstructured":"Jain, A., Nghiem, T., Morari, M., & Mangharam, R. (2018). Learning and control using gaussian processes. In: ACM\/IEEE international conference on cyber-physical systems, pp 140\u2013149.","DOI":"10.1109\/ICCPS.2018.00022"},{"key":"6103_CR32","unstructured":"Janner, M., Fu, J., Zhang, M., & Levine, S. (2019). When to trust your model: Model-based policy optimization. arXiv preprint arXiv:190608253"},{"key":"6103_CR33","unstructured":"Kamthe, S., & Deisenroth, M.P. (2018). Data-efficient reinforcement learning with probabilistic model predictive control. In: International conference on artificial intelligence and statistics."},{"key":"6103_CR34","volume-title":"Nonlinear systems","author":"HK Khalil","year":"2002","unstructured":"Khalil, H. K., & Grizzle, J. W. (2002). Nonlinear systems (Vol. 3). Prentice hall Upper Saddle River."},{"key":"6103_CR35","doi-asserted-by":"crossref","unstructured":"Koller, T., Berkenkamp, F., Turchetta, M., & Krause, A. (2018). Learning-based model predictive control for safe exploration. In: IEEE conference on decision and control, pp 6059\u20136066.","DOI":"10.1109\/CDC.2018.8619572"},{"key":"6103_CR36","doi-asserted-by":"crossref","unstructured":"Krause, A., & Guestrin, C. (2007). Nonmyopic active learning of gaussian processes: an exploration-exploitation approach. In: International conference on machine learning, pp 449\u2013456.","DOI":"10.1145\/1273496.1273553"},{"issue":"Feb","key":"6103_CR37","first-page":"235","volume":"9","author":"A Krause","year":"2008","unstructured":"Krause, A., Singh, A., & Guestrin, C. (2008). Near-optimal sensor placements in gaussian processes: Theory, efficient algorithms and empirical studies. Journal of Machine Learning Research, 9(Feb), 235\u2013284. 9(Feb):235\u2013284.","journal-title":"Journal of Machine Learning Research"},{"issue":"2","key":"6103_CR38","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1016\/j.automatica.2004.08.019","volume":"41","author":"DQ Mayne","year":"2005","unstructured":"Mayne, D. Q., Seron, M. M., & Rakovi\u0107, S. (2005). Robust model predictive control of constrained linear systems with bounded disturbances. Automatica, 41(2), 219\u2013224.","journal-title":"Automatica"},{"key":"6103_CR39","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Graves, A., Antonoglou, I., Wierstra, D., & Riedmiller, M. (2013). Playing atari with deep reinforcement learning. arXiv preprint arXiv:13125602"},{"issue":"7540","key":"6103_CR40","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A. A., Veness, J., Bellemare, M. G., et al. (2015). Human-level control through deep reinforcement learning. Nature, 518(7540), 529\u2013533.","journal-title":"Nature"},{"key":"6103_CR41","unstructured":"Petersen, K., & Pedersen, M., et\u00a0al. (2008). The matrix cookbook, vol. 7. Technical University of Denmark 15."},{"key":"6103_CR42","unstructured":"Polymenakos, K., Abate, A., & Roberts, S. (2019). Safe policy search using gaussian process models. In: Proceedings of the 18th international conference on autonomous agents and multiagent systems, pp 1565\u20131573."},{"key":"6103_CR43","doi-asserted-by":"crossref","unstructured":"Polymenakos, K., Rontsis, N., Abate, A., & Roberts, S. (2020). SafePILCO: A software tool for safe and data-efficient policy synthesis. In D. N. Jansen & A. Remke (Eds.), Gribaudo M (pp. 18\u201326). Quantitative Evaluation of Systems: Springer International Publishing.","DOI":"10.1007\/978-3-030-59854-9_3"},{"key":"6103_CR44","doi-asserted-by":"crossref","unstructured":"Prashanth, L. (2014). Policy gradients for cvar-constrained mdps. In: International conference on algorithmic learning theory, Springer, pp 155\u2013169.","DOI":"10.1007\/978-3-319-11662-4_12"},{"key":"6103_CR45","unstructured":"Prashanth, L., & Ghavamzadeh, M. (2013). Actor-critic algorithms for risk-sensitive mdps. In: Advances in neural information processing systems, pp 252\u2013260."},{"key":"6103_CR46","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/3206.001.0001","volume-title":"Gaussian processes for machine learning (Adaptive computation and machine learning)","author":"CE Rasmussen","year":"2005","unstructured":"Rasmussen, C. E., & Williams, C. K. I. (2005). Gaussian processes for machine learning (Adaptive computation and machine learning). The MIT Press."},{"key":"6103_CR47","unstructured":"Ray, A., Achiam, J., & Amodei, D. (2019). Benchmarking safe exploration in deep reinforcement learning. https:\/\/cdn.openai.com\/safexp-short.pdf."},{"key":"6103_CR48","doi-asserted-by":"publisher","first-page":"21","DOI":"10.21314\/JOR.2000.038","volume":"2","author":"RT Rockafellar","year":"2000","unstructured":"Rockafellar, R. T., Uryasev, S., et al. (2000). Optimization of conditional value-at-risk. Journal of Risk, 2, 21\u201342.","journal-title":"Journal of Risk"},{"key":"6103_CR49","unstructured":"Saphal, R., Ravindran, B., Mudigere, D., Avancha, S., & Kaul, B. (2020). Seerl: Sample efficient ensemble reinforcement learning. arXiv preprint arXiv:200105209."},{"key":"6103_CR50","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., & Moritz, P. (2015). Trust region policy optimization. In: International conference on machine learning, pp 1889\u20131897."},{"key":"6103_CR51","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms. arXiv preprint arXiv:170706347."},{"key":"6103_CR52","unstructured":"Schultheis, M., Belousov, B., Abdulsamad, H., & Peters, J. (2019). Receding horizon curiosity. In: Conference on robot learning."},{"key":"6103_CR53","unstructured":"Settles, B. (2009). Active learning literature survey. Tech. rep.: University of Wisconsin-Madison Department of Computer Sciences."},{"issue":"1","key":"6103_CR54","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1145\/584091.584093","volume":"5","author":"CE Shannon","year":"2001","unstructured":"Shannon, C. E. (2001). A mathematical theory of communication. ACM SIGMOBILE Mobile Computing and Communications Review, 5(1), 3\u201355.","journal-title":"ACM SIGMOBILE Mobile Computing and Communications Review"},{"key":"6103_CR55","unstructured":"Shyam, P., Ja\u015bkowski, W., & Gomez, F. (2019). Model-based active exploration. In: International conference on machine learning."},{"issue":"7587","key":"6103_CR56","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., Huang, A., Maddison, C. J., Guez, A., Sifre, L., Van Den Driessche, G., et al. (2016). Mastering the game of go with deep neural networks and tree search. Nature, 529(7587), 484\u2013489.","journal-title":"Nature"},{"issue":"7676","key":"6103_CR57","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1038\/nature24270","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver, D., Schrittwieser, J., Simonyan, K., Antonoglou, I., Huang, A., Guez, A., et al. (2017). Mastering the game of go without human knowledge. Nature, 550(7676), 354\u2013359.","journal-title":"Nature"},{"key":"6103_CR58","unstructured":"Srinivas, A., Laskin, M., & Abbeel, P. (2020). Curl: Contrastive unsupervised representations for reinforcement learning. arXiv preprint arXiv:200404136."},{"key":"6103_CR59","unstructured":"Sutton, R.S., & Barto, A.G. (2018). Reinforcement learning: An introduction. MIT press"},{"key":"6103_CR4","unstructured":"van Amersfoort, J., Smith, L., Teh, Y.W., & Gal, Y. (2020). Simple and scalable epistemic uncertainty estimation using a single deep deterministic neural network. arXiv preprint arXiv:200302037"},{"key":"6103_CR61","unstructured":"Zimmer, C., Meister, M., & Nguyen-Tuong, D. (2018). Safe active learning for time-series modeling with gaussian processes. In: Advances in neural information processing systems, pp 2730\u20132739."}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-021-06103-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-021-06103-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-021-06103-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,4]],"date-time":"2023-01-04T01:04:27Z","timestamp":1672794267000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-021-06103-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,1]]},"references-count":61,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2022,1]]}},"alternative-id":["6103"],"URL":"https:\/\/doi.org\/10.1007\/s10994-021-06103-6","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,1]]},"assertion":[{"value":"23 October 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 September 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 October 2021","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 January 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors received a salary from Huawei while the research was conducted.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"While all baselines can be readily found online such as ,  and . We are currently working on releasing the SAMBA agents.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}