{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T20:06:48Z","timestamp":1779307608535,"version":"3.51.4"},"reference-count":26,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,12,14]],"date-time":"2024-12-14T00:00:00Z","timestamp":1734134400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,14]],"date-time":"2024-12-14T00:00:00Z","timestamp":1734134400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Front. Comput. Sci."],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s11704-024-3946-y","type":"journal-article","created":{"date-parts":[[2024,12,14]],"date-time":"2024-12-14T00:33:40Z","timestamp":1734136420000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Offline model-based reinforcement learning with causal structured world models"],"prefix":"10.1007","volume":"19","author":[{"given":"Zhengmao","family":"Zhu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Honglong","family":"Tian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xionghui","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kun","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,14]]},"reference":[{"key":"3946_CR1","volume-title":"BDD100K: a diverse driving video database with scalable annotation tooling","author":"F Yu","year":"2018","unstructured":"Yu F, Xian W, Chen Y, Liu F, Liao M, Madhavan V, Darrell T. BDD100K: a diverse driving video database with scalable annotation tooling. 2018, arXiv preprint arXiv: 1805.04687"},{"issue":"1","key":"3946_CR2","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1038\/s41591-018-0310-5","volume":"25","author":"O Gottesman","year":"2019","unstructured":"Gottesman O, Johansson F, Komorowski M, Faisal A, Sontag D, Doshi-Velez F, Celi L A. Guidelines for reinforcement learning in healthcare. Nature Medicine, 2019, 25(1): 16\u201318","journal-title":"Nature Medicine"},{"key":"3946_CR3","first-page":"1185","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"T Yu","year":"2020","unstructured":"Yu T, Thomas G, Yu L, Ermon S, Zou J, Levine S, Finn C, Ma T. MOPO: model-based offline policy optimization. In: Proceedings of the 34th International Conference on Neural Information Processing Systems. 2020, 1185"},{"key":"3946_CR4","volume-title":"Proceedings of the 8th International Conference on Learning Representations","author":"Y Bengio","year":"2020","unstructured":"Bengio Y, Deleu T, Rahaman N, Ke N R, Lachapelle S, Bilaniuk O, Goyal A, Pal C J. A meta-transfer objective for learning to disentangle causal mechanisms. In: Proceedings of the 8th International Conference on Learning Representations. 2020"},{"key":"3946_CR5","first-page":"1049","volume-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","author":"P de Haan","year":"2019","unstructured":"de Haan P, Jayaraman D, Levine S. Causal confusion in imitation learning. In: Proceedings of the 33rd International Conference on Neural Information Processing Systems. 2019, 1049"},{"key":"3946_CR6","first-page":"5","volume-title":"Proceedings of the 17th International Conference on Autonomous Agents and MultiAgent Systems","author":"J Tenenbaum","year":"2018","unstructured":"Tenenbaum J. Building machines that learn and think like people. In: Proceedings of the 17th International Conference on Autonomous Agents and MultiAgent Systems. 2018, 5"},{"key":"3946_CR7","volume-title":"Proceedings of the 40th Annual Meeting of the Cognitive Science Society","author":"M Edmonds","year":"2018","unstructured":"Edmonds M, Kubricht J, Summers C, Zhu Y, Rothrock B, Zhu S C, Lu H. Human causal transfer: challenges for deep reinforcement learning. In: Proceedings of the 40th Annual Meeting of the Cognitive Science Society. 2018"},{"key":"3946_CR8","volume-title":"Causation, Prediction, and Search","author":"P Spirtes","year":"2000","unstructured":"Spirtes P, Glymour C N, Scheines R. Causation, Prediction, and Search. 2nd ed. Cambridge: MIT Press, 2000","edition":"2nd ed."},{"key":"3946_CR9","first-page":"804","volume-title":"Proceedings of the 27th Conference on Uncertainty in Artificial Intelligence","author":"K Zhang","year":"2011","unstructured":"Zhang K, Peters J, Janzing D, Sch\u00f6lkopf B. Kernel-based conditional independence test and application in causal discovery. In: Proceedings of the 27th Conference on Uncertainty in Artificial Intelligence. 2011, 804\u2013813"},{"key":"3946_CR10","doi-asserted-by":"publisher","first-page":"855","DOI":"10.1145\/1273496.1273604","volume-title":"Proceedings of the 24th International Conference on Machine Learning","author":"X Sun","year":"2007","unstructured":"Sun X, Janzing D, Sch\u00f6lkopf B, Fukumizu K. A kernel-based causal learning algorithm. In: Proceedings of the 24th International Conference on Machine Learning. 2007, 855\u2013862"},{"key":"3946_CR11","first-page":"1","volume-title":"Innovations in Machine Learning: Theory and Applications","author":"D Heckerman","year":"2006","unstructured":"Heckerman D, Meek C, Cooper G. A Bayesian approach to causal discovery. In: Holmes D E, Jain L C, eds. Innovations in Machine Learning: Theory and Applications. Berlin, Heidelberg: Springer, 2006, 1\u201328"},{"key":"3946_CR12","first-page":"825","volume-title":"Proceedings of the 20th National Conference on Artificial Intelligence","author":"D Margaritis","year":"2005","unstructured":"Margaritis D. Distribution-free learning of Bayesian network structure in continuous domains. In: Proceedings of the 20th National Conference on Artificial Intelligence. 2005, 825\u2013830"},{"key":"3946_CR13","volume-title":"Learning neural causal models from unknown interventions","author":"N R Ke","year":"2019","unstructured":"Ke N R, Bilaniuk O, Goyal A, Bauer S, Larochelle H, Sch\u00f6lkopf B, Mozer M C, Pal C, Bengio Y. Learning neural causal models from unknown interventions. 2019, arXiv preprint arXiv: 1910.01075"},{"key":"3946_CR14","first-page":"23151","volume-title":"Proceedings of the 39th International Conference on Machine Learning","author":"Z Wang","year":"2022","unstructured":"Wang Z, Xiao X, Xu Z, Zhu Y, Stone P. Causal dynamics learning for task-independent state abstraction. In: Proceedings of the 39th International Conference on Machine Learning. 2022, 23151\u201323180"},{"issue":"5","key":"3946_CR15","first-page":"679","volume":"6","author":"R Bellman","year":"1957","unstructured":"Bellman R. A Markovian decision process. Journal of Mathematics and Mechanics, 1957, 6(5): 679\u2013684","journal-title":"Journal of Mathematics and Mechanics"},{"key":"3946_CR16","volume-title":"Proceedings of the 6th International Conference on Learning Representations","author":"T Kurutach","year":"2018","unstructured":"Kurutach T, Clavera I, Duan Y, Tamar A, Abbeel P. Model-ensemble trust-region policy optimization. In: Proceedings of the 6th International Conference on Learning Representations. 2018"},{"key":"3946_CR17","first-page":"1714","volume-title":"Proceedings of the IEEE International Conference on Robotics and Automation (ICRA)","author":"G Williams","year":"2017","unstructured":"Williams G, Wagener N, Goldfain B, Drews P, Rehg J M, Boots B, Theodorou E A. Information theoretic MPC for model-based reinforcement learning. In: Proceedings of the IEEE International Conference on Robotics and Automation (ICRA). 2017, 1714\u20131721"},{"key":"3946_CR18","first-page":"1830","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"R Kidambi","year":"2020","unstructured":"Kidambi R, Rajeswaran A, Netrapalli P, Joachims T. MOReL: model-based offline reinforcement learning. In: Proceedings of the 34th International Conference on Neural Information Processing Systems. 2020, 1830"},{"key":"3946_CR19","volume-title":"Solving Rubik\u2019s cube with a robot hand","author":"I Akkaya","year":"2019","unstructured":"Akkaya I, Andrychowicz M, Chociej M, Litwin M, McGrew B, Petron A, Paino A, Plappert M, Powell G, Ribas R, Schneider J, Tezak N, Tworek J, Welinder P, Weng L, Yuan Q, Zaremba W, Zhang L. Solving Rubik\u2019s cube with a robot hand. 2019, arXiv preprint arXiv: 1910.07113"},{"issue":"1","key":"3946_CR20","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1080\/00401706.2000.10485983","volume":"42","author":"A E Hoerl","year":"2000","unstructured":"Hoerl A E, Kennard R W. Ridge regression: biased estimation for nonorthogonal problems. Technometrics, 2000, 42(1): 80\u201386","journal-title":"Technometrics"},{"key":"3946_CR21","volume-title":"Causality: Models, Reasoning, and Inference","author":"J Pearl","year":"2000","unstructured":"Pearl J. Causality: Models, Reasoning, and Inference. Cambridge: Cambridge University Press, 2000"},{"key":"3946_CR22","volume-title":"Probabilistic Graphical Models: Principles and Techniques","author":"D Koller","year":"2009","unstructured":"Koller D, Friedman N. Probabilistic Graphical Models: Principles and Techniques. Cambridge: MIT Press, 2009"},{"key":"3946_CR23","first-page":"2218","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"T Yu","year":"2021","unstructured":"Yu T, Kumar A, Rafailov R, Rajeswaran A, Levine S, Finn C. COMBO: conservative offline model-based policy optimization. In: Proceedings of the 35th International Conference on Neural Information Processing Systems. 2021, 2218"},{"key":"3946_CR24","first-page":"5026","volume-title":"Proceedings of the IEEE\/RSJ International Conference on Intelligent Robots and Systems","author":"E Todorov","year":"2012","unstructured":"Todorov E, Erez T, Tassa Y. MuJoCo: a physics engine for model-based control. In: Proceedings of the IEEE\/RSJ International Conference on Intelligent Robots and Systems. 2012, 5026\u20135033"},{"key":"3946_CR25","volume-title":"D4RL: datasets for deep data-driven reinforcement learning","author":"J Fu","year":"2020","unstructured":"Fu J, Kumar A, Nachum O, Tucker G, Levine S. D4RL: datasets for deep data-driven reinforcement learning. 2020, arXiv preprint arXiv: 2004.07219"},{"key":"3946_CR26","first-page":"1320","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"T Xu","year":"2020","unstructured":"Xu T, Li Z, Yu Y. Error bounds of imitating policies and environments. In: Proceedings of the 34th International Conference on Neural Information Processing Systems. 2020, 1320"}],"container-title":["Frontiers of Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-024-3946-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11704-024-3946-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11704-024-3946-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T19:23:44Z","timestamp":1779305024000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11704-024-3946-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,14]]},"references-count":26,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["3946"],"URL":"https:\/\/doi.org\/10.1007\/s11704-024-3946-y","relation":{},"ISSN":["2095-2228","2095-2236"],"issn-type":[{"value":"2095-2228","type":"print"},{"value":"2095-2236","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,14]]},"assertion":[{"value":"23 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Competing interests\n                      The authors declare that they have no competing interests or financial conflicts to disclose.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics"}}],"article-number":"194347"}}