{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T20:22:19Z","timestamp":1769545339632,"version":"3.49.0"},"publisher-location":"Cham","reference-count":61,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032156310","type":"print"},{"value":"9783032156327","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-15632-7_4","type":"book-chapter","created":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T07:22:56Z","timestamp":1769498576000},"page":"60-77","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Baba Is LLM: Reasoning in\u00a0a\u00a0Game with\u00a0Dynamic Rules"],"prefix":"10.1007","author":[{"given":"Fien","family":"van Wetten","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aske","family":"Plaat","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Max","family":"van Duijn","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,28]]},"reference":[{"issue":"6624","key":"4_CR1","doi-asserted-by":"publisher","first-page":"1067","DOI":"10.1126\/science.ade9097","volume":"378","author":"A Bakhtin","year":"2022","unstructured":"Bakhtin, A., et al.: Human-level play in the game of diplomacy by combining language models with strategic reasoning. Science 378(6624), 1067\u20131074 (2022)","journal-title":"Science"},{"key":"4_CR2","unstructured":"Berner, C., et\u00a0al.: Dota 2 with large scale deep reinforcement learning, December 2019. arXiv:1912.06680 [cs]"},{"key":"4_CR3","unstructured":"Bjorklund, I.: cot-logic-reasoning (2025). https:\/\/huggingface.co\/datasets\/isaiahbjork\/cot-logic-reasoning"},{"issue":"6374","key":"4_CR4","doi-asserted-by":"publisher","first-page":"418","DOI":"10.1126\/science.aao1733","volume":"359","author":"N Brown","year":"2018","unstructured":"Brown, N., Sandholm, T.: Superhuman AI for heads-up no-limit poker: Libratus beats top professionals. Science 359(6374), 418\u2013424 (2018). https:\/\/doi.org\/10.1126\/science.aao1733","journal-title":"Science"},{"key":"4_CR5","doi-asserted-by":"crossref","unstructured":"Campbell, M., Hoane\u00a0Jr, A.J., Hsu, F.h.: Deep blue. Artif. Intell. 134(1-2), 57\u201383 (2002)","DOI":"10.1016\/S0004-3702(01)00129-1"},{"key":"4_CR6","doi-asserted-by":"publisher","unstructured":"Charity, M., Togelius, J.: Keke AI competition: solving puzzle levels in a dynamically changing mechanic space. In: 2022 IEEE Conference on Games (CoG), pp. 570\u2013575, August 2022. https:\/\/doi.org\/10.1109\/CoG51982.2022.9893650, iSSN: 2325-4289","DOI":"10.1109\/CoG51982.2022.9893650"},{"key":"4_CR7","doi-asserted-by":"publisher","unstructured":"Charity, M., Khalifa, A., Togelius, J.: Baba is y\u2019all: collaborative mixed-initiative level design. In: 2020 IEEE Conference on Games (CoG), pp. 542\u2013549 (2020). https:\/\/doi.org\/10.1109\/CoG47356.2020.9231807","DOI":"10.1109\/CoG47356.2020.9231807"},{"key":"4_CR8","unstructured":"Chaudhari, S., et al.: RLHF Deciphered: A Critical Analysis of Reinforcement Learning from Human Feedback for LLMs, April 2024. arXiv:2404.08555 [cs]"},{"key":"4_CR9","doi-asserted-by":"crossref","unstructured":"Chu, Z., et al.: A survey of chain of thought reasoning: Advances, frontiers and future. In: Association for Computational Linguistics (2024)","DOI":"10.18653\/v1\/2024.acl-long.65"},{"key":"4_CR10","unstructured":"Cloos, N., et al.: Baba is ai: Break the rules to beat the benchmark. arXiv preprint arXiv:2407.13729 (2024)"},{"key":"4_CR11","unstructured":"Dong, Q., et al.: A survey on in-context learning. In: Association for Computational Linguistics (2023)"},{"key":"4_CR12","unstructured":"Feng, X., et al.: Chessgpt: Bridging policy learning and language modeling. Advances in Neural Information Processing Systems 36 (2024)"},{"key":"4_CR13","unstructured":"Gemini, T.: Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context, December 2024. arXiv:2403.05530 [cs]"},{"key":"4_CR14","unstructured":"Han, Z., Gao, C., Liu, J., Zhang, J., Zhang, S.Q.: Parameter-Efficient Fine-Tuning for Large Models: A Comprehensive Survey, September 2024. arXiv:2403.14608 [cs]"},{"key":"4_CR15","unstructured":"Hu, E.J., et al.: LoRA: Low-Rank Adaptation of Large Language Models, October 2021. arXiv:2106.09685 [cs]"},{"key":"4_CR16","doi-asserted-by":"crossref","unstructured":"Huang, J., Chang, K.C.C.: Towards reasoning in large language models: a survey. In: Assoc for Computational Linguistics (2023)","DOI":"10.18653\/v1\/2023.findings-acl.67"},{"key":"4_CR17","unstructured":"Hurst, A., et\u00a0al.: Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)"},{"key":"4_CR18","unstructured":"Jeong, C.: Fine-tuning and utilization methods of domain-specific llms. arXiv preprint arXiv:2401.02981 (2024)"},{"key":"4_CR19","unstructured":"Jiang, A.Q., et\u00a0al.: Mistral 7B, October 2023. arXiv:2310.06825 [cs]"},{"key":"4_CR20","unstructured":"Jiang, A.Q., et\u00a0al.: Mixtral of Experts, January 2024. arXiv:2401.04088 [cs]"},{"key":"4_CR21","doi-asserted-by":"publisher","unstructured":"Kamath, U., Keenan, K., Somers, G., Sorenson, S.: Prompt-based Learning. In: Large Language Models: A Deep Dive: Bridging Theory and Practice, pp. 83\u2013133. Springer Nature Switzerland, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-65647-7_3","DOI":"10.1007\/978-3-031-65647-7_3"},{"key":"4_CR22","unstructured":"Karvonen, A.: Emergent world models and latent variable estimation in chess-playing language models. arXiv preprint arXiv:2403.15498 (2024)"},{"key":"4_CR23","first-page":"22199","volume":"35","author":"T Kojima","year":"2022","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners. Adv. Neural. Inf. Process. Syst. 35, 22199\u201322213 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR24","unstructured":"Li, K., Hopkins, A.K., Bau, D., Vi\u00e9gas, F., Pfister, H., Wattenberg, M.: Emergent world representations: exploring a sequence model trained on a synthetic task. ICLR (2023)"},{"key":"4_CR25","doi-asserted-by":"publisher","unstructured":"Li, Y., Wang, H., Zhang, C.: Assessing logical puzzle solving in large language models: insights from a minesweeper case study. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 59\u201381. Association for Computational Linguistics, Mexico City, Mexico (2024). https:\/\/doi.org\/10.18653\/v1\/2024.naacl-long.4","DOI":"10.18653\/v1\/2024.naacl-long.4"},{"key":"4_CR26","first-page":"46534","volume":"36","author":"A Madaan","year":"2023","unstructured":"Madaan, A., et al.: Self-refine: iterative refinement with self-feedback. Adv. Neural. Inf. Process. Syst. 36, 46534\u201346594 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR27","doi-asserted-by":"crossref","unstructured":"Marincioni, A., et al.: The effect of llm-based npc emotional states on player emotions: an analysis of interactive game play. In: 2024 IEEE Conference on Games (CoG), pp.\u00a01\u20136. IEEE (2024)","DOI":"10.1109\/CoG60054.2024.10645631"},{"key":"4_CR28","unstructured":"Mirchandani, S., et al.: Large Language Models as General Pattern Machines, October 2023. arXiv:2307.04721 [cs]"},{"key":"4_CR29","first-page":"50358","volume":"36","author":"N Muennighoff","year":"2023","unstructured":"Muennighoff, N., et al.: Scaling data-constrained language models. Adv. Neural. Inf. Process. Syst. 36, 50358\u201350376 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR30","doi-asserted-by":"crossref","unstructured":"M\u00fcller-Brockhausen, M., Barbero, G., Preuss, M.: Chatter generation through language models. In: 2023 IEEE Conference on Games (CoG), pp.\u00a01\u20136. IEEE (2023)","DOI":"10.1109\/CoG57401.2023.10333244"},{"key":"4_CR31","doi-asserted-by":"crossref","unstructured":"Nanda, N., Lee, A., Wattenberg, M.: Emergent linear representations in world models of self-supervised sequence models. arXiv preprint arXiv:2309.00941 (2023)","DOI":"10.18653\/v1\/2023.blackboxnlp-1.2"},{"key":"4_CR32","unstructured":"Naveed, H., et al.: A Comprehensive Overview of Large Language Models, October 2024. arXiv:2307.06435 [cs]"},{"key":"4_CR33","doi-asserted-by":"crossref","unstructured":"Noever, D.A., Burdick, R.: Puzzle solving without search or human knowledge: An unnatural language approach (2021). https:\/\/api.semanticscholar.org\/CorpusID:237431487","DOI":"10.5121\/csit.2022.120902"},{"key":"4_CR34","doi-asserted-by":"crossref","unstructured":"Plaat, A.: Learning to play: reinforcement learning and games. Springer Nature (2020)","DOI":"10.1007\/978-3-030-59238-7"},{"key":"4_CR35","doi-asserted-by":"crossref","unstructured":"Plaat, A., van Duijn, M., van Stein, N., Preuss, M., van\u00a0der Putten, P., Batenburg, K.J.: Agentic large language models, a survey. arXiv preprint arXiv:2503.23037 (2025)","DOI":"10.1613\/jair.1.18675"},{"key":"4_CR36","doi-asserted-by":"crossref","unstructured":"Plaat, A., Wong, A., Verberne, S., Broekens, J., van Stein, N., Back, T.: Reasoning with Large Language Models, a Survey, July 2024. arXiv:2407.11511 [cs]","DOI":"10.1145\/3774896"},{"key":"4_CR37","doi-asserted-by":"crossref","unstructured":"Press, O., Zhang, M., Min, S., Schmidt, L., Smith, N.A., Lewis, M.: Measuring and Narrowing the Compositionality Gap in Language Models, October 2023. arXiv:2210.03350 [cs]","DOI":"10.18653\/v1\/2023.findings-emnlp.378"},{"key":"4_CR38","unstructured":"Radford, A., Narasimhan, K., Salimans, T., Sutskever, I.: Improving Language Understanding by Generative Pre-Training (2018)"},{"issue":"425","key":"4_CR39","doi-asserted-by":"publisher","first-page":"113","DOI":"10.1093\/mind\/107.425.113","volume":"107","author":"P Saka","year":"1998","unstructured":"Saka, P.: Quotation and the use-mention distinction. Mind 107(425), 113\u2013135 (1998)","journal-title":"Mind"},{"issue":"7839","key":"4_CR40","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1038\/s41586-020-03051-4","volume":"588","author":"J Schrittwieser","year":"2020","unstructured":"Schrittwieser, J., et al.: Mastering atari, go, chess and shogi by planning with a learned model. Nature 588(7839), 604\u2013609 (2020)","journal-title":"Nature"},{"key":"4_CR41","unstructured":"Schultz, J., et\u00a0al.: Mastering board games by external and internal planning with language models. arXiv preprint arXiv:2412.12119 (2024)"},{"key":"4_CR42","first-page":"8634","volume":"36","author":"N Shinn","year":"2023","unstructured":"Shinn, N., Cassano, F., Gopinath, A., Narasimhan, K., Yao, S.: Reflexion: language agents with verbal reinforcement learning. Adv. Neural. Inf. Process. Syst. 36, 8634\u20138652 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR43","doi-asserted-by":"crossref","unstructured":"Silver, D., et\u00a0al.: Mastering the game of go with deep neural networks and tree search. Nature 529(7587), 484\u2013489 (2016)","DOI":"10.1038\/nature16961"},{"issue":"6419","key":"4_CR44","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.1126\/science.aar6404","volume":"362","author":"D Silver","year":"2018","unstructured":"Silver, D., et al.: A general reinforcement learning algorithm that masters chess, shogi, and go through self-play. Science 362(6419), 1140\u20131144 (2018)","journal-title":"Science"},{"key":"4_CR45","unstructured":"Teikari, A.: Baba is you. Game [PC]. (March 2019). Hempuli Oy, Finland (2019)"},{"key":"4_CR46","unstructured":"Topsakal, O., Edell, C.J., Harper, J.B.: Evaluating large language models with grid-based game competitions: an extensible llm benchmark and leaderboard. arXiv preprint arXiv:2407.07796 (2024)"},{"key":"4_CR47","unstructured":"Valmeekam, K., Stechly, K., Gundawar, A., Kambhampati, S.: Planning in Strawberry Fields: Evaluating and Improving the Planning and Scheduling Capabilities of LRM o1, October 2024. arXiv:2410.02162 [cs]"},{"key":"4_CR48","unstructured":"Vaswani, A., et al.: Attention Is All You Need, August 2023. arXiv:1706.03762 [cs]"},{"key":"4_CR49","doi-asserted-by":"crossref","unstructured":"Wang, L., et al.: Plan-and-solve prompting: Improving zero-shot chain-of-thought reasoning by large language models. arXiv preprint arXiv:2305.04091 (2023)","DOI":"10.18653\/v1\/2023.acl-long.147"},{"key":"4_CR50","doi-asserted-by":"crossref","unstructured":"Wang, L., et al.: Plan-and-Solve Prompting: Improving Zero-Shot Chain-of-Thought Reasoning by Large Language Models, May 2023. arXiv:2305.04091 [cs]","DOI":"10.18653\/v1\/2023.acl-long.147"},{"key":"4_CR51","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. Adv. Neural. Inf. Process. Syst. 35, 24824\u201324837 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR52","unstructured":"van Wetten, F., Plaat, A., van Duijn, M.: Baba is llm: reasoning in a game with dynamic rules. arXiv preprint arXiv:2506.19095 (2025)"},{"key":"4_CR53","unstructured":"van Wetten, F.: Llm_babaisyou. https:\/\/github.com\/fien99\/LLM_babaisyou (2025)"},{"key":"4_CR54","unstructured":"van Wetten, F.: Mistral_7b_instruct-baba (2025). https:\/\/huggingface.co\/Migthytwig\/Mistral_7B_instruct-baba"},{"key":"4_CR55","unstructured":"van Wetten, F.: Olmo_7b_instruct-baba (2025). https:\/\/huggingface.co\/Migthytwig\/olmo_7B_instruct-baba"},{"key":"4_CR56","doi-asserted-by":"crossref","unstructured":"Wilson, S.: A bridge from the use-mention distinction to natural language processing. The Semantics and Pragmatics of Quotation, pp. 79\u201396 (2017)","DOI":"10.1007\/978-3-319-68747-6_4"},{"key":"4_CR57","unstructured":"Xu, L., Xie, H., Qin, S.Z.J., Tao, X., Wang, F.L.: Parameter-Efficient Fine-Tuning Methods for Pretrained Language Models: A Critical Review and Assessment, December 2023. arXiv:2312.12148 [cs]"},{"key":"4_CR58","first-page":"11809","volume":"36","author":"S Yao","year":"2023","unstructured":"Yao, S., Yu, D., Zhao, J., Shafran, I., Griffiths, T., Cao, Y., Narasimhan, K.: Tree of thoughts: deliberate problem solving with large language models. Adv. Neural. Inf. Process. Syst. 36, 11809\u201311822 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"4_CR59","unstructured":"Yasunaga, M., et al.: Large language models as analogical reasoners. arXiv preprint arXiv:2310.01714 (2023)"},{"key":"4_CR60","unstructured":"Zhang, S., et al.: Instruction Tuning for Large Language Models: A Survey, March 2024. arXiv:2308.10792 [cs]"},{"key":"4_CR61","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Han, X., Li, H., Chen, K., Lin, S.: Complete chess games enable llm become a chess master. arXiv preprint arXiv:2501.17186 (2025)","DOI":"10.18653\/v1\/2025.naacl-short.1"}],"container-title":["Communications in Computer and Information Science","Computational Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-15632-7_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T07:23:06Z","timestamp":1769498586000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-15632-7_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032156310","9783032156327"],"references-count":61,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-15632-7_4","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"28 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"IJCCI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Joint Conference on Computational Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Marbella","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Spain","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 October 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 October 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ijcci2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ijcci.scitevents.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}