{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T14:49:20Z","timestamp":1782312560586,"version":"3.54.5"},"publisher-location":"Cham","reference-count":37,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032061058","type":"print"},{"value":"9783032061065","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,10,3]],"date-time":"2025-10-03T00:00:00Z","timestamp":1759449600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,10,3]],"date-time":"2025-10-03T00:00:00Z","timestamp":1759449600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-06106-5_13","type":"book-chapter","created":{"date-parts":[[2025,10,2]],"date-time":"2025-10-02T10:08:38Z","timestamp":1759399718000},"page":"216-232","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Bilevel Reinforcement Learning Framework with\u00a0Language Prior Knowledge"],"prefix":"10.1007","author":[{"given":"Xue","family":"Yan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yan","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinyu","family":"Cui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Filippos","family":"Christianos","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haifeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"David","family":"Mguni","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,10,3]]},"reference":[{"key":"13_CR1","unstructured":"Albrecht, S.V., Christianos, F., Sch\u00e4fer, L.: Multi-Agent Reinforcement Learning: Foundations and Modern Approaches. MIT Press (2023). https:\/\/www.marl-book.com"},{"key":"13_CR2","doi-asserted-by":"publisher","first-page":"270","DOI":"10.1016\/j.ijar.2018.10.001","volume":"103","author":"AE Allahverdyan","year":"2018","unstructured":"Allahverdyan, A.E., Galstyan, A., Abbas, A.E., Struzik, Z.R.: Adaptive decision making via entropy minimization. Int. J. Approx. Reas. 103, 270\u2013287 (2018)","journal-title":"Int. J. Approx. Reas."},{"key":"13_CR3","unstructured":"Brooks, E., Walls, L., Lewis, R.L., Singh, S.: In-context policy iteration. arXiv preprint arXiv:2210.03821 (2022)"},{"key":"13_CR4","unstructured":"Carta, T., Romac, C., Wolf, T., Lamprier, S., Sigaud, O., Oudeyer, P.Y.: Grounding large language models in interactive environments with online reinforcement learning. arXiv preprint arXiv:2302.02662 (2023)"},{"key":"13_CR5","unstructured":"Chen, L., et\u00a0al.: Introspective tips: large language model for in-context decision making. arXiv preprint arXiv:2305.11598 (2023)"},{"key":"13_CR6","unstructured":"Cheng, C.A., Kolobov, A., Misra, D., Nie, A., Swaminathan, A.: LLF-Bench: benchmark for interactive learning from language feedback. arXiv preprint arXiv:2312.06853 (2023)"},{"key":"13_CR7","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1007\/s10479-007-0176-2","volume":"153","author":"B Colson","year":"2007","unstructured":"Colson, B., Marcotte, P., Savard, G.: An overview of bilevel optimization. Ann. Oper. Res. 153, 235\u2013256 (2007)","journal-title":"Ann. Oper. Res."},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"C\u00f4t\u00e9, M.A., et\u00a0al.: TextWorld: a learning environment for text-based games. In: Computer Games: 7th Workshop, CGW 2018, Held in Conjunction with the 27th International Conference on Artificial Intelligence, IJCAI 2018, Stockholm, Sweden, July 13, 2018, Revised Selected Papers 7, pp. 41\u201375. Springer (2019)","DOI":"10.1007\/978-3-030-24337-1_3"},{"key":"13_CR9","unstructured":"Eysenbach, B., Levine, S.: Maximum entropy RL (provably) solves some robust RL problems. arXiv preprint arXiv:2103.06257 (2021)"},{"key":"13_CR10","unstructured":"Gao, L., et al.: Pal: program-aided language models. In: International Conference on Machine Learning, pp. 10764\u201310799. PMLR (2023)"},{"key":"13_CR11","unstructured":"Haarnoja, T., et\u00a0al.: Soft actor-critic algorithms and applications. arXiv preprint arXiv:1812.05905 (2018)"},{"key":"13_CR12","unstructured":"Hao, S., et al.: Reasoning with language model is planning with world model. arXiv preprint arXiv:2305.14992 (2023)"},{"key":"13_CR13","doi-asserted-by":"crossref","unstructured":"Hinz, A.M., Klav\u017ear, S., Milutinovi\u0107, U., Petr, C.: The tower of Hanoi-Myths and Maths. Springer (2013)","DOI":"10.1007\/978-3-0348-0237-6"},{"key":"13_CR14","unstructured":"Hu, E.J., et\u00a0al.: Lora: low-rank adaptation of large language models. In: International Conference on Learning Representations"},{"key":"13_CR15","unstructured":"Jang, Y., Lee, J., Kim, K.E.: Gpt-Critic: offline reinforcement learning for end-to-end task-oriented dialogue systems. In: International Conference on Learning Representations (2021)"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"Lin, B.Y., et al.: SwiftSage: a generative agent with fast and slow thinking for complex interactive tasks. Adv. Neural Info. Process. Syst. 36 (2024)","DOI":"10.52202\/075280-1034"},{"key":"13_CR17","unstructured":"Lu, P., et al.: Dynamic prompt learning via policy gradient for semi-structured mathematical reasoning. arXiv preprint arXiv:2209.14610 (2022)"},{"key":"13_CR18","unstructured":"Mnih, V., et al.: Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning (2016)"},{"key":"13_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2020.106203","volume":"91","author":"M Pakseresht","year":"2020","unstructured":"Pakseresht, M., Mahdavi, I., Shirazi, B., Mahdavi-Amiri, N.: Co-reconfiguration of product family and supply chain using leader-follower Stackelberg game theory: bi-level multi-objective optimization. Appl. Soft Comput. 91, 106203 (2020)","journal-title":"Appl. Soft Comput."},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Park, J.S., O\u2019Brien, J.C., Cai, C.J., Morris, M.R., Liang, P., Bernstein, M.S.: Generative agents: interactive simulacra of human behavior. arXiv preprint arXiv:2304.03442 (2023)","DOI":"10.1145\/3586183.3606763"},{"key":"13_CR21","unstructured":"Rae, J.W., et\u00a0al.: Scaling language models: methods, analysis and insights from training gopher. arXiv preprint arXiv:2112.11446 (2021)"},{"key":"13_CR22","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"13_CR23","unstructured":"Shah, D., Equi, M.R., Osi\u0144ski, B., Xia, F., Levine, S., et\u00a0al.: Navigation with large language models: semantic guesswork as a heuristic for planning. In: 7th Annual Conference on Robot Learning (2023)"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Shinn, N., Cassano, F., Gopinath, A., Narasimhan, K., Yao, S.: Reflexion: language agents with verbal reinforcement learning. Adv. Neural Info. Process. Syst. 36 (2024)","DOI":"10.52202\/075280-0377"},{"key":"13_CR25","unstructured":"Shridhar, M., Yuan, X., C\u00f4t\u00e9, M.A., Bisk, Y., Trischler, A., Hausknecht, M.: ALFWorld: aligning text and embodied environments for interactive learning. arXiv preprint arXiv:2010.03768 (2020)"},{"key":"13_CR26","doi-asserted-by":"crossref","unstructured":"Sordoni, A., et al.: Deep language networks: joint prompt training of stacked LLMs using variational inference. arXiv preprint arXiv:2306.12509 (2023)","DOI":"10.52202\/075280-2534"},{"key":"13_CR27","unstructured":"Tan, W., Zhang, W., Liu, S., Zheng, L., Wang, X., Bo, A.: True knowledge comes from practice: aligning large language models with embodied environments via reinforcement learning. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"13_CR28","unstructured":"Wang, X., et al.: Self-consistency improves chain of thought reasoning in language models. arXiv preprint arXiv:2203.11171 (2022)"},{"key":"13_CR29","doi-asserted-by":"crossref","unstructured":"Wei, J., et al.: Chain of thought prompting elicits reasoning in large language models. arXiv preprint arXiv:2201.11903 (2022)","DOI":"10.52202\/068431-1800"},{"key":"13_CR30","doi-asserted-by":"crossref","unstructured":"Yan, X., Guo, J., Lou, X., Wang, J., Zhang, H., Du, Y.: An efficient end-to-end training approach for zero-shot human-ai coordination. Adv. Neural Info. Process. Syst. 36 (2024)","DOI":"10.52202\/075280-0119"},{"key":"13_CR31","doi-asserted-by":"crossref","unstructured":"Yao, S., et al.: Tree of thoughts: deliberate problem solving with large language models. arXiv preprint arXiv:2305.10601 (2023)","DOI":"10.52202\/075280-0517"},{"key":"13_CR32","unstructured":"Yao, S., et al.: React: synergizing reasoning and acting in language models. arXiv preprint arXiv:2210.03629 (2022)"},{"key":"13_CR33","unstructured":"Yao, W., et\u00a0al.: RetroFormer: retrospective large language agents with policy gradient optimization. arXiv preprint arXiv:2308.02151 (2023)"},{"key":"13_CR34","unstructured":"Zhang, C., et\u00a0al.: ProAgent: building proactive cooperative AI with large language models. arXiv preprint arXiv:2308.11339 (2023)"},{"key":"13_CR35","unstructured":"Zhang, J., Yu, H., Xu, W.: Hierarchical reinforcement learning by discovering intrinsic options. arXiv preprint arXiv:2101.06521 (2021)"},{"key":"13_CR36","doi-asserted-by":"crossref","unstructured":"Zhou, R., Du, S.S., Li, B.: Reflect-RL: two-player online RL fine-tuning for LMS. arXiv preprint arXiv:2402.12621 (2024)","DOI":"10.18653\/v1\/2024.acl-long.56"},{"key":"13_CR37","unstructured":"Zhou, Y., et al.: Large language models are human-level prompt engineers. arXiv preprint arXiv:2211.01910 (2022)"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases. Research Track"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-06106-5_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T14:21:58Z","timestamp":1782310918000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-06106-5_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,3]]},"ISBN":["9783032061058","9783032061065"],"references-count":37,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-06106-5_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10,3]]},"assertion":[{"value":"3 October 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECML PKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Porto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Portugal","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecml2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecmlpkdd.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}