{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T00:20:50Z","timestamp":1759191650711,"version":"3.44.0"},"publisher-location":"Cham","reference-count":19,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032060778","type":"print"},{"value":"9783032060785","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T00:00:00Z","timestamp":1759190400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T00:00:00Z","timestamp":1759190400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-06078-5_12","type":"book-chapter","created":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T18:50:10Z","timestamp":1759171810000},"page":"205-223","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Advancing Multi-step Mathematical Reasoning in\u00a0Large Language Models Through Multi-layered Self-reflection with\u00a0Auto-prompting"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-6856-2804","authenticated-orcid":false,"given":"Andr\u00e9","family":"de Souza Loureiro","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8664-9692","authenticated-orcid":false,"given":"Jorge","family":"Valverde-Rebaza","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6000-3452","authenticated-orcid":false,"given":"Julieta","family":"Noguez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7374-3526","authenticated-orcid":false,"given":"David","family":"Escarcega","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2309-3487","authenticated-orcid":false,"given":"Ricardo","family":"Marcacini","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,9,30]]},"reference":[{"issue":"12","key":"12_CR1","doi-asserted-by":"publisher","first-page":"304","DOI":"10.1145\/3664194","volume":"56","author":"F Yu","year":"2024","unstructured":"Yu, F., Zhang, H., Tiwari, P., Wang, B.: Natural language reasoning. A Survey. ACM Comput. Surv. 56(12), 304 (2024). https:\/\/doi.org\/10.1145\/3664194","journal-title":"A Survey. ACM Comput. Surv."},{"key":"12_CR2","doi-asserted-by":"publisher","unstructured":"Cobbe, K., Kosaraju, V., Bavarian, M., et al. 2021. Training Verifiers to Solve Math Word Problems. arXiv 2110.14168. https:\/\/doi.org\/10.48550\/arXiv.2110.14168","DOI":"10.48550\/arXiv.2110.14168"},{"key":"12_CR3","unstructured":"DeepSeek-AI, Guo, D., Yang, D., Zhang, H., Song, J., et al.: DeepSeek-R1: incentivizing reasoning capability in LLMs via reinforcement learning (2025). arXiv 2501.12948. URL: https:\/\/arxiv.org\/abs\/2501.12948"},{"key":"12_CR4","first-page":"1","volume":"7","author":"J Dem\u0161ar","year":"2006","unstructured":"Dem\u0161ar, J.: Statistical comparisons of classifiers over multiple data sets. J. Mach. Learn. Res. 7, 1\u201330 (2006)","journal-title":"J. Mach. Learn. Res."},{"key":"12_CR5","unstructured":"Liu, F., AlDahoul, N., Eady, G., Zaki, Y., AlShebli, B., Rahwan, T: Self-Reflection Outcome is Sensitive to Prompt Construction (2024). arXiv 2406.10400. URL: https:\/\/arxiv.org\/abs\/2406.10400"},{"key":"12_CR6","unstructured":"Minaee, S., et al.: Large language models: a survey (2024). arXiv 2402.06196. URL: https:\/\/arxiv.org\/abs\/2402.06196"},{"key":"12_CR7","doi-asserted-by":"publisher","unstructured":"Mirzadeh, I., Alizadeh, K., Shahrokhi, H., Tuzel, O., Bengio, S., Farajtabar, M.: GSM-Symbolic: understanding the limitations of mathematical reasoning in large language models (2024). arXiv 2410.05229. https:\/\/doi.org\/10.48550\/arXiv.2410.05229","DOI":"10.48550\/arXiv.2410.05229"},{"key":"12_CR8","unstructured":"OpenAI. 2025. OpenAI o3-mini. URL: https:\/\/openai.com\/index\/openai-o3-mini\/. Accessed 04 Feb 2025"},{"key":"12_CR9","unstructured":"OpenAI, Achiam, J., Adler, S., Agarwal, S., et al.: GPT-4 Technical report (2024). arXiv 2303.08774. URL: https:\/\/arxiv.org\/abs\/2303.08774"},{"key":"12_CR10","doi-asserted-by":"publisher","unstructured":"Renze, M., Guven, E.: Self-Reflection in LLM agents: effects on problem-solving performance (2024). arXiv 2405.06682. https:\/\/doi.org\/10.48550\/arXiv.2405.06682","DOI":"10.48550\/arXiv.2405.06682"},{"key":"12_CR11","doi-asserted-by":"publisher","unstructured":"Valmeekam, K., Stechly, K., Kambhampati, S.: LLMs still can\u2019t plan; can LRMs? A preliminary evaluation of OpenAI\u2019s o1 on planbench (2024). arXiv 2409.13373. https:\/\/doi.org\/10.48550\/arXiv.2409.13373","DOI":"10.48550\/arXiv.2409.13373"},{"key":"12_CR12","doi-asserted-by":"publisher","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models (2022). arXiv 2201.11903. https:\/\/doi.org\/10.48550\/arXiv.2201.11903","DOI":"10.48550\/arXiv.2201.11903"},{"key":"12_CR13","unstructured":"Xu, M., Ning, Y., Li, Y., et al.: Reasoning based on symbolic and parametric knowledge bases: a survey (2025). arXiv 2501.01030. URL: https:\/\/arxiv.org\/abs\/2501.01030"},{"key":"12_CR14","unstructured":"Zhong, Q., Wang, K., Xu, Z., Liu, J., Ding, L.: Achieving $$>$$97% on GSM8K: deeply understanding the problems makes LLMs better solvers for math word problems (2024). arXiv 2404.14963"},{"key":"12_CR15","doi-asserted-by":"publisher","unstructured":"Wang, L., et al.: Plan-and-solve prompting: improving zero-shot chain-of-thought reasoning by large language models. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (, vol. 1: Long Papers), pp. 2609\u20132634 (2023). ACL. URL: https:\/\/aclanthology.org\/2023.acl-long.147\/. https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.147","DOI":"10.18653\/v1\/2023.acl-long.147"},{"key":"12_CR16","unstructured":"Hendrycks, D., et al.: Measuring mathematical problem solving with the MATH dataset. In: NeurIPS (2021)"},{"key":"12_CR17","unstructured":"Balunovi\u0107, M., Dekoninck, J., Petrov, I., Jovanovi\u0107, N., Vechev, M.: MathArena: evaluating LLMs on uncontaminated math competitions (2025). SRI Lab, ETH Zurich. URL: https:\/\/matharena.ai\/. Accessed 12 April 2025"},{"key":"12_CR18","unstructured":"Gemini-Team, Anil, R., Borgeaud, S., Alayrac, J.-B., et al.: Gemini: a family of highly capable multimodal models (2025). arXiv 2312.11805. URL: https:\/\/arxiv.org\/abs\/2312.11805"},{"key":"12_CR19","unstructured":"OpenAI. OpenAI o3 and o4-mini System Card. System Card. OpenAI (2025). Accessed 12 April 2025"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases. Research Track"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-06078-5_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T18:50:16Z","timestamp":1759171816000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-06078-5_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,30]]},"ISBN":["9783032060778","9783032060785"],"references-count":19,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-06078-5_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,30]]},"assertion":[{"value":"30 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECML PKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Porto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Portugal","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecml2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecmlpkdd.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}