{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T05:14:31Z","timestamp":1783401271352,"version":"3.54.6"},"publisher-location":"Cham","reference-count":31,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032308597","type":"print"},{"value":"9783032308603","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-30860-3_11","type":"book-chapter","created":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T04:21:25Z","timestamp":1783398085000},"page":"155-171","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Representation Robustness Under Executable Reasoning Constraints in Large Language Models for Mathematical Problem Solving"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5225-952X","authenticated-orcid":false,"given":"Sagnik","family":"Nath","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7650-477X","authenticated-orcid":false,"given":"Edith Aurora","family":"Graf","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0017-2569","authenticated-orcid":false,"given":"Liang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0620-7622","authenticated-orcid":false,"given":"Diego","family":"Zapata-Rivera","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,7]]},"reference":[{"issue":"2","key":"11_CR1","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1007\/s44366-025-0059-6","volume":"2","author":"Q Zhu","year":"2025","unstructured":"Zhu, Q., Wang, M., Zhang, T.H.H.: Current trends and future prospects of large-scale foundation model in K-12 education. Front. Digit. Educ. 2(2), 22 (2025)","journal-title":"Front. Digit. Educ."},{"key":"11_CR2","doi-asserted-by":"crossref","unstructured":"Ahn, J., Verma, R., Lou, R., Liu, D., Zhang, R., Yin, W.: Large language models for mathematical reasoning: progresses and challenges (2024). arXiv:2402.00157","DOI":"10.18653\/v1\/2024.eacl-srw.17"},{"key":"11_CR3","unstructured":"Mirzadeh, I., Alizadeh, K., Shahrokhi, H., Tuzel, O., Bengio, S., Farajtabar, M.: GSM-symbolic: understanding the limitations of mathematical reasoning in large language models (2024). arXiv:2410.05229"},{"key":"11_CR4","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Zhu, Y., Antognini, D., Kim, Y., Zhang, Y.: Paraphrase and solve: exploring and exploiting the impact of surface form on mathematical reasoning in large language models (2024). arXiv:2404.11500","DOI":"10.18653\/v1\/2024.naacl-long.153"},{"issue":"2","key":"11_CR5","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1207\/s15430421tip4002_6","volume":"40","author":"SJ Pape","year":"2001","unstructured":"Pape, S.J., Tchoshanov, M.A.: The role of representation(s) in developing mathematical understanding. Theory Pract. 40(2), 118\u2013127 (2001)","journal-title":"Theory Pract."},{"key":"11_CR6","doi-asserted-by":"crossref","unstructured":"Arora, D., Singh, H.: Have llms advanced enough? a challenging problem solving benchmark for large language models. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.468"},{"issue":"2","key":"11_CR7","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1007\/BF00132515","volume":"10","author":"R Mayer","year":"1981","unstructured":"Mayer, R.: Frequency norms and structural analysis of algebra story problems into families, categories, and templates. Instr. Sci. 10(2), 135\u2013175 (1981)","journal-title":"Instr. Sci."},{"issue":"2","key":"11_CR8","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1207\/s15327809jls1302_1","volume":"13","author":"K Koedinger","year":"2004","unstructured":"Koedinger, K., Nathan, M.: The real story behind story problems: effects of representations on quantitative reasoning. J. Learn. Sci. 13(2), 129\u2013164 (2004)","journal-title":"J. Learn. Sci."},{"key":"11_CR9","doi-asserted-by":"crossref","unstructured":"Lewkowycz, A., et al.: Solving quantitative reasoning problems with language models. In: Advances in Neural Information Processing Systems (2022)","DOI":"10.52202\/068431-0278"},{"key":"11_CR10","doi-asserted-by":"crossref","unstructured":"Saba, W.: Stochastic LLMs do not understand language: towards symbolic, explainable and ontologically based LLMs. In: International Conference on Conceptual Modeling (2023)","DOI":"10.1007\/978-3-031-47262-6_1"},{"key":"11_CR11","unstructured":"Gao, L., et al.: PAL: program-aided language models. In: Proceedings of Machine Learning Research (2023)"},{"key":"11_CR12","doi-asserted-by":"crossref","unstructured":"Schick, T.: Language models can teach themselves to use tools. In: Advances in Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-2997"},{"key":"11_CR13","doi-asserted-by":"crossref","unstructured":"Steinbach, M., Bhandari, S., Meyer, J., Pardos, Z.: When LLMs hallucinate: Examining the effects of erroneous feedback in math tutoring systems. In: Proceedings of the Twelfth ACM Conference on Learning@ Scale (2025)","DOI":"10.1145\/3698205.3729555"},{"key":"11_CR14","doi-asserted-by":"crossref","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. In: Advances in Neural Information Processing Systems (2022)","DOI":"10.52202\/068431-1800"},{"key":"11_CR15","unstructured":"HCI 2026 Dataset. https:\/\/github.com\/sagniknath91\/HCI-2026-dataset\/blob\/main\/datasheet.csv"},{"issue":"2","key":"11_CR16","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1016\/0010-0285(76)90022-0","volume":"8","author":"H Simon","year":"1976","unstructured":"Simon, H., Hayes, J.: The understanding process: problem isomorphs. Cogn. Psychol. 8(2), 165\u2013190 (1976)","journal-title":"Cogn. Psychol."},{"key":"11_CR17","unstructured":"Bejar, I.I.: Generative testing: from conception to implementation. In: Irvine, S.H., Kyllonen, P.C. (eds.) Item Generation for Test Development, pp. 199\u2013218. Lawrence Erlbaum Associates, Mahwah, NJ (2002)"},{"issue":"1","key":"11_CR18","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1016\/S0732-3123(99)80058-3","volume":"17","author":"B Greer","year":"1998","unstructured":"Greer, B., Harel, G.: The role of isomorphisms in mathematical cognition. J. Math. Behav. 17(1), 5\u201324 (1998)","journal-title":"J. Math. Behav."},{"key":"11_CR19","unstructured":"Maher, C.: The longitudinal study. In: Combinatorics and Reasoning: Representing, Justifying and Building Isomorphisms, pp. 3\u20138. Springer Netherlands, Dordrecht (2010)"},{"key":"11_CR20","unstructured":"Glazer, E., et al.: Frontiermath: a benchmark for evaluating advanced mathematical reasoning in AI (2024). arXiv:2411.04872"},{"key":"11_CR21","unstructured":"Wei, J., et al.: Emergent abilities of large language models (2022). arXiv:2206.07682"},{"key":"11_CR22","unstructured":"Kaplan, J., et al.: Scaling laws for neural language models (2020). arXiv:2001.08361"},{"key":"11_CR23","doi-asserted-by":"crossref","unstructured":"Trevi\u00f1o, E., Contant, H., Ngai, J., Neubig, G., Wang, Z.: Benchmarking failures in tool-augmented language models (2025). arXiv:2503.14227","DOI":"10.18653\/v1\/2025.naacl-long.149"},{"key":"11_CR24","doi-asserted-by":"crossref","unstructured":"Abdelkarim, S., Lu, D., Flores, D., Jaeggi, S., Baldi, P.: Evaluating the intelligence of large language models: a comparative study using verbal and visual IQ tests. In: Computers in Human Behavior: Artificial Humans, p. 100170 (2025)","DOI":"10.1016\/j.chbah.2025.100170"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Nath, S., Yoon, S.: WIP: beyond code: evaluating ChatGPT, Gemini, Claude, and Meta AI as AI tutors in computer science and engineering education. In: IEEE Frontiers in Education Conference (FIE) (2024)","DOI":"10.1109\/FIE61694.2024.10893528"},{"key":"11_CR26","unstructured":"OpenRouter. https:\/\/openrouter.ai\/. Accessed 20 January 2026"},{"issue":"2","key":"11_CR27","doi-asserted-by":"publisher","first-page":"152","DOI":"10.5395\/rde.2017.42.2.152","volume":"42","author":"H Kim","year":"2017","unstructured":"Kim, H.: Statistical notes for clinical researchers: Chi-squared test and Fisher\u2019s exact test. Restor. Dentist. Endodont. 42(2), 152 (2017)","journal-title":"Restor. Dentist. Endodont."},{"key":"11_CR28","unstructured":"Structured Outputs. Gemini API. https:\/\/ai.google.dev\/gemini-api\/docs\/structured-output?example=recipe"},{"key":"11_CR29","unstructured":"Steiner, T.: Structured Output Support for the Prompt API. https:\/\/developer.chrome.com\/docs\/ai\/structured-output-for-prompt-api"},{"key":"11_CR30","unstructured":"Castillo, D.: The Good, the Bad, and the Ugly of Gemini\u2019s Structured Outputs. https:\/\/dylancastillo.co\/posts\/gemini-structured-outputs.html"},{"key":"11_CR31","doi-asserted-by":"crossref","unstructured":"Wu, Z., et al.: Reasoning or reciting? Exploring the capabilities and limitations of language models through counterfactual tasks. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (2024)","DOI":"10.18653\/v1\/2024.naacl-long.102"}],"container-title":["Lecture Notes in Computer Science","Artificial Intelligence in HCI"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-30860-3_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T04:22:05Z","timestamp":1783398125000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-30860-3_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032308597","9783032308603"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-30860-3_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"7 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"HCII","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Human-Computer Interaction","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Montreal, QC","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"hcii2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2026.hci.international\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}