{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T07:47:04Z","timestamp":1782546424734,"version":"3.54.5"},"publisher-location":"Cham","reference-count":31,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032297594","type":"print"},{"value":"9783032297600","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,6,28]],"date-time":"2026-06-28T00:00:00Z","timestamp":1782604800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,28]],"date-time":"2026-06-28T00:00:00Z","timestamp":1782604800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-29760-0_45","type":"book-chapter","created":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T07:08:25Z","timestamp":1782544105000},"page":"407-416","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Reliability as\u00a0a\u00a0Teammate: Budgeted Verifier-in-the-Loop (BVIL) Policies for\u00a0Reliable LLM Tutoring Actions"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-8879-7622","authenticated-orcid":false,"given":"Partha Sarathi","family":"Purkayastha","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,28]]},"reference":[{"key":"45_CR1","doi-asserted-by":"publisher","unstructured":"Aleven, V., McLaren, B.M., Sewall, J., Koedinger, K.R.: The cognitive tutor authoring tools (CTAT): preliminary evaluation of efficiency gains. In: Intelligent Tutoring Systems. LNCS, vol.\u00a04053, pp. 61\u201370. Springer (2006). https:\/\/doi.org\/10.1007\/11774303_7","DOI":"10.1007\/11774303_7"},{"issue":"1","key":"45_CR2","doi-asserted-by":"publisher","first-page":"224","DOI":"10.1007\/s40593-015-0088-2","volume":"26","author":"V Aleven","year":"2016","unstructured":"Aleven, V., McLaren, B.M., Sewall, J., et al.: Example-tracing tutors: intelligent tutor development for non-programmers. Int. J. Artif. Intell. Educ. 26(1), 224\u2013269 (2016). https:\/\/doi.org\/10.1007\/s40593-015-0088-2","journal-title":"Int. J. Artif. Intell. Educ."},{"issue":"1","key":"45_CR3","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1016\/0004-3702(90)90093-F","volume":"42","author":"JR Anderson","year":"1990","unstructured":"Anderson, J.R., Boyle, C.F., Corbett, A.T., Lewis, M.W.: Cognitive modeling and intelligent tutoring. Artif. Intell. 42(1), 7\u201349 (1990)","journal-title":"Artif. Intell."},{"key":"45_CR4","unstructured":"Brown, T.B..: Language models are few-shot learners (2020). arXiv:2005.14165 arXiv preprint"},{"key":"45_CR5","unstructured":"Cobbe, K., et al.: Training verifiers to solve math word problems (2021). arXiv:2110.14168 arXiv preprint"},{"issue":"4","key":"45_CR6","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1007\/BF01099821","volume":"4","author":"AT Corbett","year":"1994","unstructured":"Corbett, A.T., Anderson, J.R.: Knowledge tracing: modeling the acquisition of procedural knowledge. User Model. User-Adap. Inter. 4(4), 253\u2013278 (1994)","journal-title":"User Model. User-Adap. Inter."},{"key":"45_CR7","doi-asserted-by":"publisher","unstructured":"Dhuliawala, S., Komeili, M., Xu, J.: Chain-of-verification reduces hallucination in large language models. In: Findings of the Association for Computational Linguistics: ACL 2024, pp. 1895\u20131914. Association for Computational Linguistics (2024). https:\/\/doi.org\/10.18653\/v1\/2024.findings-acl.111","DOI":"10.18653\/v1\/2024.findings-acl.111"},{"key":"45_CR8","unstructured":"Gou, Z., Shao, Z., Gong, Y.: CRITIC: Large language models can self-correct with tool-interactive critiquing (2023). arXiv:2305.11738 arXiv preprint"},{"issue":"2","key":"45_CR9","doi-asserted-by":"publisher","first-page":"180","DOI":"10.3758\/BF03195563","volume":"36","author":"AC Graesser","year":"2004","unstructured":"Graesser, A.C., Lu, S., Jackson, G.T., et al.: AutoTutor: a tutor with dialogue in natural language. Behav. Res. Meth. Instr. Comput 36(2), 180\u2013192 (2004). https:\/\/doi.org\/10.3758\/BF03195563","journal-title":"Behav. Res. Meth. Instr. Comput"},{"key":"45_CR10","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1007\/s44217-025-00424-7","volume":"4","author":"S Guizani","year":"2025","unstructured":"Guizani, S., Mazhar, T., Shahzad, T., Ahmad, W., Bibi, A., Hamam, H., et al.: A systematic literature review to implement large language model in higher education: issues and solutions. Discover Educ. 4, 35 (2025). https:\/\/doi.org\/10.1007\/s44217-025-00424-7","journal-title":"Discover Educ."},{"key":"45_CR11","unstructured":"Huang, J.: Large language models cannot self-correct reasoning yet. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"45_CR12","doi-asserted-by":"crossref","unstructured":"Kotalwar, N., Gotovos, A., Singla, A.: Hints-in-browser: benchmarking language models for programming feedback generation. In: Advances in Neural Information Processing Systems, vol. 37, pp. 29864\u201329877. Curran Associates, Inc (2024)","DOI":"10.52202\/079017-0940"},{"key":"45_CR13","doi-asserted-by":"crossref","unstructured":"Koutcheme, C., Woodrow, J., Piech, C.: Aligning small language models for programming feedback: towards scalable coding support in a massive global course. In: Proceedings of the 57th ACM Technical Symposium on Computer Science Education V. 1, pp. 610\u2013616. Association for Computing Machinery (2026). 10.1145\/3770762.3772539","DOI":"10.1145\/3770762.3772539"},{"key":"45_CR14","unstructured":"Lewis, P., Perez, E., Piktus, A.: Retrieval-augmented generation for knowledge-intensive NLP tasks (2020). arXiv:2005.11401 arXiv preprint"},{"key":"45_CR15","unstructured":"MacLellan, C.J.: Closing the loop between learning theory and educational data: a computational theory of learning with the Apprentice Learner architecture. Ph.D. Thesis, University of Pittsburgh (2016)"},{"issue":"1","key":"45_CR16","doi-asserted-by":"publisher","first-page":"76","DOI":"10.1007\/s40593-020-00214-2","volume":"32","author":"CJ MacLellan","year":"2022","unstructured":"MacLellan, C.J., Koedinger, K.R.: Domain-general tutor authoring with apprentice learner models. Int. J. Artif. Intell. Educ. 32(1), 76\u2013117 (2022). https:\/\/doi.org\/10.1007\/s40593-020-00214-2","journal-title":"Int. J. Artif. Intell. Educ."},{"key":"45_CR17","doi-asserted-by":"crossref","unstructured":"Madaan, A., et al.: Self-refine: iterative refinement with self-feedback (2023). arXiv:2303.17651 arXiv preprint","DOI":"10.52202\/075280-2019"},{"key":"45_CR18","doi-asserted-by":"publisher","unstructured":"Manakul, P., Liusie, A., Gales, M.J.F.: SelfCheckGPT: zero-resource black-box hallucination detection for generative large language models. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 9004\u20139031. Association for Computational Linguistics (2023). https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.557","DOI":"10.18653\/v1\/2023.emnlp-main.557"},{"key":"45_CR19","unstructured":"Ouyang, L., et al.: Training language models to follow instructions with human feedback (2022). arXiv:2203.02155 arXiv preprint"},{"key":"45_CR20","doi-asserted-by":"crossref","unstructured":"Pardos, Z.A., Tang, M., Anastasopoulos, I., et\u00a0al.: OATutor: an open-source adaptive tutoring system and curated content library for learning sciences research. In: Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems, pp. 1\u201317. Association for Computing Machinery (2023). 10.1145\/3544548.3581574","DOI":"10.1145\/3544548.3581574"},{"key":"45_CR21","doi-asserted-by":"publisher","unstructured":"Park, G., Song, J., Choi, G.: K-NLPers at BEA 2025 shared task: evaluating the quality of AI tutor responses with GPT-4.1. In: Proceedings of the 20th Workshop on Innovative Use of NLP for Building Educational Applications, pp. 1145\u20131163. Association for Computational Linguistics (2025). https:\/\/doi.org\/10.18653\/v1\/2025.bea-1.90","DOI":"10.18653\/v1\/2025.bea-1.90"},{"key":"45_CR22","doi-asserted-by":"crossref","unstructured":"Pathak, A., Gandhi, R., Uttam, V.: Rubric is all you need: improving LLM-based code evaluation with question-specific rubrics. In: Proceedings of the 2025 ACM Conference on International Computing Education Research, pp. 181\u2013195. Association for Computing Machinery (2025)","DOI":"10.1145\/3702652.3744220"},{"key":"45_CR23","unstructured":"Qwen Team: Qwen3 technical report (2025). arXiv:2505.09388 arXiv preprint"},{"key":"45_CR24","doi-asserted-by":"crossref","unstructured":"Schick, T., Dwivedi-Yu, J., Dessi, R.: ToolFormer: language models can teach themselves to use tools (2023). arXiv:2302.04761 arXiv preprint","DOI":"10.52202\/075280-2997"},{"key":"45_CR25","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2025.100529","volume":"10","author":"Y Shi","year":"2026","unstructured":"Shi, Y., Yu, K., Dong, Y., Chen, F.: Large language models in education: a systematic review of empirical applications, benefits, and challenges. Comput. Educ. Artif. Intell 10, 100529 (2026). https:\/\/doi.org\/10.1016\/j.caeai.2025.100529","journal-title":"Comput. Educ. Artif. Intell"},{"key":"45_CR26","doi-asserted-by":"crossref","unstructured":"Shinn, N., Cassano, F., Berman, E.: Reflexion: language agents with verbal reinforcement learning (2023). arXiv:2303.11366 arXiv preprint","DOI":"10.52202\/075280-0377"},{"issue":"4","key":"45_CR27","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1080\/00461520.2011.611369","volume":"46","author":"K VanLehn","year":"2011","unstructured":"VanLehn, K.: The relative effectiveness of human tutoring, intelligent tutoring systems, and other tutoring systems. Educat. Psychol. 46(4), 197\u2013221 (2011). https:\/\/doi.org\/10.1080\/00461520.2011.611369","journal-title":"Educat. Psychol."},{"key":"45_CR28","unstructured":"Wang, X., Wei, J., Schuurmans, D.: Self-consistency improves chain of thought reasoning in language models (2022). arXiv:2203.11171 arXiv preprint"},{"key":"45_CR29","doi-asserted-by":"crossref","unstructured":"Wei, J., Wang, X., Schuurmans, D.: Chain-of-thought prompting elicits reasoning in large language models (2022). arXiv:2201.11903 arXiv preprint","DOI":"10.52202\/068431-1800"},{"key":"45_CR30","doi-asserted-by":"crossref","unstructured":"Weitekamp, D., Siddiqui, M.N., MacLellan, C.J.: TutorGYM: a testbed for evaluating AI agents as tutors and students. arXiv preprint arXiv:2505.01563 (2025)","DOI":"10.1007\/978-3-031-98420-4_26"},{"key":"45_CR31","doi-asserted-by":"crossref","unstructured":"Zeng, Z., Wang, J., Yang, J., et\u00a0al.: PrivacyRestore: privacy-preserving inference in large language models via privacy removal and restoration. In: Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics, (vol. 1: Long Papers), pp. 10821\u201310855. Association for Computational Linguistics, Vienna, Austria (2025). 10.18653\/v1\/2025.acl-long.532","DOI":"10.18653\/v1\/2025.acl-long.532"}],"container-title":["Lecture Notes in Computer Science","Artificial Intelligence in Education"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-29760-0_45","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T07:08:36Z","timestamp":1782544116000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-29760-0_45"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,28]]},"ISBN":["9783032297594","9783032297600"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-29760-0_45","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,28]]},"assertion":[{"value":"28 June 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"AIED","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Intelligence in Education","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Seoul","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Korea (Republic of)","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 June 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"aied2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.aied-conference.org\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}