{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,26]],"date-time":"2026-08-26T10:12:37Z","timestamp":1787739157034,"version":"build-2784847793"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,9]],"date-time":"2024-07-09T00:00:00Z","timestamp":1720483200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,9]]},"DOI":"10.1145\/3657604.3664693","type":"proceedings-article","created":{"date-parts":[[2024,7,15]],"date-time":"2024-07-15T15:49:38Z","timestamp":1721058578000},"page":"300-304","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":36,"title":["Can Large Language Models Make the Grade? An Empirical Study Evaluating LLMs Ability To Mark Short Answer Questions in K-12 Education"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-8850-067X","authenticated-orcid":false,"given":"Owen","family":"Henkel","sequence":"first","affiliation":[{"name":"University of Oxford, Oxford, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-7392-4711","authenticated-orcid":false,"given":"Libby","family":"Hills","sequence":"additional","affiliation":[{"name":"Jacobs Foundation, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9524-8037","authenticated-orcid":false,"given":"Adam","family":"Boxer","sequence":"additional","affiliation":[{"name":"Carousel Learning, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0508-1895","authenticated-orcid":false,"given":"Bill","family":"Roberts","sequence":"additional","affiliation":[{"name":"Legible Labs, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8932-1489","authenticated-orcid":false,"given":"Zach","family":"Levonian","sequence":"additional","affiliation":[{"name":"Digital Harbor Foundation, Baltimore, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,15]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.4074\/S0003503314004047"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.2307\/3315487"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1177\/014662169001400101"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11092-008--9068--5"},{"key":"e_1_3_2_1_5_1","volume-title":"et al","author":"Bommasani R.","year":"2022","unstructured":"Bommasani, R. et al. 2022. On the Opportunities and Risks of Foundation Models. arXiv."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1111\/jcal.12793"},{"key":"e_1_3_2_1_7_1","unstructured":"Brown T.B. Mann B. Ryder N. Subbiah M. Kaplan J. Dhariwal P. Neelakantan A. Shyam P. Sastry G. Askell A. Agarwal S. Herbert-Voss A. Krueger G. and Henighan T. 2020. Language Models are Few-Shot Learners. (2020)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40593-014-0026--8"},{"key":"e_1_3_2_1_9_1","unstructured":"Cain K. and Oakhill J. 2007. Children's comprehension problems in oral and written language a cognitive perspective. Guilford Press."},{"key":"e_1_3_2_1_10_1","volume-title":"Automated Summarization Evaluation (ASE) Using Natural Language Processing Tools. Artificial Intelligence in Education","author":"Crossley S.A.","unstructured":"Crossley, S.A., Kim, M., Allen, L. and McNamara, D. 2019. Automated Summarization Evaluation (ASE) Using Natural Language Processing Tools. Artificial Intelligence in Education. S. Isotani, E. Mill\u00e1n, A. Ogan, P. Hastings, B. McLaren, and R. Luckin, eds. Springer International Publishing. 84--95."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1222"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Fernandez N. Ghosh A. Liu N. Wang Z. Choffin B. Baraniuk R. and Lan A. 2023. Automated Scoring for Reading Comprehension via In-context BERT Tuning. arXiv.","DOI":"10.1007\/978-3-031-11644-5_69"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2305016120"},{"key":"e_1_3_2_1_14_1","volume-title":"Fill-In Problems: The Trade-off Between Scalability and Learning. Proceedings of the 14th Learning Analytics and Knowledge Conference (Kyoto Japan","author":"Gurung A.","year":"2024","unstructured":"Gurung, A., Vanacore, K., Mcreynolds, A.A., Ostrow, K.S., Worden, E., Sales, A.C. and Heffernan, N.T. 2024. Multiple Choice vs. Fill-In Problems: The Trade-off Between Scalability and Learning. Proceedings of the 14th Learning Analytics and Knowledge Conference (Kyoto Japan, Mar. 2024), 507--517."},{"key":"e_1_3_2_1_15_1","volume-title":"Visible learning: a synthesis of over 800 meta-analyses relating to achievement","author":"Hattie J.","unstructured":"Hattie, J. 2010. Visible learning: a synthesis of over 800 meta-analyses relating to achievement. Routledge."},{"key":"e_1_3_2_1_16_1","volume-title":"Supporting Foundational Literacy Assessment in LMICs: Can LLMs Grade Short-answer Reading Comprehension Questions?","author":"Henkel O.","year":"2023","unstructured":"Henkel, O., Hills, L., Roberts, B. and McGrane, J. 2023. Supporting Foundational Literacy Assessment in LMICs: Can LLMs Grade Short-answer Reading Comprehension Questions? (2023)."},{"key":"e_1_3_2_1_17_1","unstructured":"Kuzman T. Mozeti? I. and Ljube?i? N. 2023. ChatGPT: Beginning of an End of Manual Linguistic Data Annotation? Use Case of Automatic Genre Identification. arXiv."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.3758\/s13428-012-0211--3"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1207\/S1532690XCI2103_02"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.bea-1.15"},{"key":"e_1_3_2_1_21_1","unstructured":"Pearson P.D. and Hamm D.N. 2006. The Assessment of Reading Comprehension: A Review of Practices- Past Present and Future. Children's reading comprehension and assessment. Lawrence Erlbaum Associates."},{"key":"e_1_3_2_1_22_1","unstructured":"Perez E. Kiela D. and Cho K. 2021. True Few-Shot Learning with Language Models. (2021)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.3115\/1609829.1609831"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.3102\/0034654307313795"},{"key":"e_1_3_2_1_25_1","unstructured":"Zhao S. Li B. Reed C. Xu P. and Keutzer K. 2020. Multi-source Domain Adaptation in the Deep Learning Era: A Systematic Survey. arXiv."}],"event":{"name":"L@S '24: Eleventh ACM Conference on Learning @ Scale","location":"Atlanta GA USA","acronym":"L@S '24"},"container-title":["Proceedings of the Eleventh ACM Conference on Learning @ Scale"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3657604.3664693","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3657604.3664693","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T15:39:23Z","timestamp":1755877163000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3657604.3664693"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,9]]},"references-count":25,"alternative-id":["10.1145\/3657604.3664693","10.1145\/3657604"],"URL":"https:\/\/doi.org\/10.1145\/3657604.3664693","relation":{},"subject":[],"published":{"date-parts":[[2024,7,9]]},"assertion":[{"value":"2024-07-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}