{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T21:34:30Z","timestamp":1777152870514,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","funder":[{"name":"Xi'an Jiaotong-Liverpool University","award":["TDF22\/23-R25-185"],"award-info":[{"award-number":["TDF22\/23-R25-185"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,17]]},"DOI":"10.1145\/3698205.3729551","type":"proceedings-article","created":{"date-parts":[[2025,7,17]],"date-time":"2025-07-17T14:35:08Z","timestamp":1752762908000},"page":"105-115","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["LLMarking: Adaptive Automatic Short-Answer Grading Using Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-5778-4007","authenticated-orcid":false,"given":"Hanling","family":"Wang","sequence":"first","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6120-8064","authenticated-orcid":false,"given":"Banghao","family":"Chi","sequence":"additional","affiliation":[{"name":"College of Liberal Arts &amp; Sciences, University of Illinois at Urbana-Champagne, Urbana, IL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2847-9516","authenticated-orcid":false,"given":"Yufei","family":"Wu","sequence":"additional","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2257-2010","authenticated-orcid":false,"given":"Kexin","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1463-6147","authenticated-orcid":false,"given":"Di","family":"Wu","sequence":"additional","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1086-9463","authenticated-orcid":false,"given":"Songning","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2181-9941","authenticated-orcid":false,"given":"Yiwei","family":"Li","sequence":"additional","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8207-9929","authenticated-orcid":false,"given":"Hanyan","family":"Niu","sequence":"additional","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1024-5442","authenticated-orcid":false,"given":"Xiaohui","family":"Zhu","sequence":"additional","affiliation":[{"name":"School of Advanced Technology, Xi'an Jiaotong-Liverpool University, Suzhou, Jiangsu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,17]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Automated Short Answer Grading Using Deep Learning: A Survey","author":"Bonthu Sridevi","unstructured":"Sridevi Bonthu, S. Rama Sree, and M. H. M. Krishna Prasad. 2021. Automated Short Answer Grading Using Deep Learning: A Survey. In Machine Learning and Knowledge Extraction, Andreas Holzinger, Peter Kieseberg, A. Min Tjoa, and Edgar Weippl (Eds.). Springer International Publishing, Cham, 61-78."},{"key":"e_1_3_2_2_2_1","unstructured":"Tom B. Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell Sandhini Agarwal Ariel Herbert-Voss Gretchen Krueger Tom Henighan Rewon Child Aditya Ramesh Daniel M. Ziegler Jeffrey Wu Clemens Winter Christopher Hesse Mark Chen Eric Sigler Mateusz Litwin Scott Gray Benjamin Chess Jack Clark Christopher Berner Sam McCandlish Alec Radford Ilya Sutskever and Dario Amodei. 2020. Language Models are Few-Shot Learners. arXiv:2005.14165 [cs.CL] https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40593-014-0026-8"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i21.30363"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1053"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2014.09.005"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10115-023-01892-9"},{"key":"e_1_3_2_2_8_1","volume-title":"Human: Rethinking Automated Assessment with Large Language Models. arXiv preprint arXiv:2405.19694","author":"Gobbo George Del","year":"2023","unstructured":"George Del Gobbo et al. 2023. Grade Like a Human: Rethinking Automated Assessment with Large Language Models. arXiv preprint arXiv:2405.19694 (2023). Available at arXiv: https:\/\/arxiv.org\/abs\/2405.19694."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-03928-8_31"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"V. Hackl A. E. M\u00fcller M. Granitzer and M. Sailer. 2023. Is GPT-4 a Reliable Rater? Evaluating Consistency in GPT-4 Text Ratings. arXiv preprint (2023). arXiv:2308.02575 [cs.CL]","DOI":"10.3389\/feduc.2023.1272229"},{"key":"e_1_3_2_2_11_1","volume-title":"Conferences in Research and Practice in Information Technology Series 66 (01","author":"Haley Debra","year":"2007","unstructured":"Debra Haley, Pete Thomas, Anne De Roeck, and Marian Petre. 2007. Measuring improvement in latent semantic analysis-based marking systems: Using a computer to mark questions about HTML. Conferences in Research and Practice in Information Technology Series 66 (01 2007)."},{"key":"e_1_3_2_2_12_1","first-page":"13","volume-title":"ETS: Domain Adaptation and Stacking for Short Answer Scoring. In Second Joint Conference on Lexical and Computational Semantics (*SEM)","volume":"279","author":"Heilman Michael","year":"2013","unstructured":"Michael Heilman and Nitin Madnani. 2013. ETS: Domain Adaptation and Stacking for Short Answer Scoring. In Second Joint Conference on Lexical and Computational Semantics (*SEM), Volume 2: Proceedings of the Seventh International Workshop on Semantic Evaluation (SemEval 2013), Suresh Manandhar and Deniz Yuret (Eds.). Association for Computational Linguistics, Atlanta, Georgia, USA, 275-279. https:\/\/aclanthology.org\/S13-2046"},{"key":"e_1_3_2_2_13_1","unstructured":"Fan Huang Haewoon Kwak and Jisun An. 2024. Token-Ensemble Text Generation: On Attacking the Automatic AI-Generated Text Detection. arXiv:2402.11167 [cs.CL] https:\/\/arxiv.org\/abs\/2402.11167"},{"key":"e_1_3_2_2_14_1","volume-title":"Gilpin","author":"Huang Shiyuan","year":"2023","unstructured":"Shiyuan Huang, Siddarth Mamidanna, Shreedhar Jangam, Yilun Zhou, and Leilani H. Gilpin. 2023. Can Large Language Models Explain Themselves? A Study of LLM-Generated Self-Explanations. arXiv:2310.11207 [cs.CL] https:\/\/arxiv.org\/abs\/2310.11207"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s44163-024-00147-y"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","unstructured":"Sachin Kumar Soumen Chakrabarti and Shourya Roy. 2017. Earth Mover's Distance Pooling over Siamese LSTMs for Automatic Short Answer Grading. 2046-2052. doi:10.24963\/ijcai.2017\/284","DOI":"10.24963\/ijcai.2017\/284"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_2_2_18_1","volume-title":"Generating with confidence: Uncertainty quantification for black-box large language models. arXiv preprint arXiv:2305.19187","author":"Lin Zhen","year":"2023","unstructured":"Zhen Lin, Shubhendu Trivedi, and Jimeng Sun. 2023. Generating with confidence: Uncertainty quantification for black-box large language models. arXiv preprint arXiv:2305.19187 (2023)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i09.7062"},{"key":"e_1_3_2_2_20_1","unstructured":"Ahmed Magooda Mohamed Zahran Mohsen Rashwan Hazem Raafat and Magda Fayek. 2016. Vector Based Techniques for Short Answer Grading."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271755"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10350"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Ethan Perez Saffron Huang Francis Song Trevor Cai Roman Ring John Aslanides Amelia Glaese Nat McAleese and Geoffrey Irving. 2022. Red Teaming Language Models with Language Models. arXiv:2202.03286 [cs.CL] https:\/\/arxiv.org\/abs\/2202.03286","DOI":"10.18653\/v1\/2022.emnlp-main.225"},{"key":"e_1_3_2_2_24_1","unstructured":"Stephen Pulman. 2005. Information Extraction and Machine Learning: Automarking Short Free Text Responses to Science Questions. (2005)."},{"key":"e_1_3_2_2_25_1","first-page":"9 05","volume-title":"Proceedings of the SecondWorkshop on Building Educational Applications Using NLP, Jill Burstein and Claudia Leacock (Eds.). Association for Computational Linguistics","author":"Stephen","unstructured":"Stephen G. Pulman and Jana Z. Sukkarieh. 2005. Automatic Short Answer Marking. In Proceedings of the SecondWorkshop on Building Educational Applications Using NLP, Jill Burstein and Claudia Leacock (Eds.). Association for Computational Linguistics, Ann Arbor, Michigan, 9-16. https:\/\/aclanthology.org\/W05-0202"},{"key":"e_1_3_2_2_26_1","unstructured":"Kamesh R. 2024. Think Beyond Size: Adaptive Prompting for More Effective Reasoning. arXiv:2410.08130 [cs.LG] https:\/\/arxiv.org\/abs\/2410.08130"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/SCSE61872.2024.10550624"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2020.02.171"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-6119"},{"key":"e_1_3_2_2_30_1","volume-title":"Context Injection Attacks on Large Language Models. arXiv preprint arXiv:2405.20234","author":"Wei Cheng'an","year":"2024","unstructured":"Cheng'an Wei, Kai Chen, Yue Zhao, Yujia Gong, Lu Xiang, and Shenchen Zhu. 2024. Context Injection Attacks on Large Language Models. arXiv preprint arXiv:2405.20234 (2024)."},{"key":"e_1_3_2_2_31_1","volume-title":"Dynamic Prompting: A Unified Framework for Prompt Tuning. arXiv:2303.02909 [cs.CL] https:\/\/arxiv.org\/abs\/2303.02909","author":"Yang Xianjun","year":"2023","unstructured":"Xianjun Yang,Wei Cheng, Xujiang Zhao,Wenchao Yu, Linda Petzold, and Haifeng Chen. 2023. Dynamic Prompting: A Unified Framework for Prompt Tuning. arXiv:2303.02909 [cs.CL] https:\/\/arxiv.org\/abs\/2303.02909"},{"key":"e_1_3_2_2_32_1","volume-title":"Short Answer Grading Using Oneshot Prompting and Text Similarity Scoring Model. arXiv","author":"Yoon S.-Y.","year":"2023","unstructured":"S.-Y. Yoon. 2023. Short Answer Grading Using Oneshot Prompting and Text Similarity Scoring Model. arXiv (2023). arXiv:2305.18638 [cs.CL]"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Haifeng Zhao Yuguang Jin and Leilei Ma. 2025. Dynamic Prompt Adjustment for Multi-Label Class-Incremental Learning. arXiv:2501.00340 [cs.CV] https:\/\/arxiv.org\/abs\/2501.00340","DOI":"10.1007\/978-981-96-2882-7_6"},{"key":"e_1_3_2_2_34_1","unstructured":"Andy Zou Zifan Wang Nicholas Carlini Milad Nasr J. Zico Kolter and Matt Fredrikson. 2023. Universal and Transferable Adversarial Attacks on Aligned Language Models. arXiv:2307.15043 [cs.CL] https:\/\/arxiv.org\/abs\/2307.15043"}],"event":{"name":"L@S '25: Twelfth ACM Conference on Learning @ Scale","location":"Palermo Italy","acronym":"L@S '25"},"container-title":["Proceedings of the Twelfth ACM Conference on Learning @ Scale"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3698205.3729551","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T13:15:41Z","timestamp":1755868541000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3698205.3729551"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,17]]},"references-count":34,"alternative-id":["10.1145\/3698205.3729551","10.1145\/3698205"],"URL":"https:\/\/doi.org\/10.1145\/3698205.3729551","relation":{},"subject":[],"published":{"date-parts":[[2025,7,17]]},"assertion":[{"value":"2025-07-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}