{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T12:08:04Z","timestamp":1783166884988,"version":"3.54.6"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,3,21]],"date-time":"2025-03-21T00:00:00Z","timestamp":1742515200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,21]],"date-time":"2025-03-21T00:00:00Z","timestamp":1742515200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2101104"],"award-info":[{"award-number":["2101104"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["213884"],"award-info":[{"award-number":["213884"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2223768"],"award-info":[{"award-number":["2223768"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Tech Know Learn"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s10758-025-09836-8","type":"journal-article","created":{"date-parts":[[2025,3,21]],"date-time":"2025-03-21T19:46:25Z","timestamp":1742586385000},"page":"669-684","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":31,"title":["Unveiling Scoring Processes: Dissecting the Differences Between LLMs and Human Graders in Automatic Scoring"],"prefix":"10.1007","volume":"31","author":[{"given":"Xuansheng","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Padmaja Pravin","family":"Saraf","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gyeonggeon","family":"Lee","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ehsan","family":"Latif","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ninghao","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4519-1931","authenticated-orcid":false,"given":"Xiaoming","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,21]]},"reference":[{"key":"9836_CR1","volume":"5","author":"A Bewersdorff","year":"2023","unstructured":"Bewersdorff, A., Se\u00dfler, K., Baur, A., Kasneci, E., & Nerdel, C. (2023). Assessing student errors in experimentation using artificial intelligence and large language models: A comparative study with human raters. Computers and Education: Artificial Intelligence, 5, 100177.","journal-title":"Computers and Education: Artificial Intelligence"},{"key":"9836_CR2","unstructured":"Brown, T.B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G. & Askell, A. (2020). Language models are few-shot learners. arXiv preprint arXiv:2005.14165."},{"key":"9836_CR3","doi-asserted-by":"crossref","unstructured":"Cohn, C., Hutchins, N., Le, T. & Biswas, G. (2024). A chain-of-thought prompting approach with llms for evaluating students\u2019 formative assessment responses in science. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 23182\u201323190.","DOI":"10.1609\/aaai.v38i21.30364"},{"key":"9836_CR4","doi-asserted-by":"crossref","unstructured":"Du, M., He, F., Zou, N., Tao, D. & Hu, X. (2023). Shortcut learning of large language models in natural language understanding.","DOI":"10.1145\/3596490"},{"key":"9836_CR5","unstructured":"Google Deepmind. (2024). Gemini 1.5: Unlocking Multimodal Understanding Across Millions of Tokens of Context. Google Deepmind"},{"issue":"4","key":"9836_CR6","doi-asserted-by":"publisher","first-page":"612","DOI":"10.1109\/TE.2005.856149","volume":"48","author":"AC Graesser","year":"2005","unstructured":"Graesser, A. C., Chipman, P., Haynes, B. C., & Olney, A. (2005). Autotutor: An intelligent tutoring system with mixed-initiative dialogue. IEEE Transactions on Education, 48(4), 612\u2013618.","journal-title":"IEEE Transactions on Education"},{"key":"9836_CR7","unstructured":"Guo, S., Latif, E., Zhou, Y., Huang, X. & Zhai, X. (2024). Using generative ai and multi-agents to provide automatic feedback. arXiv preprint arXiv:2411.07407."},{"key":"9836_CR8","unstructured":"Han, J., Yoo, H., Myung, J., Kim, M., Lim, H., Kim, Y., Lee, T.Y., Hong, H., Kim, J., Ahn, S.-Y., Oh, A. (2023). Fabric: Automated scoring and feedback generation for essays. arXiv preprint arXiv:2310.05191."},{"key":"9836_CR9","doi-asserted-by":"crossref","unstructured":"Haque, S., Eberhart, Z., Bansal, A. & McMillan, C. (2022). Semantic similarity metrics for evaluating source code summarization. In: Proceedings of the 30th IEEE\/ACM International Conference on Program Comprehension, pp. 36\u201347.","DOI":"10.1145\/3524610.3527909"},{"key":"9836_CR10","unstructured":"Harris, C.J., Krajcik, J.S. & Pellegrino, J.W. (2024). Creating and Using Instructionally Supportive Assessments in NGSS Classrooms. NSTA Press, National Science Teaching Association. ???"},{"key":"9836_CR11","unstructured":"Jiang, A.Q., Sablayrolles, A., Roux, A., Mensch, A., Savary, B., Bamford, C., Chaplot, D.S., Casas, D.d.l., Hanna, E.B., Bressand, F., Lengyel, G. (2024). Mixtral of experts. arXiv preprint arXiv:2401.04088."},{"key":"9836_CR12","doi-asserted-by":"publisher","first-page":"22199","DOI":"10.52202\/068431-1613","volume":"35","author":"T Kojima","year":"2022","unstructured":"Kojima, T., Gu, S. S., Reid, M., Matsuo, Y., & Iwasawa, Y. (2022). Large language models are zero-shot reasoners. Advances in Neural Information Processing Systems, 35, 22199\u201322213.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"9836_CR13","unstructured":"Latif, E., Mai, G., Nyaaba, M., Wu, X., Liu, N., Lu, G., Li, S., Liu, T. & Zhai, X. (2023). Artificial general intelligence (agi) for education. arXiv preprint arXiv:2304.12479 1."},{"key":"9836_CR14","volume":"6","author":"E Latif","year":"2024","unstructured":"Latif, E., & Zhai, X. (2024). Fine-tuning chatgpt for automatic scoring. Computers and Education: Artificial Intelligence, 6, 100210.","journal-title":"Computers and Education: Artificial Intelligence"},{"key":"9836_CR15","unstructured":"Martin, H.T., Stone, K. & Albert, P. (2023). Llama 2: Open foundation and fine-tuned chat models. arXiv:2307.09288."},{"key":"9836_CR16","unstructured":"OpenAI. (2023). Gpt-4 technical report. arXiv:2303.08774."},{"key":"9836_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.tsc.2023.101356","volume":"49","author":"P Organisciak","year":"2023","unstructured":"Organisciak, P., Acar, S., Dumas, D., & Berthiaume, K. (2023). Beyond semantic distance: Automated scoring of divergent thinking greatly improves with large language models. Thinking Skills and Creativity, 49, 101356.","journal-title":"Thinking Skills and Creativity"},{"key":"9836_CR18","doi-asserted-by":"crossref","unstructured":"Parikh, D., Lu, Y., Xin, Y., Wu, D., Pelz, J. & Lu, G. (2019). Where am i looking: Localizing gaze in reconstructed 3d space. In: 2019 IEEE Global Conference on Signal and Information Processing (GlobalSIP), pp. 1\u20135. IEEE","DOI":"10.1109\/GlobalSIP45357.2019.8969158"},{"key":"9836_CR19","doi-asserted-by":"crossref","unstructured":"Reimers, N. & Gurevych, I. (2019). Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084.","DOI":"10.18653\/v1\/D19-1410"},{"key":"9836_CR20","doi-asserted-by":"crossref","unstructured":"Tang, R., Kong, D., Huang, L. & Xue, H. (2023). Large language models can be lazy learners: Analyze shortcuts in in-context learning. In: Findings of the Association for Computational Linguistics: ACL 2023, pp. 4645\u20134657.","DOI":"10.18653\/v1\/2023.findings-acl.284"},{"key":"9836_CR21","doi-asserted-by":"crossref","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E., Le, Q. & Zhou, D. (2022). Chain-of-thought prompting elicits reasoning in large language models. arXiv:2201.11903.","DOI":"10.52202\/068431-1800"},{"key":"9836_CR22","doi-asserted-by":"publisher","first-page":"24824","DOI":"10.52202\/068431-1800","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q. V., & Zhou, D. (2022). Chain-of-thought prompting elicits reasoning in large language models. Advances in Neural Information Processing Systems, 35, 24824\u201324837.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"1","key":"9836_CR23","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1002\/tea.21864","volume":"61","author":"CD Wilson","year":"2024","unstructured":"Wilson, C. D., Haudek, K. C., Osborne, J. F., Buck Bracey, Z. E., Cheuk, T., Donovan, B. M., Stuhlsatz, M. A., Santiago, M. M., & Zhai, X. (2024). Using automated analysis to assess middle school students\u2019 competence with scientific argumentation. Journal of Research in Science Teaching, 61(1), 38\u201369.","journal-title":"Journal of Research in Science Teaching"},{"key":"9836_CR24","doi-asserted-by":"crossref","unstructured":"Wu, X., He, X., Liu, T., Liu, N. & Zhai, X. (2023). Matching exemplar as next sentence prediction (mensp): Zero-shot prompt learning for automatic scoring in science education. arXiv:2301.08771.","DOI":"10.1007\/978-3-031-36272-9_33"},{"key":"9836_CR25","doi-asserted-by":"crossref","unstructured":"Wu, S., Koo, M., Blum, L., Black, A., Kao, L., Scalzo, F. & Kurtz, I. (2023). A comparative study of open-source large language models, gpt-4 and claude 2: Multiple-choice test taking in nephrology. arXiv:2308.04709.","DOI":"10.1056\/AIdbp2300092"},{"key":"9836_CR26","doi-asserted-by":"crossref","unstructured":"Wu, X., Yao, W., Chen, J., Pan, X., Wang, X., Liu, N. & Yu, D. (2024). From language modeling to instruction following: Understanding the behavior shift in llms after instruction tuning. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), pp. 2341\u20132369.","DOI":"10.18653\/v1\/2024.naacl-long.130"},{"key":"9836_CR27","unstructured":"Wu, X., Zhao, H., Zhu, Y., Shi, Y., Yang, F., Liu, T., Zhai, X., Yao, W., Li, J., Du, M., Liu, N. (2024). Usable xai: 10 strategies towards exploiting explainability in the llm era. arXiv preprint arXiv:2403.08946."},{"key":"9836_CR28","first-page":"2171206","volume":"2022","author":"D Wu","year":"2022","unstructured":"Wu, D., Wang, M., & Li, X. (2022). Automatic scoring for translations based on language models. Computational Intelligence and Neuroscience, 2022, 2171206.","journal-title":"Computational Intelligence and Neuroscience"},{"key":"9836_CR29","unstructured":"Xia, W., Mao, S. & Zheng, C. (2024). Empirical study of large language models as automated essay scoring tools in english composition_taking toefl independent writing task for example. arXiv preprint arXiv:2401.03401."},{"issue":"1","key":"9836_CR30","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1111\/bjet.13370","volume":"55","author":"L Yan","year":"2024","unstructured":"Yan, L., Sha, L., Zhao, L., Li, Y., Martinez-Maldonado, R., Chen, G., Li, X., Jin, Y., & Ga\u0161evi\u0107, D. (2024). Practical and ethical challenges of large language models in education: A systematic scoping review. British Journal of Educational Technology, 55(1), 90\u2013112.","journal-title":"British Journal of Educational Technology"},{"key":"9836_CR31","doi-asserted-by":"crossref","unstructured":"Zhai, X. (2024). Ai and machine learning for next generation sci-ence assessments. Machine Learning, Natural Language Processing, and Psychometrics, 201.","DOI":"10.1108\/979-8-88730-606-320251011"},{"issue":"1","key":"9836_CR32","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1007\/s11423-020-09917-8","volume":"69","author":"X Zhai","year":"2021","unstructured":"Zhai, X. (2021). Advancing automatic guidance in virtual science inquiry: From ease of use to personalization. Educational Technology Research and Development, 69(1), 255\u2013258.","journal-title":"Educational Technology Research and Development"},{"issue":"3","key":"9836_CR33","doi-asserted-by":"publisher","first-page":"42","DOI":"10.1145\/3589649","volume":"29","author":"X Zhai","year":"2023","unstructured":"Zhai, X. (2023). Chatgpt for next generation science learning. XRDS: Crossroads, The ACM Magazine for Students, 29(3), 42\u201346.","journal-title":"XRDS: Crossroads, The ACM Magazine for Students"},{"issue":"9","key":"9836_CR34","doi-asserted-by":"publisher","first-page":"1430","DOI":"10.1002\/tea.21658","volume":"57","author":"X Zhai","year":"2020","unstructured":"Zhai, X., C Haudek, K., Shi, L., H Nehm, R., & Urban-Lurain, M. (2020). From substitution to redefinition: A framework of machine learning-based science assessment. Journal of Research in Science Teaching, 57(9), 1430\u20131459.","journal-title":"Journal of Research in Science Teaching"},{"issue":"10","key":"9836_CR35","doi-asserted-by":"publisher","first-page":"1765","DOI":"10.1002\/tea.21773","volume":"59","author":"X Zhai","year":"2022","unstructured":"Zhai, X., He, P., & Krajcik, J. (2022). Applying machine learning to automatically assess scientific models. Journal of Research in Science Teaching, 59(10), 1765\u20131794.","journal-title":"Journal of Research in Science Teaching"},{"issue":"2","key":"9836_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3639372","volume":"15","author":"H Zhao","year":"2024","unstructured":"Zhao, H., Chen, H., Yang, F., Liu, N., Deng, H., Cai, H., Wang, S., Yin, D., & Du, M. (2024). Explainability for large language models: A survey. ACM Transactions on Intelligent Systems and Technology, 15(2), 1\u201338.","journal-title":"ACM Transactions on Intelligent Systems and Technology"},{"key":"9836_CR37","doi-asserted-by":"crossref","unstructured":"Zheng, L., Chiang, W.-L., Sheng, Y., Zhuang, S., Wu, Z., Zhuang, Y., Lin, Z., Li, Z., Li, D., Xing, E., et al. (2024). Judging llm-as-a-judge with mt-bench and chatbot arena. NIPS 36.","DOI":"10.52202\/075280-2020"}],"container-title":["Technology, Knowledge and Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10758-025-09836-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10758-025-09836-8","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10758-025-09836-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T05:57:15Z","timestamp":1782971835000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10758-025-09836-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,21]]},"references-count":37,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["9836"],"URL":"https:\/\/doi.org\/10.1007\/s10758-025-09836-8","relation":{},"ISSN":["2211-1662","2211-1670"],"issn-type":[{"value":"2211-1662","type":"print"},{"value":"2211-1670","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,21]]},"assertion":[{"value":"22 February 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 March 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}