{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T07:47:03Z","timestamp":1782546423385,"version":"3.54.5"},"publisher-location":"Cham","reference-count":16,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032297594","type":"print"},{"value":"9783032297600","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,6,28]],"date-time":"2026-06-28T00:00:00Z","timestamp":1782604800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,28]],"date-time":"2026-06-28T00:00:00Z","timestamp":1782604800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-29760-0_9","type":"book-chapter","created":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T07:07:46Z","timestamp":1782544066000},"page":"75-83","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Beyond the\u00a0Gold Standard: Reliability Estimation of\u00a0Human and\u00a0GenAI Scoring"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-5995-219X","authenticated-orcid":false,"given":"Ji Yoon","family":"Jung","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8771-4780","authenticated-orcid":false,"given":"Ummugul","family":"Bezirhan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1298-9701","authenticated-orcid":false,"given":"Matthias","family":"von Davier","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,28]]},"reference":[{"key":"9_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.asw.2025.100988","volume":"66","author":"MD Shermis","year":"2025","unstructured":"Shermis, M.D.: Using ChatGPT to score essays and short-form constructed responses. Assess. Writ. 66, 100988 (2025)","journal-title":"Assess. Writ."},{"key":"9_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2024.100210","volume":"6","author":"E Latif","year":"2024","unstructured":"Latif, E., Zhai, X.: Fine-tuning ChatGPT for automatic scoring. Comput. Educ. Artifi. Intell. 6, 100210 (2024)","journal-title":"Comput. Educ. Artifi. Intell."},{"key":"9_CR3","doi-asserted-by":"publisher","DOI":"10.1016\/j.caeai.2025.100375","volume":"8","author":"JY Jung","year":"2025","unstructured":"Jung, J.Y., Tyack, L., von Davier, M.: Towards the implementation of automated scoring in international large-scale assessments: scalability and quality control. Comput. Educ. Artifi. Intell. 8, 100375 (2025)","journal-title":"Comput. Educ. Artifi. Intell."},{"issue":"1","key":"9_CR4","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1111\/j.1745-3992.2011.00223.x","volume":"31","author":"DM Williamson","year":"2012","unstructured":"Williamson, D.M., Xi, X., Breyer, F.J.: A framework for evaluation and use of automated scoring. Educ. Meas. Issues Pract. 31(1), 2\u201313 (2012)","journal-title":"Educ. Meas. Issues Pract."},{"issue":"1","key":"9_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1177\/026553229501200101","volume":"12","author":"A Brown","year":"1995","unstructured":"Brown, A.: The effect of rater variables in the development of an occupation-specific language performance test. Lang. Test. 12(1), 1\u201315 (1995)","journal-title":"Lang. Test."},{"issue":"4","key":"9_CR6","doi-asserted-by":"publisher","first-page":"479","DOI":"10.1177\/0265532214530699","volume":"31","author":"G Ling","year":"2014","unstructured":"Ling, G., Mollaun, P., Xi, X.: A study on the impact of fatigue on human raters when scoring speaking responses. Lang. Test. 31(4), 479\u2013499 (2014)","journal-title":"Lang. Test."},{"issue":"4","key":"9_CR7","first-page":"386","volume":"4","author":"CM Myford","year":"2003","unstructured":"Myford, C.M., Wolfe, E.W.: Detecting and measuring rater effects using many-facet Rasch measurement: Part I. J. Appl. Meas. 4(4), 386\u2013422 (2003)","journal-title":"J. Appl. Meas."},{"key":"9_CR8","doi-asserted-by":"publisher","unstructured":"Jung, J.Y., Bezirhan, U., von Davier, M.: Reconceptualizing scoring reliability through linguistic similarity. Educ. Psychol. Measurem. (2025). https:\/\/doi.org\/10.1177\/00131644251397428","DOI":"10.1177\/00131644251397428"},{"issue":"3","key":"9_CR9","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1111\/j.1745-3992.2012.00238.x","volume":"31","author":"II Bejar","year":"2012","unstructured":"Bejar, I.I.: Rater cognition: implications for validity. Educ. Meas. Issues Pract. 31(3), 2\u20139 (2012)","journal-title":"Educ. Meas. Issues Pract."},{"key":"9_CR10","first-page":"1","volume":"13","author":"CA McClellan","year":"2010","unstructured":"McClellan, C.A.: Constructed-response scoring\u2013doing it right. R&D Connections 13, 1\u20137 (2010)","journal-title":"R&D Connections"},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Miao, J., Cao, Y.: Development and evaluation of a partial double scoring procedure for preservice teacher portfolio assessment. AERA Online Paper Repository (2019)","DOI":"10.3102\/1432878"},{"issue":"1","key":"9_CR12","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1111\/emip.12636","volume":"44","author":"Y Xu","year":"2025","unstructured":"Xu, Y., Wind, S.A.: Examining the psychometric impact of targeted and random double-scoring in mixed-format assessments. Educ. Meas. Issues Pract. 44(1), 18\u201330 (2025)","journal-title":"Educ. Meas. Issues Pract."},{"issue":"2","key":"9_CR13","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1080\/08957347.2022.2067543","volume":"35","author":"YA Song","year":"2022","unstructured":"Song, Y.A., Lee, W.C.: Effects of using double ratings as item scores on IRT proficiency estimation. Appl. Measur. Educ. 35(2), 95\u2013115 (2022)","journal-title":"Appl. Measur. Educ."},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"Powers, D.E., Escoffery, D.S., Duchnowski, M.P.: Validating automated essay scoring: a (modest) refinement of the \u201cgold standard.\u201d Appl. Measur. Educ. 28(2), 130\u2013142 (2015)","DOI":"10.1080\/08957347.2014.1002920"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Kojima, T., Gu, S.S., Reid, M., Matsuo, Y., Iwasawa, Y.: Large language models are zero-shot reasoners. In: Proceedings of the 35th International Conference on Neural Information Processing Systems (NeurIPS), pp. 22199\u201322213 (2022)","DOI":"10.52202\/068431-1613"},{"key":"9_CR16","unstructured":"Jung, J.Y., Bezirhan, U., von Davier, M.: Optimizing reliability scoring for ILSAs. In: Proceedings of the AI in Measurement and Evaluation Conference (AIME-Con): Full Papers, pp. 43\u201349 (2025)"}],"container-title":["Lecture Notes in Computer Science","Artificial Intelligence in Education"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-29760-0_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T07:07:52Z","timestamp":1782544072000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-29760-0_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,28]]},"ISBN":["9783032297594","9783032297600"],"references-count":16,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-29760-0_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,28]]},"assertion":[{"value":"28 June 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"AIED","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Intelligence in Education","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Seoul","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Korea (Republic of)","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 June 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"aied2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.aied-conference.org\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}