{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T23:56:08Z","timestamp":1772927768368,"version":"3.50.1"},"reference-count":24,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"3","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2026,3,1]]},"DOI":"10.1587\/transinf.2024edp7281","type":"journal-article","created":{"date-parts":[[2025,8,31]],"date-time":"2025-08-31T22:07:06Z","timestamp":1756678026000},"page":"419-429","source":"Crossref","is-referenced-by-count":0,"title":["Error Candidate Cross-Validation for Efficient Annotation Error Detection"],"prefix":"10.1587","volume":"E109.D","author":[{"given":"Kenji","family":"AKIYAMA","sequence":"first","affiliation":[{"name":"Tokyo University of Agriculture and Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kanako","family":"KOMIYA","sequence":"additional","affiliation":[{"name":"Tokyo University of Agriculture and Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takafumi","family":"SAITO","sequence":"additional","affiliation":[{"name":"Tokyo University of Agriculture and Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","unstructured":"[1] J. Risch and R. Krestel, \u201cBagging BERT Models for Robust Aggression Identification,\u201d Language Resources and Evaluation Conference, 2020."},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] J.-C. Klie, B. Webber, and I. Gurevych, \u201cAnnotation Error Detection: Analyzing the Past and Present for a More Coherent Future,\u201d Computational Linguistics, vol.49, no.1, pp.157-198, 2023. 10.1162\/coli_a_00464","DOI":"10.1162\/coli_a_00464"},{"key":"3","unstructured":"[3] H. Halteren, \u201cThe Detection of Inconsistency in Manually Tagged Text,\u201d International Conference on Computational Linguistics, 2000."},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] H. Loftsson, \u201cCorrecting a PoS-tagged corpus using three complementary methods,\u201d Conference of the European Chapter of the Association for Computational Linguistics, pp.523-531, 2009. 10.3115\/1609067.1609125","DOI":"10.3115\/1609067.1609125"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] H. Amir, T. Miller, and G. Savova, \u201cSpotting Spurious Data with Neural Networks,\u201d North American Chapter of the Association for Computational Linguistics, pp.2006-2016, 2018. 10.18653\/v1\/n18-1182","DOI":"10.18653\/v1\/N18-1182"},{"key":"6","doi-asserted-by":"publisher","unstructured":"[6] M.M. Tatsuoka, F.M. Lord, M.R. Novick, and A. Birnbaum, \u201cStatistical Theories of Mental Test Scores,\u201d Journal of the American Statistical Association, vol.66, no.335, pp.651-652 1971. 10.2307\/2283550","DOI":"10.2307\/2283550"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] P. Rodriguez, J. Barrow, A. Hoyle, J.P. Lalor, R. Jia, and J. Boyd-Graber, \u201cEvaluation Examples are not Equally Informative: How should that change NLP Leaderboards?\u201d Annual Meeting of the Association for Computational Linguistics, pp.4486-4503, 2021. 10.18653\/v1\/2021.acl-long.346","DOI":"10.18653\/v1\/2021.acl-long.346"},{"key":"8","unstructured":"[8] D. Hendrycks and K. Gimpel, \u201cA Baseline for Detecting Misclassified and Out-of-Distribution Examples in Neural Networks,\u201d Proceedings of International Conference on Learning Representations, pp.1-12, 2016."},{"key":"9","unstructured":"[9] D. Dligach and M. Palmer, \u201cReducing the need for double annotation,\u201d Proceedings of the Fifth Law Workshop (LAW V), pp.65-73, 2011."},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] K. Gligori\u2019c, T. Zrnic, C. Lee, E.J. Candes and D. Jurafsky, \u201cCan Unconfident LLM Annotations Be Used for Confident Conclusions?\u201d arXiv:2408.15204, 2024.","DOI":"10.18653\/v1\/2025.naacl-long.179"},{"key":"11","doi-asserted-by":"publisher","unstructured":"[11] Z. Zhu, Y. Wang, S. Yang, L. Long, R. Wu, X. Tang, J. Zhao, and H. Wang, \u201cCORAL: Collaborative Automatic Labeling System Based on Large Language Models,\u201d Proc. VLDB Endow., Collaborative Automatic Labeling System Based on Large Language Models, vol.17, no.12, pp.4401-4404, 2024. 10.14778\/3685800.3685885","DOI":"10.14778\/3685800.3685885"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] H. Kim, K. Mitra, R.L. Chen, S. Rahman, and D. Zhang, \u201cMEGAnno+: A Human-LLM Collaborative Annotation System,\u201d Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics: System Demonstrations, pp.168-176, 2024. 10.18653\/v1\/2024.eacl-demo.18","DOI":"10.18653\/v1\/2024.eacl-demo.18"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] L. Weber-Genzel, S. Peng, M.-C. De Marneffe, and B. Plank, \u201cVariErr NLI: Separating Annotation Error from Human Label Variation,\u201d Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics, pp.2256-2269, 2024. 10.18653\/v1\/2024.acl-long.123","DOI":"10.18653\/v1\/2024.acl-long.123"},{"key":"14","unstructured":"[14] Open AI, \u201cGPT-4 Technical Report,\u201d arXiv:2303.08774, 2023."},{"key":"15","unstructured":"[15] K. Tsuji, T. Hiraoka, Y. Cheng, and T. Iwakura, \u201cSubRegWeigh: Effective and Efficient Annotation Weighing with Subword Regularization,\u201d Proceedings of the 31st International Conference on Computational Linguistics, pp.1908-1921, 2025."},{"key":"16","unstructured":"[16] V. Sanh, L. Debut, J. Chaumond, and T. Wolf, \u201cDistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter,\u201d Proceedings of the 5th Workshop on Energy Efficient Machine Learning and Cognitive Computing, 2020."},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] A. Wang, A. Singh, J. Michael, F. Hill, O. Levy, and S. Bowman, \u201cGLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding,\u201d Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP, pp.353-355, 2018. 10.18653\/v1\/w18-5446","DOI":"10.18653\/v1\/W18-5446"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] C.T. Hemphill, J.J. Godfrey, and G.R. Doddington, \u201cThe ATIS Spoken Language Systems Pilot Corpus,\u201d Human Language Technology - The Baltic Perspectiv, pp.96-101, 1990. 10.3115\/116580.116613","DOI":"10.3115\/116580.116613"},{"key":"19","unstructured":"[19] C.G. Northcutt, A. Athalye, and J. Mueller, \u201cPervasive Label Errors in Test Sets Destabilize Machine Learning Benchmarks,\u201d International Conference on Learning Representations Workshop Track (ICLR), 2021."},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] R. Socher, A. Perelygin, J. Wu, J. Chuang, C.D. Manning, A.Y. Ng, and C. Potts, \u201cRecursive Deep Models for Semantic Compositionality Over a Sentiment Treebank,\u201d Conference on Empirical Methods in Natural Language Processing, pp.1631-1642, 2013. 10.18653\/v1\/d13-1170","DOI":"10.18653\/v1\/D13-1170"},{"key":"21","doi-asserted-by":"publisher","unstructured":"[21] A. Zeldes, \u201cThe GUM corpus: Creating multilayer resources in the classroom,\u201d Language Resources and Evaluation, vol.51, no.3, pp.581-612, 2016. 10.1007\/s10579-016-9343-x","DOI":"10.1007\/s10579-016-9343-x"},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] B. Plank, D. Hovy, and A. S\u00f8gaard, \u201cLearning part-of-speech taggers with inter-annotator agreement loss,\u201d Conference of the European Chapter of the Association for Computational Linguistics, 2014. 10.3115\/v1\/e14-1078","DOI":"10.3115\/v1\/E14-1078"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] E.F. Tjong, K. Sang, and F.D. Meulder, \u201cIntroduction to the CoNLL-2003 Shared Task: Language-Independent Named Entity Recognition,\u201d Conference on Computational Natural Language Learning, vol.4, pp.142-147, 2003. 10.3115\/1119176.1119195","DOI":"10.3115\/1119176.1119195"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] S. Larson, A. Cheung, A. Mahendran, K. Leach, and J.K. Kummerfeld, \u201cInconsistencies in Crowdsourced Slot-Filling Annotations: A Typology and Identification Methods,\u201d International Conference on Computational Linguistics, pp.5035-5046, 2020. 10.18653\/v1\/2020.coling-main.442","DOI":"10.18653\/v1\/2020.coling-main.442"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E109.D\/3\/E109.D_2024EDP7281\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T04:10:41Z","timestamp":1772856641000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E109.D\/3\/E109.D_2024EDP7281\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,1]]},"references-count":24,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024edp7281","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"value":"0916-8532","type":"print"},{"value":"1745-1361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,1]]},"article-number":"2024EDP7281"}}