{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T06:10:28Z","timestamp":1780467028411,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,12]],"date-time":"2026-04-12T00:00:00Z","timestamp":1775952000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,12]]},"DOI":"10.1145\/3786149.3788306","type":"proceedings-article","created":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T05:45:46Z","timestamp":1780465546000},"page":"33-36","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["OLAF: Towards Robust LLM-Based Annotation Framework in Empirical Software Engineering"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2477-9971","authenticated-orcid":false,"given":"Mia Mohammad","family":"Imran","sequence":"first","affiliation":[{"name":"Computer Science, Missouri University of Science and Technology, Rolla, MO, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8634-524X","authenticated-orcid":false,"given":"Tarannum Shaila","family":"Zaman","sequence":"additional","affiliation":[{"name":"Information Systems, University of Maryland Baltimore County, Baltimore, MF, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,2]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/SaTML64287.2025.00011"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/MSR66628.2025.00086"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICMLA58977.2023.00089"},{"key":"e_1_3_3_1_5_2","unstructured":"Sebastian Baltes Florian Angermeir Chetan Arora Marvin\u00a0Mu\u00f1oz Bar\u00f3n Chunyang Chen et\u00a0al. 2025. Guidelines for Empirical Studies in Software Engineering involving Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2508.15503"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.275"},{"key":"e_1_3_3_1_7_2","unstructured":"Julian Baumann Thomas Wachinger Leonhard Dorsch et\u00a0al. 2025. Large Language Model Hacking: Quantifying the Hidden Risks of Using LLMs for Text Annotation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2509.08825 (2025)."},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/3715275.3732194"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.852"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.2"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"crossref","unstructured":"Alexander Churchill Shamitha Pichika Chengxin Xu and Ying Liu. 2025. GPT models for text annotation: An empirical exploration in public policy research. Policy Studies Journal (2025).","DOI":"10.31235\/osf.io\/6fpgj"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Jacob Cohen. 1960. A Coefficient of Agreement for Nominal Scales. Educational and Psychological Measurement (1960).","DOI":"10.1177\/001316446002000104"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.366"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"Fabrizio Gilardi Meysam Alizadeh and Ma\u00ebl Kubli. 2023. ChatGPT outperforms crowd workers for text-annotation tasks. Proceedings of the National Academy of Sciences 120 30 (2023) e2305016120.","DOI":"10.1073\/pnas.2305016120"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-00262-6_7"},{"key":"e_1_3_3_1_16_2","volume-title":"Handbook of inter-rater reliability: The definitive guide to measuring the extent of agreement among raters","author":"Gwet Kilem\u00a0L","year":"2014","unstructured":"Kilem\u00a0L Gwet. 2014. Handbook of inter-rater reliability: The definitive guide to measuring the extent of agreement among raters. Advanced Analytics, LLC."},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"crossref","unstructured":"Matthias Haucke Rink Hoekstra and Don Van\u00a0Ravenzwaaij. 2021. When numbers fail: do researchers agree on operationalization of published research? Royal Society Open Science 8 9 (2021) 191354.","DOI":"10.1098\/rsos.191354"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642834"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1145\/3194932.3194938"},{"key":"e_1_3_3_1_20_2","volume-title":"Content Analysis: An Introduction to Its Methodology (4th ed.)","author":"Krippendorff Klaus","year":"2018","unstructured":"Klaus Krippendorff. 2018. Content Analysis: An Introduction to Its Methodology (4th ed.). Sage Publications."},{"key":"e_1_3_3_1_21_2","first-page":"2277","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics","author":"Li Minzhi","year":"2025","unstructured":"Minzhi Li, Zhengyuan Liu, Shumin Deng, Shafiq Joty, Nancy Chen, and Min-Yen Kan. 2025. DnA-Eval: Enhancing Large Language Model Evaluation through Decomposition and Aggregation. In Proceedings of the 31st International Conference on Computational Linguistics. 2277\u20132290."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"crossref","unstructured":"Q\u00a0Vera Liao and Jennifer\u00a0Wortman Vaughan. 2024. AI Transparency in the Age of LLMs: A Human-Centered Research Roadmap. Harvard Data Science ReviewSpecial Issue 5 (2024).","DOI":"10.1162\/99608f92.8036d03b"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/RE59067.2024.00046"},{"key":"e_1_3_3_1_24_2","volume-title":"Human-in-the-Loop Machine Learning: Active learning and annotation for human-centered AI","author":"Monarch Robert\u00a0Munro","year":"2021","unstructured":"Robert\u00a0Munro Monarch. 2021. Human-in-the-Loop Machine Learning: Active learning and annotation for human-centered AI. Simon and Schuster."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"Ernesto\u00a0Lang Oreamuno Rohan\u00a0Faiyaz Khan Abdul\u00a0Ali Bangash Catherine Stinson and Bram Adams. 2024. The state of documentation practices of third-party machine learning models and datasets. IEEE Software 41 5 (2024) 52\u201359.","DOI":"10.1109\/MS.2024.3366111"},{"key":"e_1_3_3_1_26_2","volume-title":"Proceedings of the 2008 conference on EMNLP","author":"Snow Rion","year":"2008","unstructured":"Rion Snow, Brendan O\u2019connor, Dan Jurafsky, and Andrew\u00a0Y Ng. 2008. Cheap and fast\u2013but is it good? evaluating non-expert annotations for natural language tasks. In Proceedings of the 2008 conference on EMNLP."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/WSESE66602.2025.00011"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"Ruiqi Wang Jiyu Guo Cuiyun Gao Guodong Fan Chun\u00a0Yong Chong and Xin Xia. 2025. Can llms replace human evaluators? an empirical study of llm-as-a-judge in software engineering. Proceedings of the ACM on Software Engineering 2 ISSTA (2025) 1955\u20131977.","DOI":"10.1145\/3728963"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3641960"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Dustin Wright and Isabelle Augenstein. 2025. Aggregating soft labels from crowd annotations improves uncertainty estimation under distribution shift. PLoS One 20 6 (2025) e0323064.","DOI":"10.1371\/journal.pone.0323064"}],"event":{"name":"WSESE '26: 2026 IEEE\/ACM International Workshop on Methodological Issues with Empirical Studies in Software Engineering","location":"Rio de Janeiro Brazil","acronym":"WSESE '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","IEEE CS","Faculty of Engineering of University of Porto"]},"container-title":["Proceedings of the 2026 IEEE\/ACM International Workshop on Methodological Issues with Empirical Studies in Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3786149.3788306","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T05:46:46Z","timestamp":1780465606000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3786149.3788306"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":29,"alternative-id":["10.1145\/3786149.3788306","10.1145\/3786149"],"URL":"https:\/\/doi.org\/10.1145\/3786149.3788306","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-06-02","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}