{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T22:08:50Z","timestamp":1784412530006,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":106,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,11]],"date-time":"2024-05-11T00:00:00Z","timestamp":1715385600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,11]]},"DOI":"10.1145\/3613904.3641960","type":"proceedings-article","created":{"date-parts":[[2024,5,11]],"date-time":"2024-05-11T08:39:12Z","timestamp":1715416752000},"page":"1-21","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":117,"title":["Human-LLM Collaborative Annotation Through Effective Verification of LLM Labels"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0213-6425","authenticated-orcid":false,"given":"Xinru","family":"Wang","sequence":"first","affiliation":[{"name":"Purdue University, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0137-7171","authenticated-orcid":false,"given":"Hannah","family":"Kim","sequence":"additional","affiliation":[{"name":"Megagon Labs, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4210-1582","authenticated-orcid":false,"given":"Sajjadur","family":"Rahman","sequence":"additional","affiliation":[{"name":"Megagon Labs, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5898-243X","authenticated-orcid":false,"given":"Kushan","family":"Mitra","sequence":"additional","affiliation":[{"name":"Megagon Labs, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2371-1186","authenticated-orcid":false,"given":"Zhengjie","family":"Miao","sequence":"additional","affiliation":[{"name":"School of Computing Science, Simon Fraser University, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,5,11]]},"reference":[{"key":"e_1_3_3_3_1_1","unstructured":"[n. d.]. BERT base model (uncased). https:\/\/huggingface.co\/bert-base-uncased. Accessed: 2023-09-14."},{"key":"e_1_3_3_3_2_1","unstructured":"[n. d.]. Introducing ChatGPT. https:\/\/openai.com\/blog\/chatgpt. Accessed: 2023-09-14."},{"key":"e_1_3_3_3_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3241916"},{"key":"e_1_3_3_3_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445717"},{"key":"e_1_3_3_3_6_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3173574.3173951"},{"key":"e_1_3_3_3_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_9_1","volume-title":"Advances in Neural Information Processing Systems, Vol.\u00a033. Curran Associates","author":"Brown Tom","year":"1877","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared\u00a0D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel Ziegler, Jeffrey Wu, Clemens Winter, Chris Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. In Advances in Neural Information Processing Systems, Vol.\u00a033. Curran Associates, Inc., 1877\u20131901. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/1457c0d6bfcb4967418bfb8ac142f64a-Paper.pdf"},{"key":"e_1_3_3_3_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_11_1","volume-title":"Two Failures of Self-Consistency in the Multi-Step Reasoning of LLMs. Transactions on Machine Learning Research","author":"Chen Angelica","year":"2024","unstructured":"Angelica Chen, Jason Phang, Alicia Parrish, Vishakh Padmakumar, Chen Zhao, Samuel\u00a0R. Bowman, and Kyunghyun Cho. 2024. Two Failures of Self-Consistency in the Multi-Step Reasoning of LLMs. Transactions on Machine Learning Research (2024). https:\/\/openreview.net\/forum?id=5nBqY1y96B"},{"key":"e_1_3_3_3_12_1","doi-asserted-by":"publisher","unstructured":"Karl Cobbe Vineet Kosaraju Mohammad Bavarian Mark Chen Heewoo Jun Lukasz Kaiser Matthias Plappert Jerry Tworek Jacob Hilton Reiichiro Nakano Christopher Hesse and John Schulman. 2021. Training Verifiers to Solve Math Word Problems. https:\/\/doi.org\/10.48550\/ARXIV.2110.14168","DOI":"10.48550\/ARXIV.2110.14168"},{"key":"e_1_3_3_3_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/bf00994018"},{"key":"e_1_3_3_3_14_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00425"},{"key":"e_1_3_3_3_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_17_1","doi-asserted-by":"publisher","unstructured":"Qingxiu Dong Lei Li Damai Dai Ce Zheng Zhiyong Wu Baobao Chang Xu Sun Jingjing Xu Lei Li and Zhifang Sui. 2023. A Survey on In-context Learning. https:\/\/doi.org\/10.48550\/ARXIV.2301.00234","DOI":"10.48550\/ARXIV.2301.00234"},{"key":"e_1_3_3_3_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512943"},{"key":"e_1_3_3_3_19_1","doi-asserted-by":"publisher","unstructured":"Sharon\u00a0A Ferguson Paula\u00a0Akemi Aoyagui and Anastasia Kuzminykh. 2023. Something Borrowed: Exploring the Influence of AI-Generated Explanation Text on the Composition of Human Explanations. In Extended Abstracts of the 2023 CHI Conference on Human Factors in Computing Systems (Hamburg Germany) (CHI EA \u201923). Association for Computing Machinery New York NY USA Article 253 7\u00a0pages. https:\/\/doi.org\/10.1145\/3544549.3585727","DOI":"10.1145\/3544549.3585727"},{"key":"e_1_3_3_3_20_1","doi-asserted-by":"publisher","DOI":"10.5555\/1868720.1868727"},{"key":"e_1_3_3_3_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581352"},{"key":"e_1_3_3_3_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IGARSS52108.2023.10283015"},{"key":"e_1_3_3_3_25_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2305016120"},{"key":"e_1_3_3_3_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3287560.3287563"},{"key":"e_1_3_3_3_27_1","volume-title":"Proceedings of the 34th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a070)","author":"Guo Chuan","year":"2017","unstructured":"Chuan Guo, Geoff Pleiss, Yu Sun, and Kilian\u00a0Q. Weinberger. 2017. On Calibration of Modern Neural Networks. In Proceedings of the 34th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a070). PMLR, 1321\u20131330. https:\/\/proceedings.mlr.press\/v70\/guo17a.html"},{"key":"e_1_3_3_3_28_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.05153"},{"key":"e_1_3_3_3_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_30_1","doi-asserted-by":"publisher","unstructured":"Xingwei He Zhenghao Lin Yeyun Gong A-Long Jin Hang Zhang Chen Lin Jian Jiao Siu\u00a0Ming Yiu Nan Duan and Weizhu Chen. 2023. AnnoLLM: Making Large Language Models to Be Better Crowdsourced Annotators. https:\/\/doi.org\/10.48550\/ARXIV.2303.16854","DOI":"10.48550\/ARXIV.2303.16854"},{"key":"e_1_3_3_3_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/hcomp.v2i1.13213"},{"key":"e_1_3_3_3_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.1995.598994"},{"key":"e_1_3_3_3_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_35_1","doi-asserted-by":"publisher","unstructured":"Saurav Kadavath Tom Conerly Amanda Askell Tom Henighan Dawn Drain Ethan Perez Nicholas Schiefer Zac Hatfield-Dodds Nova DasSarma Eli Tran-Johnson Scott Johnston Sheer El-Showk Andy Jones Nelson Elhage Tristan Hume Anna Chen Yuntao Bai Sam Bowman Stanislav Fort Deep Ganguli Danny Hernandez Josh Jacobson Jackson Kernion Shauna Kravec Liane Lovitt Kamal Ndousse Catherine Olsson Sam Ringer Dario Amodei Tom Brown Jack Clark Nicholas Joseph Ben Mann Sam McCandlish Chris Olah and Jared Kaplan. 2022. Language Models (Mostly) Know What They Know. https:\/\/doi.org\/10.48550\/ARXIV.2207.05221","DOI":"10.48550\/ARXIV.2207.05221"},{"key":"e_1_3_3_3_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581001"},{"key":"e_1_3_3_3_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3129669"},{"key":"e_1_3_3_3_38_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022"},{"key":"e_1_3_3_3_39_1","doi-asserted-by":"publisher","unstructured":"Johnson Kuan and Jonas Mueller. 2022. Back to the Basics: Revisiting Out-of-Distribution Detection Baselines. https:\/\/doi.org\/10.48550\/ARXIV.2207.03061","DOI":"10.48550\/ARXIV.2207.03061"},{"key":"e_1_3_3_3_40_1","doi-asserted-by":"publisher","unstructured":"Tsung-Ting Kuo Jina Huh Jihoon Kim Robert El-Kareh Siddharth Singh Stephanie\u00a0Feudjio Feupe Vincent Kuri Gordon Lin Michele\u00a0E. Day Lucila Ohno-Machado and Chun-Nan Hsu. 2018. The Impact of Automatic Pre-annotation in Clinical Note Data Element Extraction - the CLEAN Tool. https:\/\/doi.org\/10.48550\/ARXIV.1808.03806","DOI":"10.48550\/ARXIV.1808.03806"},{"key":"e_1_3_3_3_41_1","doi-asserted-by":"publisher","unstructured":"Taja Kuzman Igor Mozeti\u010d and Nikola Ljube\u0161i\u0107. 2023. ChatGPT: Beginning of an End of Manual Linguistic Data Annotation? Use Case of Automatic Genre Identification. https:\/\/doi.org\/10.48550\/ARXIV.2303.03953","DOI":"10.48550\/ARXIV.2303.03953"},{"key":"e_1_3_3_3_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3287560.3287590"},{"key":"e_1_3_3_3_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610206"},{"key":"e_1_3_3_3_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_45_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_46_1","doi-asserted-by":"publisher","unstructured":"Shiyang Li Jianshu Chen Yelong Shen Zhiyu Chen Xinlu Zhang Zekun Li Hong Wang Jing Qian Baolin Peng Yi Mao Wenhu Chen and Xifeng Yan. 2022. Explanations from Large Language Models Make Small Reasoners Better. https:\/\/doi.org\/10.48550\/ARXIV.2210.06726","DOI":"10.48550\/ARXIV.2210.06726"},{"key":"e_1_3_3_3_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_49_1","volume-title":"Holistic Evaluation of Language Models. Transactions on Machine Learning Research","author":"Liang Percy","year":"2023","unstructured":"Percy Liang, Rishi Bommasani, Tony Lee, Dimitris Tsipras, Dilara Soylu, Michihiro Yasunaga, Yian Zhang, Deepak Narayanan, Yuhuai Wu, Ananya Kumar, Benjamin Newman, Binhang Yuan, Bobby Yan, Ce Zhang, Christian\u00a0Alexander Cosgrove, Christopher\u00a0D Manning, Christopher Re, Diana Acosta-Navas, Drew\u00a0Arad Hudson, Eric Zelikman, Esin Durmus, Faisal Ladhak, Frieda Rong, Hongyu Ren, Huaxiu Yao, Jue WANG, Keshav Santhanam, Laurel Orr, Lucia Zheng, Mert Yuksekgonul, Mirac Suzgun, Nathan Kim, Neel Guha, Niladri\u00a0S. Chatterji, Omar Khattab, Peter Henderson, Qian Huang, Ryan\u00a0Andrew Chi, Sang\u00a0Michael Xie, Shibani Santurkar, Surya Ganguli, Tatsunori Hashimoto, Thomas Icard, Tianyi Zhang, Vishrav Chaudhary, William Wang, Xuechen Li, Yifan Mai, Yuhui Zhang, and Yuta Koreeda. 2023. Holistic Evaluation of Language Models. Transactions on Machine Learning Research (2023). https:\/\/openreview.net\/forum?id=iO4LZibEqW Featured Certification, Expert Certification."},{"key":"e_1_3_3_3_50_1","volume-title":"Teaching Models to Express Their Uncertainty in Words. Transactions on Machine Learning Research","author":"Lin Stephanie","year":"2022","unstructured":"Stephanie Lin, Jacob Hilton, and Owain Evans. 2022. Teaching Models to Express Their Uncertainty in Words. Transactions on Machine Learning Research (2022). https:\/\/openreview.net\/forum?id=8s8K2UZGTZ"},{"key":"e_1_3_3_3_51_1","doi-asserted-by":"publisher","DOI":"10.1136\/amiajnl-2013-001837"},{"key":"e_1_3_3_3_52_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3479552"},{"key":"e_1_3_3_3_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"e_1_3_3_3_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581058"},{"key":"e_1_3_3_3_57_1","doi-asserted-by":"publisher","DOI":"10.5555\/1599081.1599147"},{"key":"e_1_3_3_3_58_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0162)","author":"Majumder Bodhisattwa\u00a0Prasad","year":"2022","unstructured":"Bodhisattwa\u00a0Prasad Majumder, Oana Camburu, Thomas Lukasiewicz, and Julian Mcauley. 2022. Knowledge-Grounded Self-Rationalization via Extractive and Natural Language Explanations. In Proceedings of the 39th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0162). PMLR, 14786\u201314801. https:\/\/proceedings.mlr.press\/v162\/majumder22a.html"},{"key":"e_1_3_3_3_59_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_60_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i17.17745"},{"key":"e_1_3_3_3_61_1","volume-title":"Proceedings of the Thirteenth Language Resources and Evaluation Conference. European Language Resources Association","author":"Mikulov\u00e1 Marie","year":"2022","unstructured":"Marie Mikulov\u00e1, Milan Straka, Jan \u0160t\u011bp\u00e1nek, Barbora \u0160t\u011bp\u00e1nkov\u00e1, and Jan Hajic. 2022. Quality and Efficiency of Manual Annotation: Pre-annotation Bias. In Proceedings of the Thirteenth Language Resources and Evaluation Conference. European Language Resources Association, Marseille, France, 2909\u20132918. https:\/\/aclanthology.org\/2022.lrec-1.312"},{"key":"e_1_3_3_3_62_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_63_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_64_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0202)","author":"Ni Ansong","year":"2023","unstructured":"Ansong Ni, Srini Iyer, Dragomir Radev, Veselin Stoyanov, Wen-Tau Yih, Sida Wang, and Xi\u00a0Victoria Lin. 2023. LEVER: Learning to Verify Language-to-Code Generation with Execution. In Proceedings of the 40th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0202). PMLR, 26106\u201326128. https:\/\/proceedings.mlr.press\/v202\/ni23b.html"},{"key":"e_1_3_3_3_65_1","volume-title":"Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC\u201908)","author":"Ogren Philip","year":"2008","unstructured":"Philip Ogren, Guergana Savova, and Christopher Chute. 2008. Constructing Evaluation Corpora for Automated Clinical Named Entity Recognition. In Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC\u201908). European Language Resources Association (ELRA), Marrakech, Morocco. http:\/\/www.lrec-conf.org\/proceedings\/lrec2008\/pdf\/796_paper.pdf"},{"key":"e_1_3_3_3_67_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-023-02433-3"},{"key":"e_1_3_3_3_68_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_69_1","doi-asserted-by":"publisher","unstructured":"Nicholas Pangakis Samuel Wolken and Neil Fasching. 2023. Automated Annotation with Generative AI Requires Validation. https:\/\/doi.org\/10.48550\/ARXIV.2306.00176","DOI":"10.48550\/ARXIV.2306.00176"},{"key":"e_1_3_3_3_70_1","doi-asserted-by":"publisher","DOI":"10.5555\/1953048.2078195"},{"key":"e_1_3_3_3_71_1","volume-title":"Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC\u201908)","author":"Ringger Eric","year":"2008","unstructured":"Eric Ringger, Marc Carmen, Robbie Haertel, Kevin Seppi, Deryle Lonsdale, Peter McClanahan, James Carroll, and Noel Ellison. 2008. Assessing the Costs of Machine-Assisted Corpus Annotation through a User Study. In Proceedings of the Sixth International Conference on Language Resources and Evaluation (LREC\u201908). European Language Resources Association (ELRA), Marrakech, Morocco. http:\/\/www.lrec-conf.org\/proceedings\/lrec2008\/pdf\/832_paper.pdf"},{"key":"e_1_3_3_3_72_1","doi-asserted-by":"publisher","DOI":"10.1002\/widm.2"},{"key":"e_1_3_3_3_73_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_74_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_75_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_76_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_77_1","volume-title":"51st Annual Meeting of the Association for Computational Linguistics Proceedings of the Student Research Workshop. Association for Computational Linguistics","author":"Skeppstedt Maria","year":"2013","unstructured":"Maria Skeppstedt. 2013. Annotating named entities in clinical text by combining pre-annotation and active learning. In 51st Annual Meeting of the Association for Computational Linguistics Proceedings of the Student Research Workshop. Association for Computational Linguistics, Sofia, Bulgaria, 74\u201380. https:\/\/aclanthology.org\/P13-3011"},{"key":"e_1_3_3_3_78_1","doi-asserted-by":"publisher","DOI":"10.21248\/jlcl.31.2016.203"},{"key":"e_1_3_3_3_79_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbi.2014.05.002"},{"key":"e_1_3_3_3_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3301275.3302322"},{"key":"e_1_3_3_3_81_1","volume-title":"Artificial Intelligence in HCI","author":"Stites C.","unstructured":"Mallory\u00a0C. Stites, Megan Nyre-Yu, Blake Moss, Charles Smutz, and Michael\u00a0R. Smith. 2021. Sage Advice? The Impacts of Explanations for Machine Learning Models on Human Decision-Making in Spam Detection. In Artificial Intelligence in HCI. Springer International Publishing, Cham, 269\u2013284."},{"key":"e_1_3_3_3_82_1","volume-title":"Proceedings of the Computational Sanskrit & Digital Humanities: Selected papers presented at the 18th World Sanskrit Conference. Association for Computational Linguistics","author":"Sujoy Sarkar","year":"2023","unstructured":"Sarkar Sujoy, Amrith Krishna, and Pawan Goyal. 2023. Pre-annotation Based Approach for Development of a Sanskrit Named Entity Recognition Dataset. In Proceedings of the Computational Sanskrit & Digital Humanities: Selected papers presented at the 18th World Sanskrit Conference. Association for Computational Linguistics, Canberra, Australia (Online mode), 59\u201370. https:\/\/aclanthology.org\/2023.wsc-csdh.4"},{"key":"e_1_3_3_3_83_1","doi-asserted-by":"publisher","unstructured":"Petter T\u00f6rnberg. 2023. ChatGPT-4 Outperforms Experts and Crowd Workers in Annotating Political Twitter Messages with Zero-Shot Learning. https:\/\/doi.org\/10.48550\/ARXIV.2304.06588","DOI":"10.48550\/ARXIV.2304.06588"},{"key":"e_1_3_3_3_84_1","doi-asserted-by":"publisher","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. https:\/\/doi.org\/10.48550\/ARXIV.2302.13971","DOI":"10.48550\/ARXIV.2302.13971"},{"key":"e_1_3_3_3_85_1","unstructured":"Miles Turpin Julian Michael Ethan Perez and Samuel\u00a0R. Bowman. 2023. Language Models Don\u2019t Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting. In Thirty-seventh Conference on Neural Information Processing Systems. https:\/\/openreview.net\/forum?id=bzs4uPLXvi"},{"key":"e_1_3_3_3_86_1","doi-asserted-by":"publisher","DOI":"10.1097\/01.ccm.0000458294.39613.e1"},{"key":"e_1_3_3_3_87_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023"},{"key":"e_1_3_3_3_88_1","volume-title":"PINTO: Faithful Language Reasoning Using Prompt-Generated Rationales. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=WBXbRs63oVu","author":"Wang PeiFeng","year":"2023","unstructured":"PeiFeng Wang, Aaron Chan, Filip Ilievski, Muhao Chen, and Xiang Ren. 2023. PINTO: Faithful Language Reasoning Using Prompt-Generated Rationales. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=WBXbRs63oVu"},{"key":"e_1_3_3_3_89_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_90_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_91_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/343"},{"key":"e_1_3_3_3_92_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512240"},{"key":"e_1_3_3_3_93_1","volume-title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=1PL1NIMMrw","author":"Wang Xuezhi","year":"2023","unstructured":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc\u00a0V Le, Ed\u00a0H. Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou. 2023. Self-Consistency Improves Chain of Thought Reasoning in Language Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=1PL1NIMMrw"},{"key":"e_1_3_3_3_94_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397481.3450650"},{"key":"e_1_3_3_3_95_1","doi-asserted-by":"publisher","DOI":"10.1145\/3519266"},{"key":"e_1_3_3_3_96_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581366"},{"key":"e_1_3_3_3_97_1","volume-title":"Fei Xia, Ed Chi, Quoc\u00a0V Le, and Denny Zhou.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, brian ichter, Fei Xia, Ed Chi, Quoc\u00a0V Le, and Denny Zhou. 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Advances in Neural Information Processing Systems, Vol.\u00a035. Curran Associates, Inc., 24824\u201324837. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/9d5609613524ecf4f15af0f7b31abca4-Paper-Conference.pdf"},{"key":"e_1_3_3_3_98_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_99_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_100_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_101_1","volume-title":"An Empirical Evaluation of Confidence Elicitation in LLMs. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=gjeQKFxFpZ","author":"Xiong Miao","year":"2024","unstructured":"Miao Xiong, Zhiyuan Hu, Xinyang Lu, YIFEI LI, Jie Fu, Junxian He, and Bryan Hooi. 2024. Can LLMs Express Their Uncertainty? An Empirical Evaluation of Confidence Elicitation in LLMs. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=gjeQKFxFpZ"},{"key":"e_1_3_3_3_102_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_3_3_103_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300509"},{"key":"e_1_3_3_3_104_1","unstructured":"Ann Yuan Daphne Ippolito Vitaly Nikolaev Chris Callison-Burch Andy Coenen and Sebastian Gehrmann. 2021. SynthBio: A Case Study in Faster Curation of Text Datasets. In Thirty-fifth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 2). https:\/\/openreview.net\/forum?id=Fkpr2RYDvI1"},{"key":"e_1_3_3_3_105_1","volume-title":"BERTScore: Evaluating Text Generation with BERT. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SkeHuCVFDr","author":"Zhang Tianyi","year":"2020","unstructured":"Tianyi Zhang, Varsha Kishore, Felix Wu, Kilian\u00a0Q. Weinberger, and Yoav Artzi. 2020. BERTScore: Evaluating Text Generation with BERT. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SkeHuCVFDr"},{"key":"e_1_3_3_3_106_1","doi-asserted-by":"publisher","DOI":"10.1145\/3351095.3372852"},{"key":"e_1_3_3_3_107_1","doi-asserted-by":"publisher","unstructured":"Yiming Zhu Peixian Zhang Ehsan-Ul Haq Pan Hui and Gareth Tyson. 2023. Can ChatGPT Reproduce Human-Generated Labels? A Study of Social Computing Tasks. https:\/\/doi.org\/10.48550\/ARXIV.2304.10145","DOI":"10.48550\/ARXIV.2304.10145"},{"key":"e_1_3_3_3_108_1","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00502"}],"event":{"name":"CHI '24: CHI Conference on Human Factors in Computing Systems","location":"Honolulu HI USA","acronym":"CHI '24","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction","SIGACCESS ACM Special Interest Group on Accessible Computing"]},"container-title":["Proceedings of the CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3613904.3641960","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3613904.3641960","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T23:57:29Z","timestamp":1750291049000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3613904.3641960"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,11]]},"references-count":106,"alternative-id":["10.1145\/3613904.3641960","10.1145\/3613904"],"URL":"https:\/\/doi.org\/10.1145\/3613904.3641960","relation":{},"subject":[],"published":{"date-parts":[[2024,5,11]]},"assertion":[{"value":"2024-05-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}