{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:03:38Z","timestamp":1784138618253,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":76,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809717","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"1665-1676","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Uncertainty Quantification for Retrieval-Augmented Reasoning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0393-8662","authenticated-orcid":false,"given":"Heydar","family":"Soudani","sequence":"first","affiliation":[{"name":"Radboud University, Nijmegen, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0800-3340","authenticated-orcid":false,"given":"Hamed","family":"Zamani","sequence":"additional","affiliation":[{"name":"University of Massachusetts Amherst, Amherst, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9986-482X","authenticated-orcid":false,"given":"Faegheh","family":"Hasibi","sequence":"additional","affiliation":[{"name":"Radboud University, Nijmegen, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Summarize: A Modular Pipeline for Scientific Literature Summarization. CoRR","author":"Achkar Pierre","year":"2025","unstructured":"Pierre Achkar, Tim Gollub, and Martin Potthast. 2025. Ask, Retrieve, Summarize: A Modular Pipeline for Scientific Literature Summarization. CoRR (2025)."},{"key":"e_1_3_2_1_2_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR. OpenReview.net.","author":"Asai Akari","year":"2024","unstructured":"Akari Asai, Zeqiu Wu, Yizhong Wang, Avirup Sil, and Hannaneh Hajishirzi. 2024. Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection. In The Twelfth International Conference on Learning Representations, ICLR. OpenReview.net."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.419"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1429"},{"key":"e_1_3_2_1_5_1","volume-title":"Cycles of thought: Measuring llm confidence through stable explanations. arXiv preprint arXiv:2406.03441","author":"Becker Evan","year":"2024","unstructured":"Evan Becker and Stefano Soatto. 2024. Cycles of thought: Measuring llm confidence through stable explanations. arXiv preprint arXiv:2406.03441 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning. CoRR","author":"Chen Mingyang","year":"2025","unstructured":"Mingyang Chen, Tianpeng Li, Haoze Sun, Yijie Zhou, Chenzheng Zhu, Haofen Wang, Jeff Z. Pan, Wen Zhang, Huajun Chen, Fan Yang, Zenan Zhou, and Weipeng Chen. 2025c. ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning. CoRR (2025)."},{"key":"e_1_3_2_1_7_1","volume-title":"Revisiting RAG Ensemble: A Theoretical and Mechanistic Analysis of Multi-RAG System Collaboration. arXiv preprint arXiv:2508.13828","author":"Chen Yifei","year":"2025","unstructured":"Yifei Chen, Guanting Dong, Yutao Zhu, and Zhicheng Dou. 2025a. Revisiting RAG Ensemble: A Theoretical and Mechanistic Analysis of Multi-RAG System Collaboration. arXiv preprint arXiv:2508.13828 (2025)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2502.18036"},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems. 1-35","author":"Cox Samuel Rhys","unstructured":"Samuel Rhys Cox, Yunlong Wang, Ashraf Abdul, Christian Von Der Weth, and Brian Y. Lim. 2021. Directed diversity: Leveraging language embedding distances for collective creativity in crowd ideation. In Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems. 1-35."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657834"},{"key":"e_1_3_2_1_11_1","volume-title":"Comparing the areas under two or more correlated receiver operating characteristic curves: a nonparametric approach. Biometrics","author":"DeLong Elizabeth R","year":"1988","unstructured":"Elizabeth R DeLong, David M DeLong, and Daniel L Clarke-Pearson. 1988. Comparing the areas under two or more correlated receiver operating characteristic curves: a nonparametric approach. Biometrics (1988), 837-845."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.276"},{"key":"e_1_3_2_1_13_1","volume-title":"UProp: Investigating the Uncertainty Propagation of LLMs in Multi-Step Agentic Decision-Making. CoRR","author":"Duan Jinhao","year":"2025","unstructured":"Jinhao Duan, James Diffenderfer, Sandeep Madireddy, Tianlong Chen, Bhavya Kailkhura, and Kaidi Xu. 2025. UProp: Investigating the Uncertainty Propagation of LLMs in Multi-Step Agentic Decision-Making. CoRR, Vol. abs\/2506.17419 (2025)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.558"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.786"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-023-10562-9"},{"key":"e_1_3_2_1_17_1","volume-title":"Smoothie: Label Free Language Model Routing. In Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems","author":"Guha Neel","year":"2024","unstructured":"Neel Guha, Mayee F. Chen, Trevor Chow, Ishan S. Khare, and Christopher R\u00e9. 2024. Smoothie: Label Free Language Model Routing. In Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS."},{"key":"e_1_3_2_1_18_1","volume-title":"Variational Bayesian Last Layers. In The Twelfth International Conference on Learning Representations, ICLR. OpenReview.net.","author":"Harrison James","year":"2024","unstructured":"James Harrison, John Willes, and Jasper Snoek. 2024. Variational Bayesian Last Layers. In The Twelfth International Conference on Learning Representations, ICLR. OpenReview.net."},{"key":"e_1_3_2_1_19_1","first-page":"10371","article-title":"Retrieving, Rethinking and Revising: The Chain-of-Verification Can Improve Retrieval Augmented Generation","author":"He Bolei","year":"2024","unstructured":"Bolei He, Nuo Chen, Xinran He, Lingyong Yan, Zhenkai Wei, Jinchang Luo, and Zhen-Hua Ling. 2024. Retrieving, Rethinking and Revising: The Chain-of-Verification Can Improve Retrieval Augmented Generation. In Findings of the Association for Computational Linguistics: EMNLP. Association for Computational Linguistics, 10371-10393.","journal-title":"Findings of the Association for Computational Linguistics: EMNLP. Association for Computational Linguistics"},{"key":"e_1_3_2_1_20_1","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024","author":"Hou Bairu","year":"2024","unstructured":"Bairu Hou, Yujian Liu, Kaizhi Qian, Jacob Andreas, Shiyu Chang, and Yang Zhang. 2024. Decomposing Uncertainty for Large Language Models through Input Clarification Ensembling. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=byxXa99PtF"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730351"},{"key":"e_1_3_2_1_22_1","first-page":"14231","article-title":"Open-RAG: Enhanced Retrieval Augmented Reasoning with Open-Source Large Language Models","author":"Islam Shayekh Bin","year":"2024","unstructured":"Shayekh Bin Islam, Md. Asib Rahman, K. S. M. Tozammel Hossain, Enamul Hoque, Shafiq Joty, and Md. Rizwan Parvez. 2024. Open-RAG: Enhanced Retrieval Augmented Reasoning with Open-Source Large Language Models. In Findings of the Association for Computational Linguistics: EMNLP. Association for Computational Linguistics, 14231-14244.","journal-title":"Findings of the Association for Computational Linguistics: EMNLP. Association for Computational Linguistics"},{"key":"e_1_3_2_1_23_1","volume-title":"Hasibi Faegheh Hoveyda Mohanna, and Piepenbrock","author":"Vries Arjen","year":"2026","unstructured":"Jelle, de Vries Arjen P., de Rijke Maarten, Hasibi Faegheh Hoveyda Mohanna, and Piepenbrock. 2026. OrLog: Resolving Complex Queries with LLMs and Probabilistic Reasoning. In Advances in Information Retrieval (ECIR 2026). 98-114."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.792"},{"key":"e_1_3_2_1_25_1","first-page":"7064","article-title":"RAG-Star: Enhancing Deliberative Reasoning with Retrieval Augmented Verification and Refinement. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"Jiang Jinhao","year":"2025","unstructured":"Jinhao Jiang, Jiayi Chen, Junyi Li, Ruiyang Ren, Shijie Wang, Xin Zhao, Yang Song, and Tao Zhang. 2025a. RAG-Star: Enhancing Deliberative Reasoning with Retrieval Augmented Verification and Refinement. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL. Association for Computational Linguistics, 7064-7074.","journal-title":"NAACL. Association for Computational Linguistics"},{"key":"e_1_3_2_1_26_1","volume-title":"s3: You Don't Need That Much Data to Train a Search Agent via RL. arXiv preprint arXiv:2505.14146","author":"Jiang Pengcheng","year":"2025","unstructured":"Pengcheng Jiang, Xueqiang Xu, Jiacheng Lin, Jinfeng Xiao, Zifeng Wang, Jimeng Sun, and Jiawei Han. 2025b. s3: You Don't Need That Much Data to Train a Search Agent via RL. arXiv preprint arXiv:2505.14146 (2025)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.495"},{"key":"e_1_3_2_1_28_1","volume-title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning. CoRR","author":"Jin Bowen","year":"2025","unstructured":"Bowen Jin, Hansi Zeng, Zhenrui Yue, Dong Wang, Hamed Zamani, and Jiawei Han. 2025. Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning. CoRR (2025)."},{"key":"e_1_3_2_1_29_1","unstructured":"Saurav Kadavath et al. 2022. Language Models (Mostly) Know What They Know. Vol. abs\/2207.05221 (2022)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"e_1_3_2_1_31_1","volume-title":"What uncertainties do we need in bayesian deep learning for computer vision? Advances in neural information processing systems","author":"Kendall Alex","year":"2017","unstructured":"Alex Kendall and Yarin Gal. 2017. What uncertainties do we need in bayesian deep learning for computer vision? Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_32_1","volume-title":"Batched Self-Consistency Improves LLM Relevance Assessment and Ranking. CoRR","author":"Korikov Anton","year":"2025","unstructured":"Anton Korikov, Pan Du, Scott Sanner, and Navid Rekabsaz. 2025. Batched Self-Consistency Improves LLM Relevance Assessment and Ranking. CoRR, Vol. abs\/2505.12570 (2025)."},{"key":"e_1_3_2_1_33_1","volume-title":"Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation. In The Eleventh International Conference on Learning Representations ICLR.","author":"Kuhn Lorenz","year":"2023","unstructured":"Lorenz Kuhn, Yarin Gal, and Sebastian Farquhar. 2023. Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation. In The Eleventh International Conference on Learning Representations ICLR."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.587"},{"key":"e_1_3_2_1_35_1","volume-title":"Search-o1: Agentic Search-Enhanced Large Reasoning Models. CoRR","author":"Li Xiaoxi","year":"2025","unstructured":"Xiaoxi Li, Guanting Dong, Jiajie Jin, Yuyao Zhang, Yujia Zhou, Yutao Zhu, Peitian Zhang, and Zhicheng Dou. 2025a. Search-o1: Agentic Search-Enhanced Large Reasoning Models. CoRR (2025)."},{"key":"e_1_3_2_1_36_1","article-title":"Generating with Confidence: Uncertainty Quantification for Black-box Large Language","volume":"2024","author":"Lin Zhen","year":"2024","unstructured":"Zhen Lin, Shubhendu Trivedi, and Jimeng Sun. 2024. Generating with Confidence: Uncertainty Quantification for Black-box Large Language Models. Trans. Mach. Learn. Res., Vol. 2024 (2024).","journal-title":"Models. Trans. Mach. Learn. Res."},{"key":"e_1_3_2_1_37_1","volume-title":"Uncertainty estimation and quantification for llms: A simple supervised approach. arXiv preprint arXiv:2404.15993","author":"Liu Linyu","year":"2024","unstructured":"Linyu Liu, Yu Pan, Xiaocheng Li, and Guanting Chen. 2024. Uncertainty estimation and quantification for llms: A simple supervised approach. arXiv preprint arXiv:2404.15993 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3711896.3736569"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.322"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the 31st International Conference on Computational Linguistics, COLING. Association for Computational Linguistics, 9329-9345","author":"Madhusudhan Nishanth","year":"2025","unstructured":"Nishanth Madhusudhan, Sathwik Tejaswi Madhusudhan, Vikas Yadav, and Masoud Hashemi. 2025. Do LLMs Know When to NOT Answer? Investigating Abstention Abilities of Large Language Models. In Proceedings of the 31st International Conference on Computational Linguistics, COLING. Association for Computational Linguistics, 9329-9345."},{"key":"e_1_3_2_1_41_1","volume-title":"Uncertainty Estimation in Autoregressive Structured Prediction. In 9th International Conference on Learning Representations, ICLR.","author":"Malinin Andrey","unstructured":"Andrey Malinin and Mark J. F. Gales. 2021. Uncertainty Estimation in Autoregressive Structured Prediction. In 9th International Conference on Learning Representations, ICLR."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.546"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.319"},{"key":"e_1_3_2_1_44_1","volume-title":"Uncertainty Quantification in Retrieval Augmented Question Answering. arXiv preprint arXiv:2502.18108","author":"Perez-Beltrachini Laura","year":"2025","unstructured":"Laura Perez-Beltrachini and Mirella Lapata. 2025. Uncertainty Quantification in Retrieval Augmented Question Answering. arXiv preprint arXiv:2502.18108 (2025)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"Ofir Press Muru Zhang Sewon Min Ludwig Schmidt Noah Smith and Mike Lewis. 2023. Measuring and Narrowing the Compositionality Gap in Language Models. In Findings of the Association for Computational Linguistics: EMNLP.","DOI":"10.18653\/v1\/2023.findings-emnlp.378"},{"key":"e_1_3_2_1_46_1","volume-title":"Mutual Reasoning Makes Smaller LLMs Stronger Problem-Solver. In The Thirteenth International Conference on Learning Representations.","author":"Qi Zhenting","year":"2025","unstructured":"Zhenting Qi, Mingyuan MA, Jiahang Xu, Li Lyna Zhang, Fan Yang, and Mao Yang. 2025. Mutual Reasoning Makes Smaller LLMs Stronger Problem-Solver. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_2_1_47_1","volume-title":"Out-of-distribution detection and selective generation for conditional language models. arXiv preprint arXiv:2209.15558","author":"Ren Jie","year":"2022","unstructured":"Jie Ren, Jiaming Luo, Yao Zhao, Kundan Krishna, Mohammad Saleh, Balaji Lakshminarayanan, and Peter J Liu. 2022. Out-of-distribution detection and selective generation for conditional language models. arXiv preprint arXiv:2209.15558 (2022)."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 17th Annual International ACM-SIGIR Conference on Research and Development in Information Retrieval. 232-241","author":"Stephen","unstructured":"Stephen E. Robertson and Steve Walker. 1994. Some Simple Effective Approximations to the 2-Poisson Model for Probabilistic Weighted Retrieval. In Proceedings of the 17th Annual International ACM-SIGIR Conference on Research and Development in Information Retrieval. 232-241."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2506.10844"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657957"},{"key":"e_1_3_2_1_51_1","volume-title":"Search and Refine During Think: Autonomous Retrieval-Augmented Reasoning of LLMs. arXiv preprint arXiv:2505.11277","author":"Shi Yaorui","year":"2025","unstructured":"Yaorui Shi, Sihang Li, Chang Wu, Zhiyuan Liu, Junfeng Fang, Hengxing Cai, An Zhang, and Xiang Wang. 2025. Search and Refine During Think: Autonomous Retrieval-Augmented Reasoning of LLMs. arXiv preprint arXiv:2505.11277 (2025)."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730130"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3673791.3698415"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.852"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3795686"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.702"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730102"},{"key":"e_1_3_2_1_58_1","volume-title":"Manning","author":"Tian Katherine","year":"2023","unstructured":"Katherine Tian, Eric Mitchell, Allan Zhou, Archit Sharma, Rafael Rafailov, Huaxiu Yao, Chelsea Finn, and Christopher D. Manning. 2023. Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback. In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, EMNLP. Association for Computational Linguistics, 5433-5442."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.896"},{"key":"e_1_3_2_1_60_1","first-page":"539","article-title":"MuSiQue: Multihop Questions via Single-hop Question","volume":"10","author":"Trivedi Harsh","year":"2022","unstructured":"Harsh Trivedi, Niranjan Balasubramanian, Tushar Khot, and Ashish Sabharwal. 2022. MuSiQue: Multihop Questions via Single-hop Question Composition. Trans. Assoc. Comput. Linguistics, Vol. 10 (2022), 539-554.","journal-title":"Composition. Trans. Assoc. Comput. Linguistics"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.557"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.511"},{"key":"e_1_3_2_1_63_1","volume-title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models. In The Eleventh International Conference on Learning Representations, ICLR.","author":"Wang Xuezhi","year":"2023","unstructured":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc V. Le, Ed H. Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou. 2023. Self-Consistency Improves Chain of Thought Reasoning in Language Models. In The Eleventh International Conference on Learning Representations, ICLR."},{"key":"e_1_3_2_1_64_1","volume-title":"Heng Tao Shen, and Xiaofeng Zhu","author":"Wang Zhiyuan","year":"2024","unstructured":"Zhiyuan Wang, Jinhao Duan, Lu Cheng, Yue Zhang, Qingni Wang, Xiaoshuang Shi, Kaidi Xu, Heng Tao Shen, and Xiaofeng Zhu. 2024a. ConU: Conformal Uncertainty in Large Language Models with Correctness Coverage Guarantees. In Findings of the Association for Computational Linguistics: EMNLP. 6886-6898."},{"key":"e_1_3_2_1_65_1","first-page":"1040","article-title":"The Art of Abstention: Selective Prediction and Error Regularization for Natural Language Processing. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing","author":"Xin Ji","year":"2021","unstructured":"Ji Xin, Raphael Tang, Yaoliang Yu, and Jimmy Lin. 2021. The Art of Abstention: Selective Prediction and Error Regularization for Natural Language Processing. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, ACL\/IJCNLP. Association for Computational Linguistics, 1040-1051.","journal-title":"ACL\/IJCNLP. Association for Computational Linguistics"},{"key":"e_1_3_2_1_66_1","volume-title":"Can llms express their uncertainty? an empirical evaluation of confidence elicitation in llms. arXiv preprint arXiv:2306.13063","author":"Xiong Miao","year":"2023","unstructured":"Miao Xiong, Zhiyuan Hu, Xinyang Lu, Yifei Li, Jie Fu, Junxian He, and Bryan Hooi. 2023. Can llms express their uncertainty? an empirical evaluation of confidence elicitation in llms. arXiv preprint arXiv:2306.13063 (2023)."},{"key":"e_1_3_2_1_67_1","volume-title":"Baturalp Buyukates, Chenyang Tao, Anil Ramakrishna, Dimitrios Dimitriadis, Jieyu Zhao, and Salman Avestimehr.","author":"Yaldiz Duygu Nur","year":"2025","unstructured":"Duygu Nur Yaldiz, Yavuz Faruk Bakman, Baturalp Buyukates, Chenyang Tao, Anil Ramakrishna, Dimitrios Dimitriadis, Jieyu Zhao, and Salman Avestimehr. 2025a. Do Not Design, Learn: A Trainable Scoring Function for Uncertainty Estimation in Generative LLMs. In Findings of the Association for Computational Linguistics: NAACL 2025."},{"key":"e_1_3_2_1_68_1","volume-title":"Sungmin Kang, Alperen \u00d6zi\u015f, Hayrettin Eren Yildiz, Mitash Ashish Shah, Zhiqi Huang, Anoop Kumar, Alfy Samuel, Daben Liu, et al.","author":"Yaldiz Duygu Nur","year":"2025","unstructured":"Duygu Nur Yaldiz, Yavuz Faruk Bakman, Sungmin Kang, Alperen \u00d6zi\u015f, Hayrettin Eren Yildiz, Mitash Ashish Shah, Zhiqi Huang, Anoop Kumar, Alfy Samuel, Daben Liu, et al., 2025b. TruthTorchLM: A Comprehensive Library for Predicting Truthfulness in LLM Outputs. arXiv preprint arXiv:2507.08203 (2025)."},{"key":"e_1_3_2_1_69_1","unstructured":"An Yang et al. 2024. Qwen2.5 Technical Report. CoRR Vol. abs\/2412.15115 (2024)."},{"key":"e_1_3_2_1_70_1","volume-title":"Manning","author":"Yang Zhilin","year":"2018","unstructured":"Zhilin Yang, Peng Qi, Saizheng Zhang, Yoshua Bengio, William W. Cohen, Ruslan Salakhutdinov, and Christopher D. Manning. 2018. HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering. In Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, EMNLP. Association for Computational Linguistics, 2369-2380."},{"key":"e_1_3_2_1_71_1","volume-title":"ReAct: Synergizing Reasoning and Acting in Language Models. In The Eleventh International Conference on Learning Representations, ICLR.","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik R. Narasimhan, and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. In The Eleventh International Conference on Learning Representations, ICLR."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0491"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.131"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1339"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1181"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.302"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:07:00Z","timestamp":1784135220000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809717"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":76,"alternative-id":["10.1145\/3805712.3809717","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809717","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}