{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:08:43Z","timestamp":1784138923512,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"NSF IUCRC CRAFT Center","award":["CRAFT Grant 22017"],"award-info":[{"award-number":["CRAFT Grant 22017"]}]},{"name":"Columbia&#x27;s SIRS and STAR Program","award":["None"],"award-info":[{"award-number":["None"]}]},{"name":"Tang Family Fund for Research Innovations in FinTech, Engineering, and Business Operations","award":["None"],"award-info":[{"award-number":["None"]}]},{"name":"JPMorganChase Faculty Research Award","award":["None"],"award-info":[{"award-number":["None"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808578","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"3456-3463","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["FinAuditing: A Financial Taxonomy-Structured Multi-Document Benchmark for Evaluating LLMs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1036-9365","authenticated-orcid":false,"given":"Yan","family":"Wang","sequence":"first","affiliation":[{"name":"The Fin AI, New Haven, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3638-7816","authenticated-orcid":false,"given":"Keyi","family":"Wang","sequence":"additional","affiliation":[{"name":"Columbia University, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0316-7312","authenticated-orcid":false,"given":"Shanshan","family":"Yang","sequence":"additional","affiliation":[{"name":"Stevens Institute of Technology, Hoboken, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5350-2192","authenticated-orcid":false,"given":"Jaisal","family":"Patel","sequence":"additional","affiliation":[{"name":"Rensselaer Polytechnic Institute, Troy, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4791-3425","authenticated-orcid":false,"given":"Jeff","family":"Zhao","sequence":"additional","affiliation":[{"name":"The University of Texas at Austin, Austin, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0838-6994","authenticated-orcid":false,"given":"Fengran","family":"Mo","sequence":"additional","affiliation":[{"name":"University of Montreal, Montreal, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1484-0622","authenticated-orcid":false,"given":"Xueqing","family":"Peng","sequence":"additional","affiliation":[{"name":"The Fin AI, New Haven, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8342-1837","authenticated-orcid":false,"given":"Lingfei","family":"Qian","sequence":"additional","affiliation":[{"name":"The Fin AI, New Haven, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5741-2047","authenticated-orcid":false,"given":"Yankai","family":"Chen","sequence":"additional","affiliation":[{"name":"Mohamed bin Zayed University of Artificial Intelligence, Abu Dhabi, United Arab Emirates and McGill University, Montreal, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6117-5459","authenticated-orcid":false,"given":"V\u00edctor","family":"Guti\u00e9rrez-Basulto","sequence":"additional","affiliation":[{"name":"Cardiff University, Cardiff, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3501-3907","authenticated-orcid":false,"given":"Jimin","family":"Huang","sequence":"additional","affiliation":[{"name":"The Fin AI, New Haven, USA and The University of Manchester, Manchester, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5943-8109","authenticated-orcid":false,"given":"Guojun","family":"Xiong","sequence":"additional","affiliation":[{"name":"Harvard University, Cambridge, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9532-1709","authenticated-orcid":false,"given":"Xiao-Yang","family":"Liu","sequence":"additional","affiliation":[{"name":"Columbia University, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1556-3335","authenticated-orcid":false,"given":"Jian-Yun","family":"Nie","sequence":"additional","affiliation":[{"name":"University of Montreal, Montreal, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Finqa: A dataset of numerical reasoning over financial data. arXiv preprint arXiv:2109.00122","author":"Chen Zhiyu","year":"2021","unstructured":"Zhiyu Chen, Wenhu Chen, Charese Smiley, Sameena Shah, Iana Borova, Dylan Langdon, Reema Moussa, Matt Beane, Ting-Hao Huang, Bryan Routledge, et al., 2021. Finqa: A dataset of numerical reasoning over financial data. arXiv preprint arXiv:2109.00122 (2021)."},{"key":"e_1_3_2_1_2_1","volume-title":"Convfinqa: Exploring the chain of numerical reasoning in conversational finance question answering. arXiv preprint arXiv:2210.03849","author":"Chen Zhiyu","year":"2022","unstructured":"Zhiyu Chen, Shiyang Li, Charese Smiley, Zhiqiang Ma, Sameena Shah, and William Yang Wang. 2022. Convfinqa: Exploring the chain of numerical reasoning in conversational finance question answering. arXiv preprint arXiv:2210.03849 (2022)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72008-6"},{"key":"e_1_3_2_1_4_1","volume-title":"International Conference on Internet and Web Applications and Services.","author":"da Silva Paulo Caetano","year":"2018","unstructured":"Paulo Caetano da Silva. 2018. xAudit: Auditing Representation in XBRL Based Documents. In International Conference on Internet and Web Applications and Services."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jaccpubpol.2010.04.001"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.12608602"},{"key":"e_1_3_2_1_7_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_8_1","unstructured":"Jiawei Gu Xuhui Jiang Zhichao Shi Hexiang Tan Xuehao Zhai Chengjin Xu Wei Li Yinghan Shen Shengjie Ma Honghao Liu et al. 2024. A survey on llm-as-a-judge. arXiv preprint arXiv:2411.15594 (2024)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3677052.3698614"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052569"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.2308\/accr-51762"},{"key":"e_1_3_2_1_12_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_13_1","volume-title":"Information extraction from ESG reports using NLP: a ChatGPT comparison. Available at SSRN 4836432","author":"Katz Steven","year":"2024","unstructured":"Steven Katz, Yu Gu, and Lanxin Jiang. 2024. Information extraction from ESG reports using NLP: a ChatGPT comparison. Available at SSRN 4836432 (2024)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.126"},{"key":"e_1_3_2_1_15_1","volume-title":"Llms-as-judges: a comprehensive survey on llm-based evaluation methods. arXiv preprint arXiv:2412.05579","author":"Li Haitao","year":"2024","unstructured":"Haitao Li, Qian Dong, Junjie Chen, Huixue Su, Yujia Zhou, Qingyao Ai, Ziyi Ye, and Yiqun Liu. 2024. Llms-as-judges: a comprehensive survey on llm-based evaluation methods. arXiv preprint arXiv:2412.05579 (2024)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.2991\/csic-15.2015.76"},{"key":"e_1_3_2_1_17_1","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et al. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437 (2024)."},{"key":"e_1_3_2_1_18_1","unstructured":"Zhaowei Liu Xin Guo Fangqi Lou Lingfeng Zeng Jinyi Niu Zixuan Wang Jiajie Xu Weige Cai Ziwei Yang Xueqian Zhao Chao Li Sheng Xu Dezhi Chen Yun Chen Zuo Bai and Liwen Zhang. 2025. Fin-R1: A Large Language Model for Financial Reasoning through Reinforcement Learning. arXiv:2503.16252 [cs.CL] https:\/\/arxiv.org\/abs\/2503.16252"},{"key":"e_1_3_2_1_19_1","volume-title":"FiNER: Financial numeric entity recognition for XBRL tagging. arXiv preprint arXiv:2203.06482","author":"Loukas Lefteris","year":"2022","unstructured":"Lefteris Loukas, Manos Fergadiotis, Ilias Chalkidis, Eirini Spyropoulou, Prodromos Malakasiotis, Ion Androutsopoulos, and Georgios Paliouras. 2022. FiNER: Financial numeric entity recognition for XBRL tagging. arXiv preprint arXiv:2203.06482 (2022)."},{"key":"e_1_3_2_1_20_1","volume-title":"The llama 4 herd: The beginning of a new era of natively multimodal ai innovation. https:\/\/ai. meta.com\/blog\/llama-4-multimodal-intelligence\/, checked on","author":"Meta AI","year":"2025","unstructured":"AI Meta. 2025. The llama 4 herd: The beginning of a new era of natively multimodal ai innovation. https:\/\/ai. meta.com\/blog\/llama-4-multimodal-intelligence\/, checked on, Vol. 4, 7 (2025), 2025."},{"key":"e_1_3_2_1_21_1","volume-title":"Zheyuan Liu, Chao Zhang, Tetsuya Sakai, and Jian-Yun Nie.","author":"Mo Fengran","year":"2026","unstructured":"Fengran Mo, Zhan Su, Yuchen Hui, Jinghan Zhang, Jia Ao Sun, Zheyuan Liu, Chao Zhang, Tetsuya Sakai, and Jian-Yun Nie. 2026. Opendecoder: Open large language model decoding to incorporate document quality in rag. arXiv preprint arXiv:2601.09028 (2026)."},{"key":"e_1_3_2_1_22_1","volume-title":"Ectsum: A new benchmark dataset for bullet point summarization of long earnings call transcripts. arXiv preprint arXiv:2210.12467","author":"Mukherjee Rajdeep","year":"2022","unstructured":"Rajdeep Mukherjee, Abhinav Bohra, Akash Banerjee, Soumya Sharma, Manjunath Hegde, Afreen Shaikh, Shivani Shrivastava, Koustuv Dasgupta, Niloy Ganguly, Saptarshi Ghosh, et al., 2022. Ectsum: A new benchmark dataset for bullet point summarization of long earnings call transcripts. arXiv preprint arXiv:2210.12467 (2022)."},{"key":"e_1_3_2_1_23_1","unstructured":"Charles Packer Vivian Fang Shishir_G Patil Kevin Lin Sarah Wooders and Joseph_E Gonzalez. 2023. MemGPT: Towards LLMs as Operating Systems. (2023)."},{"key":"e_1_3_2_1_24_1","volume-title":"Huan He, Hanley Smith, Yi Han, Yueru He, Haohang Li, Yupeng Cao, et al.","author":"Qian Lingfei","year":"2025","unstructured":"Lingfei Qian, Xueqing Peng, Yan Wang, Vincent Jim Zhang, Huan He, Hanley Smith, Yi Han, Yueru He, Haohang Li, Yupeng Cao, et al., 2025a. When Agents Trade: Live Multi-Market Trading Benchmark for LLM Agents. arXiv preprint arXiv:2510.11695 (2025)."},{"key":"e_1_3_2_1_25_1","volume-title":"Fino1: On the Transferability of Reasoning Enhanced LLMs to Finance. arXiv preprint arXiv:2502.08127","author":"Qian Lingfei","year":"2025","unstructured":"Lingfei Qian, Weipeng Zhou, Yan Wang, Xueqing Peng, Jimin Huang, and Qianqian Xie. 2025b. Fino1: On the Transferability of Reasoning Enhanced LLMs to Finance. arXiv preprint arXiv:2502.08127 (2025)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.219"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-73103-8_41"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/BigData55660.2022.10020720"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.34740\/KAGGLE\/M\/3301"},{"key":"e_1_3_2_1_30_1","unstructured":"Qwen Team. 2025. Qwen3 Technical Report. arXiv:2505.09388 [cs.CL] https:\/\/arxiv.org\/abs\/2505.09388"},{"key":"e_1_3_2_1_31_1","volume-title":"FinTagging: An LLM-ready Benchmark for Extracting and Structuring Financial Information. arXiv preprint arXiv:2505.20650","author":"Wang Yan","year":"2025","unstructured":"Yan Wang, Yang Ren, Lingfei Qian, Xueqing Peng, Keyi Wang, Yi Han, Dongji Feng, Xiao-Yang Liu, Jimin Huang, and Qianqian Xie. 2025. FinTagging: An LLM-ready Benchmark for Extracting and Structuring Financial Information. arXiv preprint arXiv:2505.20650 (2025)."},{"key":"e_1_3_2_1_32_1","volume-title":"Pixiu: A large language model, instruction data and evaluation benchmark for finance. arXiv preprint arXiv:2306.05443","author":"Xie Qianqian","year":"2023","unstructured":"Qianqian Xie, Weiguang Han, Xiao Zhang, Yanzhao Lai, Min Peng, Alejandro Lopez-Lira, and Jimin Huang. 2023. Pixiu: A large language model, instruction data and evaluation benchmark for finance. arXiv preprint arXiv:2306.05443 (2023)."},{"key":"e_1_3_2_1_33_1","unstructured":"An Yang Baosong Yang Binyuan Hui Bo Zheng Bowen Yu Chang Zhou Chengpeng Li Chengyuan Li Dayiheng Liu Fei Huang Guanting Dong Haoran Wei Huan Lin Jialong Tang Jialin Wang Jian Yang Jianhong Tu Jianwei Zhang Jianxin Ma Jin Xu Jingren Zhou Jinze Bai Jinzheng He Junyang Lin Kai Dang Keming Lu Keqin Chen Kexin Yang Mei Li Mingfeng Xue Na Ni Pei Zhang Peng Wang Ru Peng Rui Men Ruize Gao Runji Lin Shijie Wang Shuai Bai Sinan Tan Tianhang Zhu Tianhao Li Tianyu Liu Wenbin Ge Xiaodong Deng Xiaohuan Zhou Xingzhang Ren Xinyu Zhang Xipin Wei Xuancheng Ren Yang Fan Yang Yao Yichang Zhang Yu Wan Yunfei Chu Yuqiong Liu Zeyu Cui Zhenru Zhang and Zhihao Fan. 2024. Qwen2 Technical Report. arXiv preprint arXiv:2407.10671 (2024)."},{"key":"e_1_3_2_1_34_1","volume-title":"MultiHiertt: Numerical reasoning over multi hierarchical tabular and textual data. arXiv preprint arXiv:2206.01347","author":"Zhao Yilun","year":"2022","unstructured":"Yilun Zhao, Yunxiang Li, Chenying Li, and Rui Zhang. 2022. MultiHiertt: Numerical reasoning over multi hierarchical tabular and textual data. arXiv preprint arXiv:2206.01347 (2022)."},{"key":"e_1_3_2_1_35_1","volume-title":"DocMath-eval: Evaluating math reasoning capabilities of LLMs in understanding long and specialized documents. arXiv preprint arXiv:2311.09805","author":"Zhao Yilun","year":"2023","unstructured":"Yilun Zhao, Yitao Long, Hongjun Liu, Ryo Kamoi, Linyong Nan, Lyuhao Chen, Yixin Liu, Xiangru Tang, Rui Zhang, and Arman Cohan. 2023. DocMath-eval: Evaluating math reasoning capabilities of LLMs in understanding long and specialized documents. arXiv preprint arXiv:2311.09805 (2023)."},{"key":"e_1_3_2_1_36_1","volume-title":"TAT-QA: A question answering benchmark on a hybrid of tabular and textual content in finance. arXiv preprint arXiv:2105.07624","author":"Zhu Fengbin","year":"2021","unstructured":"Fengbin Zhu, Wenqiang Lei, Youcheng Huang, Chao Wang, Shuo Zhang, Jiancheng Lv, Fuli Feng, and Tat-Seng Chua. 2021. TAT-QA: A question answering benchmark on a hybrid of tabular and textual content in finance. arXiv preprint arXiv:2105.07624 (2021)."}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:23:59Z","timestamp":1784136239000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808578"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":36,"alternative-id":["10.1145\/3805712.3808578","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808578","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}