{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T16:03:09Z","timestamp":1783526589669,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","funder":[{"name":"2025 Special Program for Supporting Innovative Development in Leading Industries (AI Track) under the High-Quality Industrial Development Initiative","award":["Project Name Finstep Finsmart Intelligent Service Platform \/ Project ID: 2025-GZL-RGZN-01024"],"award-info":[{"award-number":["Project Name Finstep Finsmart Intelligent Service Platform \/ Project ID: 2025-GZL-RGZN-01024"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,15]]},"DOI":"10.1145\/3768292.3770364","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T07:24:26Z","timestamp":1763105066000},"page":"656-664","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["FinResearchBench: A Logic Tree based Agent-as-a-Judge Evaluation Framework for Financial Research Agents"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-6031-5770","authenticated-orcid":false,"given":"Rui","family":"Sun","sequence":"first","affiliation":[{"name":"Stepfun, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5349-9739","authenticated-orcid":false,"given":"Zuo","family":"Bai","sequence":"additional","affiliation":[{"name":"FinStep, Shanghai, China and Stepfun, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0903-8260","authenticated-orcid":false,"given":"Wentao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Stepfun, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9221-7284","authenticated-orcid":false,"given":"Yuxiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Stepfun, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6531-359X","authenticated-orcid":false,"given":"Li","family":"Zhao","sequence":"additional","affiliation":[{"name":"Stepfun, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8853-0352","authenticated-orcid":false,"given":"Shan","family":"Sun","sequence":"additional","affiliation":[{"name":"FinStep, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2937-8799","authenticated-orcid":false,"given":"Zhengwen","family":"Qiu","sequence":"additional","affiliation":[{"name":"FinStep, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,11,14]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Anthropic. 2024. Anthropic: Introducing Claude 3.5 Sonnet. https:\/\/www.anthropic.com\/news\/claude-3-5-sonnet"},{"key":"e_1_3_3_2_3_2","unstructured":"Yushi Bai Jiajie Zhang Xin Lv Linzhi Zheng Siqi Zhu Lei Hou Yuxiao Dong Jie Tang and Juanzi Li. 2024. LongWriter: Unleashing 10 000+ Word Generation from Long Context LLMs. arxiv:https:\/\/arXiv.org\/abs\/2408.07055\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2408.07055"},{"key":"e_1_3_3_2_4_2","unstructured":"Kaiyuan Chen Yixin Ren Yang Liu Xiaobo Hu Haotong Tian Tianbao Xie Fangfu Liu Haoye Zhang Hongzhang Liu Yuan Gong Chen Sun Han Hou Hui Yang James Pan Jianan Lou Jiayi Mao Jizheng Liu Jinpeng Li Kangyi Liu Kenkun Liu Rui Wang Run Li Tong Niu Wenlong Zhang Wenqi Yan Xuanzheng Wang Yuchen Zhang Yi-Hsin Hung Yuan Jiang Zexuan Liu Zihan Yin Zijian Ma and Zhiwen Mo. 2025. xbench: Tracking Agents Productivity Scaling with Profession-Aligned Real-World Evaluations. arxiv:https:\/\/arXiv.org\/abs\/2506.13651\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2506.13651"},{"key":"e_1_3_3_2_5_2","series-title":"(NIPS \u201923)","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Deng Xiang","year":"2023","unstructured":"Xiang Deng, Yu Gu, Boyuan Zheng, Shijie Chen, Samuel Stevens, Boshi Wang, Huan Sun, and Yu Su. 2023. MIND2WEB: towards a generalist agent for the web. In Proceedings of the 37th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201923). Curran Associates Inc., Red Hook, NY, USA, Article 1220, 24\u00a0pages."},{"key":"e_1_3_3_2_6_2","unstructured":"Mingxuan Du Benfeng Xu Chiwei Zhu Xiaorui Wang and Zhendong Mao. 2025. DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents. arxiv:https:\/\/arXiv.org\/abs\/2506.11763\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2506.11763"},{"key":"e_1_3_3_2_7_2","unstructured":"FutureSearch : Nikos\u00a0I. Bosse Jon Evans Robert\u00a0G. Gambee Daniel Hnyk Peter M\u00fchlbacher Lawrence Phillips Dan Schwarz and Jack Wildman. 2025. Deep Research Bench: Evaluating AI Web Research Agents. arxiv:https:\/\/arXiv.org\/abs\/2506.06287\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2506.06287"},{"key":"e_1_3_3_2_8_2","unstructured":"Xin Guo Haotian Xia Zhaowei Liu Hanyang Cao Zhi Yang Zhiqiang Liu Sizhe Wang Jinyi Niu Chuqi Wang Yanhui Wang Xiaolong Liang Xiaoming Huang Bing Zhu Zhongyu Wei Yun Chen Weining Shen and Liwen Zhang. 2024. FinEval: A Chinese Financial Domain Knowledge Evaluation Benchmark for Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2308.09975\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2308.09975"},{"key":"e_1_3_3_2_9_2","volume-title":"ICLR","author":"Hong Sirui","year":"2024","unstructured":"Sirui Hong, Mingchen Zhuge, Jonathan Chen, Xiawu Zheng, Yuheng Cheng, Jinlin Wang, Ceyao Zhang, Zili Wang, Steven Ka\u00a0Shing Yau, Zijuan Lin, Liyang Zhou, Chenyu Ran, Lingfeng Xiao, Chenglin Wu, and J\u00fcrgen Schmidhuber. 2024. MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework. In ICLR. https:\/\/openreview.net\/forum?id=VtmBAGCN7o"},{"key":"e_1_3_3_2_10_2","unstructured":"Dawei Li Bohan Jiang Liangjie Huang Alimohammad Beigi Chengshuai Zhao Zhen Tan Amrita Bhattacharjee Yuxuan Jiang Canyu Chen Tianhao Wu et\u00a0al. 2024. From generation to judgment: Opportunities and challenges of llm-as-a-judge. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2411.16594 (2024)."},{"key":"e_1_3_3_2_11_2","unstructured":"Haitao Li Junjie Chen Jingli Yang Qingyao Ai Wei Jia Youfeng Liu Kai Lin Yueyue Wu Guozhi Yuan Yiran Hu et\u00a0al. 2024. LegalAgentBench: Evaluating LLM Agents in Legal Domain. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.17259 (2024)."},{"key":"e_1_3_3_2_12_2","volume-title":"The Twelfth International Conference on Learning Representations","author":"Liu Xiao","year":"2024","unstructured":"Xiao Liu, Hao Yu, Hanchen Zhang, Yifan Xu, Xuanyu Lei, Hanyu Lai, Yu Gu, Hangliang Ding, Kaiwen Men, Kejuan Yang, Shudan Zhang, Xiang Deng, Aohan Zeng, Zhengxiao Du, Chenhui Zhang, Sheng Shen, Tianjun Zhang, Yu Su, Huan Sun, Minlie Huang, Yuxiao Dong, and Jie Tang. 2024. AgentBench: Evaluating LLMs as Agents. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=zAdUB0aCTQ"},{"key":"e_1_3_3_2_13_2","unstructured":"Haoran Que Feiyu Duan Liqun He Yutao Mou Wangchunshu Zhou Jiaheng Liu Wenge Rong Zekun\u00a0Moore Wang Jian Yang Ge Zhang Junran Peng Zhaoxiang Zhang Songyang Zhang and Kai Chen. 2024. HelloBench: Evaluating Long Text Generation Capabilities of Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2409.16191\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2409.16191"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"D\u00a0Gordon Rohman. 1965. Pre-writing: The stage of discovery in the writing process. College Composition & Communication 16 2 (1965) 106\u2013112.","DOI":"10.58680\/ccc196521081"},{"key":"e_1_3_3_2_15_2","unstructured":"Samuel Schmidgall Yusheng Su Ze Wang Ximeng Sun Jialian Wu Xiaodong Yu Jiang Liu Michael Moor Zicheng Liu and Emad Barsoum. 2025. Agent Laboratory: Using LLM Agents as Research Assistants. arxiv:https:\/\/arXiv.org\/abs\/2501.04227\u00a0[cs.HC] https:\/\/arxiv.org\/abs\/2501.04227"},{"key":"e_1_3_3_2_16_2","unstructured":"Xuemei Tang Xufeng Duan and Zhenguang\u00a0G Cai. 2024. Are LLMs Good Literature Review Writers? Evaluating the Literature Review Writing Ability of Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.13612 (2024)."},{"key":"e_1_3_3_2_17_2","unstructured":"Gemini Team Petko Georgiev Ving\u00a0Ian Lei Ryan Burnell Libin Bai Anmol Gulati Garrett Tanzer Damien Vincent Zhufeng Pan Shibo Wang et\u00a0al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.05530 (2024)."},{"key":"e_1_3_3_2_18_2","series-title":"(NIPS \u201922)","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed\u00a0H. Chi, Quoc\u00a0V. Le, and Denny Zhou. 2022. Chain-of-thought prompting elicits reasoning in large language models. In Proceedings of the 36th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201922). Curran Associates Inc., Red Hook, NY, USA, Article 1800, 14\u00a0pages."},{"key":"e_1_3_3_2_19_2","unstructured":"Qingyun Wu Gagan Bansal Jieyu Zhang Yiran Wu Beibin Li Erkang Zhu Li Jiang Xiaoyun Zhang Shaokun Zhang Jiale Liu Ahmed\u00a0Hassan Awadallah Ryen\u00a0W White Doug Burger and Chi Wang. 2023. AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation. arxiv:https:\/\/arXiv.org\/abs\/2308.08155\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2308.08155"},{"key":"e_1_3_3_2_20_2","unstructured":"Siwei Wu Yizhi Li Xingwei Qu Rishi Ravikumar Yucheng Li Tyler Loakman Shanghaoran Quan Xiaoyong Wei Riza Batista-Navarro and Chenghua Lin. 2025. LongEval: A Comprehensive Analysis of Long-Text Generation Through a Plan-based Paradigm. arxiv:https:\/\/arXiv.org\/abs\/2502.19103\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2502.19103"},{"key":"e_1_3_3_2_21_2","series-title":"(ICML\u201924)","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Xu Haoran","year":"2024","unstructured":"Haoran Xu, Amr Sharaf, Yunmo Chen, Weiting Tan, Lingfeng Shen, Benjamin Van\u00a0Durme, Kenton Murray, and Young\u00a0Jin Kim. 2024. Contrastive preference optimization: pushing the boundaries of LLM performance in machine translation. In Proceedings of the 41st International Conference on Machine Learning (Vienna, Austria) (ICML\u201924). JMLR.org, Article 2275, 21\u00a0pages."},{"key":"e_1_3_3_2_22_2","series-title":"(NIPS \u201923)","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Thomas\u00a0L. Griffiths, Yuan Cao, and Karthik Narasimhan. 2023. Tree of thoughts: deliberate problem solving with large language models. In Proceedings of the 37th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201923). Curran Associates Inc., Red Hook, NY, USA, Article 517, 14\u00a0pages."},{"key":"e_1_3_3_2_23_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik\u00a0R Narasimhan, and Yuan Cao. 2023. ReAct: Synergizing Reasoning and Acting in Language Models. In The Eleventh International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=WE_vluYUL-X"},{"key":"e_1_3_3_2_24_2","unstructured":"Zihao Yi Jiarui Ouyang Zhe Xu Yuwen Liu Tianhao Liao Haohao Luo and Ying Shen. 2025. A Survey on Recent Advances in LLM-Based Multi-turn Dialogue Systems. arxiv:https:\/\/arXiv.org\/abs\/2402.18013\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2402.18013"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.246"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3591196.3596612"},{"key":"e_1_3_3_2_27_2","series-title":"(NIPS \u201923)","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric\u00a0P. Xing, Hao Zhang, Joseph\u00a0E. Gonzalez, and Ion Stoica. 2023. Judging LLM-as-a-judge with MT-bench and Chatbot Arena. In Proceedings of the 37th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201923). Curran Associates Inc., Red Hook, NY, USA, Article 2020, 29\u00a0pages."},{"key":"e_1_3_3_2_28_2","unstructured":"Yuxiang Zheng Dayuan Fu Xiangkun Hu Xiaojie Cai Lyumanshan Ye Pengrui Lu and Pengfei Liu. 2025. DeepResearcher: Scaling Deep Research via Reinforcement Learning in Real-world Environments. arxiv:https:\/\/arXiv.org\/abs\/2504.03160\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2504.03160"},{"key":"e_1_3_3_2_29_2","series-title":"(NIPS \u201923)","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"Zhuang Yuchen","year":"2023","unstructured":"Yuchen Zhuang, Yue Yu, Kuan Wang, Haotian Sun, and Chao Zhang. 2023. ToolQA: a dataset for LLM question answering with external tools. In Proceedings of the 37th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201923). Curran Associates Inc., Red Hook, NY, USA, Article 2180, 27\u00a0pages."},{"key":"e_1_3_3_2_30_2","unstructured":"Mingchen Zhuge Changsheng Zhao Dylan Ashley Wenyi Wang Dmitrii Khizbullin Yunyang Xiong Zechun Liu Ernie Chang Raghuraman Krishnamoorthi Yuandong Tian Yangyang Shi Vikas Chandra and J\u00fcrgen Schmidhuber. 2024. Agent-as-a-Judge: Evaluate Agents with Agents. arxiv:https:\/\/arXiv.org\/abs\/2410.10934\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2410.10934"}],"event":{"name":"ICAIF '25: 6th ACM International Conference on AI in Finance","location":"Singapore Singapore","acronym":"ICAIF '25"},"container-title":["Proceedings of the 6th ACM International Conference on AI in Finance"],"original-title":[],"deposited":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T07:25:13Z","timestamp":1763105113000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3768292.3770364"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,14]]},"references-count":29,"alternative-id":["10.1145\/3768292.3770364","10.1145\/3768292"],"URL":"https:\/\/doi.org\/10.1145\/3768292.3770364","relation":{},"subject":[],"published":{"date-parts":[[2025,11,14]]},"assertion":[{"value":"2025-11-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}