{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:36:59Z","timestamp":1776886619687,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100006374","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2023ZD0121101"],"award-info":[{"award-number":["2023ZD0121101"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Laboratory under grant","award":["231-HF-D04-01"],"award-info":[{"award-number":["231-HF-D04-01"]}]},{"name":"the Open Fund of National Key Laboratory of Parallel and Distributed Computing (PDL)","award":["NO.2024-KJWPDL-02"],"award-info":[{"award-number":["NO.2024-KJWPDL-02"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,13]]},"DOI":"10.1145\/3726302.3729965","type":"proceedings-article","created":{"date-parts":[[2025,7,14]],"date-time":"2025-07-14T14:55:26Z","timestamp":1752504926000},"page":"1444-1454","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Empowering Large Language Model Agent through Step-Level Self-Critique and Self-Training"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1385-0074","authenticated-orcid":false,"given":"Yuanzhao","family":"Zhai","sequence":"first","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6526-3333","authenticated-orcid":false,"given":"Huanxi","family":"Liu","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1835-5411","authenticated-orcid":false,"given":"Zhuo","family":"Zhang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology (Shenzhen), Shenzhen, China and Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3907-9402","authenticated-orcid":false,"given":"Tong","family":"Lin","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5997-5169","authenticated-orcid":false,"given":"Kele","family":"Xu","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4782-1645","authenticated-orcid":false,"given":"Cheng","family":"Yang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7587-8905","authenticated-orcid":false,"given":"Dawei","family":"Feng","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1236-8318","authenticated-orcid":false,"given":"Bo","family":"Ding","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3245-1901","authenticated-orcid":false,"given":"Huaimin","family":"Wang","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, China and State Key Laboratory of Complex &amp; Critical Software Environment, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-024-40231-1"},{"key":"e_1_3_2_1_2_1","volume-title":"HotpotQA: A dataset for diverse, explainable multi-hop question answering. arXiv preprint arXiv:1809.09600","author":"Yang Zhilin","year":"2018","unstructured":"Zhilin Yang, Peng Qi, Saizheng Zhang, Yoshua Bengio, William W Cohen, Ruslan Salakhutdinov, and Christopher D Manning. 2018. HotpotQA: A dataset for diverse, explainable multi-hop question answering. arXiv preprint arXiv:1809.09600 (2018)."},{"key":"e_1_3_2_1_3_1","first-page":"20744","article-title":"Webshop: Towards scalable real-world web interaction with grounded language agents","volume":"35","author":"Yao Shunyu","year":"2022","unstructured":"Shunyu Yao, Howard Chen, John Yang, and Karthik Narasimhan. 2022. Webshop: Towards scalable real-world web interaction with grounded language agents. Advances in Neural Information Processing Systems, Vol. 35 (2022), 20744--20757. http:\/\/arxiv.org\/abs\/2207.01206","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_4_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Zhou Shuyan","year":"2024","unstructured":"Shuyan Zhou, Frank F Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Yonatan Bisk, Daniel Fried, Uri Alon, et al. 2024. Webarena: A realistic web environment for building autonomous agents. The Twelfth International Conference on Learning Representations (2024)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657848"},{"key":"e_1_3_2_1_6_1","volume-title":"The Eleventh International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/2210","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik Narasimhan, and Yuan Cao. 2023. React: Synergizing reasoning and acting in language models. In The Eleventh International Conference on Learning Representations. https:\/\/arxiv.org\/abs\/2210.03629"},{"key":"e_1_3_2_1_7_1","volume-title":"International Conference on Machine Learning","author":"Xi Zhiheng","year":"2024","unstructured":"Zhiheng Xi, Wenxiang Chen, Boyang Hong, Senjie Jin, Rui Zheng, Wei He, Yiwen Ding, Shichun Liu, Xin Guo, Junzhe Wang, et al. 2024. Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning. International Conference on Machine Learning (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"International conference on machine learning. PMLR.","author":"Zhou Andy","year":"2024","unstructured":"Andy Zhou, Kai Yan, Michal Shlapentokh-Rothman, Haohan Wang, and Yu-Xiong Wang. 2024. Language agent tree search unifies reasoning acting and planning in language models. In International conference on machine learning. PMLR."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i25.34924"},{"key":"e_1_3_2_1_10_1","volume-title":"Agent q: Advanced reasoning and learning for autonomous ai agents. arXiv preprint arXiv:2408.07199","author":"Putta Pranav","year":"2024","unstructured":"Pranav Putta, Edmund Mills, Naman Garg, Sumeet Motwani, Chelsea Finn, Divyansh Garg, and Rafael Rafailov. 2024. Agent q: Advanced reasoning and learning for autonomous ai agents. arXiv preprint arXiv:2408.07199 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Shinn Noah","year":"2023","unstructured":"Noah Shinn, Federico Cassano, Ashwin Gopinath, Karthik Narasimhan, and Shunyu Yao. 2023. Reflexion: Language agents with verbal reinforcement learning. Advances in Neural Information Processing Systems, Vol. 36 (2023)."},{"key":"e_1_3_2_1_12_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Madaan Aman","year":"2023","unstructured":"Aman Madaan, Niket Tandon, Prakhar Gupta, Skyler Hallinan, Luyu Gao, Sarah Wiegreffe, Uri Alon, Nouha Dziri, Shrimai Prabhumoye, Yiming Yang, et al. 2023. Self-refine: Iterative refinement with self-feedback. Advances in Neural Information Processing Systems, Vol. 36 (2023)."},{"key":"e_1_3_2_1_13_1","unstructured":"OpenAI. 2024. Learning to Reason with Large Language Models. https:\/\/openai.com\/index\/learning-to-reason-with-llms\/."},{"key":"e_1_3_2_1_14_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_1_15_1","volume-title":"Critique ability of large language models. arXiv preprint arXiv:2310.04815","author":"Luo Liangchen","year":"2023","unstructured":"Liangchen Luo, Zi Lin, Yinxiao Liu, Lei Shu, Yun Zhu, Jingbo Shang, and Lei Meng. 2023. Critique ability of large language models. arXiv preprint arXiv:2310.04815 (2023)."},{"key":"e_1_3_2_1_16_1","volume-title":"Large Language Models Cannot Self-Correct Reasoning Yet. In The Twelfth International Conference on Learning Representations.","author":"Huang Jie","year":"2024","unstructured":"Jie Huang, Xinyun Chen, Swaroop Mishra, Huaixiu Steven Zheng, Adams Wei Yu, Xinying Song, and Denny Zhou. 2024. Large Language Models Cannot Self-Correct Reasoning Yet. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_17_1","volume-title":"Fireact: Toward language agent fine-tuning. arXiv preprint arXiv:2310.05915","author":"Chen Baian","year":"2023","unstructured":"Baian Chen, Chang Shu, Ehsan Shareghi, Nigel Collier, Karthik Narasimhan, and Shunyu Yao. 2023. Fireact: Toward language agent fine-tuning. arXiv preprint arXiv:2310.05915 (2023)."},{"key":"e_1_3_2_1_18_1","volume-title":"Agenttuning: Enabling generalized agent abilities for llms. arXiv preprint arXiv:2310.12823","author":"Zeng Aohan","year":"2023","unstructured":"Aohan Zeng, Mingdao Liu, Rui Lu, Bowen Wang, Xiao Liu, Yuxiao Dong, and Jie Tang. 2023. Agenttuning: Enabling generalized agent abilities for llms. arXiv preprint arXiv:2310.12823 (2023)."},{"key":"e_1_3_2_1_19_1","volume-title":"Agent-FLAN: Designing Data and Methods of Effective Agent Tuning for Large Language Models. arXiv preprint arXiv:2403.12881","author":"Chen Zehui","year":"2024","unstructured":"Zehui Chen, Kuikun Liu, Qiuchen Wang, Wenwei Zhang, Jiangning Liu, Dahua Lin, Kai Chen, and Feng Zhao. 2024. Agent-FLAN: Designing Data and Methods of Effective Agent Tuning for Large Language Models. arXiv preprint arXiv:2403.12881 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters. arXiv preprint arXiv:2408.03314","author":"Snell Charlie","year":"2024","unstructured":"Charlie Snell, Jaehoon Lee, Kelvin Xu, and Aviral Kumar. 2024. Scaling llm test-time compute optimally can be more effective than scaling model parameters. arXiv preprint arXiv:2408.03314 (2024)."},{"key":"e_1_3_2_1_21_1","volume-title":"Self-verification improves few-shot clinical information extraction. arXiv preprint arXiv:2306.00024","author":"Gero Zelalem","year":"2023","unstructured":"Zelalem Gero, Chandan Singh, Hao Cheng, Tristan Naumann, Michel Galley, Jianfeng Gao, and Hoifung Poon. 2023. Self-verification improves few-shot clinical information extraction. arXiv preprint arXiv:2306.00024 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Large language models are better reasoners with self-verification. arXiv preprint arXiv:2212.09561","author":"Weng Yixuan","year":"2022","unstructured":"Yixuan Weng, Minjun Zhu, Fei Xia, Bin Li, Shizhu He, Shengping Liu, Bin Sun, Kang Liu, and Jun Zhao. 2022. Large language models are better reasoners with self-verification. arXiv preprint arXiv:2212.09561 (2022)."},{"key":"e_1_3_2_1_23_1","volume-title":"Fan Yang, and Mao Yang.","author":"Qi Zhenting","year":"2024","unstructured":"Zhenting Qi, Mingyuan Ma, Jiahang Xu, Li Lyna Zhang, Fan Yang, and Mao Yang. 2024. Mutual Reasoning Makes Smaller LLMs Stronger Problem-Solvers. arXiv preprint arXiv:2408.06195 (2024)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2012.2186810"},{"key":"e_1_3_2_1_25_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Tom Griffiths, Yuan Cao, and Karthik Narasimhan. 2023. Tree of thoughts: Deliberate problem solving with large language models. Advances in Neural Information Processing Systems, Vol. 36 (2023)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.507"},{"key":"e_1_3_2_1_27_1","volume-title":"Alphazero-like tree-search can guide large language model decoding and training. arXiv preprint arXiv:2309.17179","author":"Feng Xidong","year":"2023","unstructured":"Xidong Feng, Ziyu Wan, Muning Wen, Ying Wen, Weinan Zhang, and Jun Wang. 2023. Alphazero-like tree-search can guide large language model decoding and training. arXiv preprint arXiv:2309.17179 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"AlphaMath Almost Zero: process Supervision without process. arXiv preprint arXiv:2405.03553","author":"Chen Guoxin","year":"2024","unstructured":"Guoxin Chen, Minpeng Liao, Chengxi Li, and Kai Fan. 2024. AlphaMath Almost Zero: process Supervision without process. arXiv preprint arXiv:2405.03553 (2024)."},{"key":"e_1_3_2_1_29_1","unstructured":"Liangchen Luo Yinxiao Liu Rosanne Liu Samrat Phatale Harsh Lara Yunxuan Li Lei Shu Yun Zhu Lei Meng Jiao Sun et al. 2024. Improve Mathematical Reasoning in Language Models by Automated Process Supervision. arXiv preprint arXiv:2406.06592 (2024)."},{"key":"e_1_3_2_1_30_1","unstructured":"Zhiheng Xi Yiwen Ding Wenxiang Chen Boyang Hong Honglin Guo Junzhe Wang Dingwen Yang Chenyang Liao Xin Guo Wei He et al. 2024. AgentGym: Evolving Large Language Model-based Agents across Diverse Environments. arXiv preprint arXiv:2406.04151 (2024)."},{"key":"e_1_3_2_1_31_1","volume-title":"Oh (Eds.)","volume":"35","author":"Zelikman Eric","year":"2022","unstructured":"Eric Zelikman, Yuhuai Wu, Jesse Mu, and Noah Goodman. 2022. STaR: Bootstrapping Reasoning With Reasoning. In Advances in Neural Information Processing Systems, S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Eds.), Vol. 35. Curran Associates, Inc., 15476--15488. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/639a9a172c044fbb64175b5fad42e9a5-Paper-Conference.pdf"},{"key":"e_1_3_2_1_32_1","volume-title":"Yifei Liu, Ning Shang, Youran Sun, Yi Zhu, Fan Yang, and Mao Yang.","author":"Guan Xinyu","year":"2025","unstructured":"Xinyu Guan, Li Lyna Zhang, Yifei Liu, Ning Shang, Youran Sun, Yi Zhu, Fan Yang, and Mao Yang. 2025. rStar-Math: Small LLMs Can Master Math Reasoning with Self-Evolved Deep Thinking. arXiv preprint arXiv:2501.04519 (2025)."},{"key":"e_1_3_2_1_33_1","volume-title":"Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning. arXiv preprint arXiv:2405.00451","author":"Xie Yuxi","year":"2024","unstructured":"Yuxi Xie, Anirudh Goyal, Wenyue Zheng, Min-Yen Kan, Timothy P Lillicrap, Kenji Kawaguchi, and Michael Shieh. 2024. Monte Carlo Tree Search Boosts Reasoning via Iterative Preference Learning. arXiv preprint arXiv:2405.00451 (2024)."},{"key":"e_1_3_2_1_34_1","volume-title":"Scaling relationship on learning mathematical reasoning with large language models. arXiv preprint arXiv:2308.01825","author":"Yuan Zheng","year":"2023","unstructured":"Zheng Yuan, Hongyi Yuan, Chengpeng Li, Guanting Dong, Chuanqi Tan, and Chang Zhou. 2023. Scaling relationship on learning mathematical reasoning with large language models. arXiv preprint arXiv:2308.01825 (2023)."},{"key":"e_1_3_2_1_35_1","volume-title":"The Curse of Recursion: Training on Generated Data Makes Models Forget. ArXiv","author":"Shumailov Ilia","year":"2023","unstructured":"Ilia Shumailov, Zakhar Shumaylov, Yiren Zhao, Yarin Gal, Nicolas Papernot, and Ross Anderson. 2023. The Curse of Recursion: Training on Generated Data Makes Models Forget. ArXiv, Vol. abs\/2305.17493 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258987240"},{"key":"e_1_3_2_1_36_1","unstructured":"Matthias Gerstgrasser Rylan Schaeffer Apratim Dey Rafael Rafailov Henry Sleight John Hughes Tomasz Korbak Rajashree Agrawal Dhruv Pai Andrey Gromov et al. 2024. Is model collapse inevitable? breaking the curse of recursion by accumulating real and synthetic data. arXiv preprint arXiv:2404.01413 (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"Progress or regress? self-improvement reversal in post-training. arXiv preprint arXiv:2407.05013","author":"Wu Ting","year":"2024","unstructured":"Ting Wu, Xuefeng Li, and Pengfei Liu. 2024. Progress or regress? self-improvement reversal in post-training. arXiv preprint arXiv:2407.05013 (2024)."},{"key":"e_1_3_2_1_38_1","volume-title":"Understanding the Effects of RLHF on LLM Generalisation and Diversity. In The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Kirk Robert","year":"2024","unstructured":"Robert Kirk, Ishita Mediratta, Christoforos Nalmpantis, Jelena Luketina, Eric Hambro, Edward Grefenstette, and Roberta Raileanu. 2024. Understanding the Effects of RLHF on LLM Generalisation and Diversity. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7--11, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=PXD3FAVHJT"},{"key":"e_1_3_2_1_39_1","volume-title":"Iterative reasoning preference optimization. arXiv preprint arXiv:2404.19733","author":"Pang Richard Yuanzhe","year":"2024","unstructured":"Richard Yuanzhe Pang, Weizhe Yuan, Kyunghyun Cho, He He, Sainbayar Sukhbaatar, and Jason Weston. 2024. Iterative reasoning preference optimization. arXiv preprint arXiv:2404.19733 (2024)."},{"key":"e_1_3_2_1_40_1","volume-title":"LLM Self-Training via Process Reward Guided Tree Search. arXiv preprint arXiv:2406.03816","author":"Zhang Dan","year":"2024","unstructured":"Dan Zhang, Sining Zhoubian, Yisong Yue, Yuxiao Dong, and Jie Tang. 2024. ReST-MCTS*: LLM Self-Training via Process Reward Guided Tree Search. arXiv preprint arXiv:2406.03816 (2024)."},{"key":"e_1_3_2_1_41_1","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume":"35","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al. 2022. Training language models to follow instructions with human feedback. Advances in Neural Information Processing Systems, Vol. 35 (2022), 27730--27744.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_42_1","volume-title":"Let's Verify Step by Step. arXiv preprint arXiv:2305.20050","author":"Lightman Hunter","year":"2023","unstructured":"Hunter Lightman, Vineet Kosaraju, Yura Burda, Harri Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe. 2023. Let's Verify Step by Step. arXiv preprint arXiv:2305.20050 (2023)."},{"key":"e_1_3_2_1_43_1","volume-title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations. arXiv preprint arXiv:2312.08935","author":"Wang Peiyi","year":"2023","unstructured":"Peiyi Wang, Lei Li, Zhihong Shao, R.X. Xu, Damai Dai, Yifei Li, Deli Chen, Y.Wu, and Zhifang Sui. 2023. Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations. arXiv preprint arXiv:2312.08935 (2023)."},{"key":"e_1_3_2_1_44_1","volume-title":"The Thirty-eighth Annual Conference on Neural Information Processing Systems.","author":"Lan Tian","unstructured":"Tian Lan, Wenwei Zhang, Chen Xu, Heyan Huang, Dahua Lin, Kai Chen, and Xian-Ling Mao. [n.,d.]. CriticEval: Evaluating Large-scale Language Model as Critic. In The Thirty-eighth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_45_1","volume-title":"CriticBench: Benchmarking LLMs for Critique-Correct Reasoning. arXiv preprint arXiv:2402.14809","author":"Lin Zicheng","year":"2024","unstructured":"Zicheng Lin, Zhibin Gou, Tian Liang, Ruilin Luo, Haowei Liu, and Yujiu Yang. 2024. CriticBench: Benchmarking LLMs for Critique-Correct Reasoning. arXiv preprint arXiv:2402.14809 (2024)."},{"key":"e_1_3_2_1_46_1","volume-title":"Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, et al.","author":"Silver David","year":"2016","unstructured":"David Silver, Aja Huang, Chris J Maddison, Arthur Guez, Laurent Sifre, George Van Den Driessche, Julian Schrittwieser, Ioannis Antonoglou, Veda Panneershelvam, Marc Lanctot, et al. 2016. Mastering the game of Go with deep neural networks and tree search. nature, Vol. 529, 7587 (2016), 484--489."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/11871842_29"},{"key":"e_1_3_2_1_48_1","volume-title":"Trial and Error: Exploration-Based Trajectory Optimization for LLM Agents. arXiv preprint arXiv:2403.02502","author":"Song Yifan","year":"2024","unstructured":"Yifan Song, Da Yin, Xiang Yue, Jie Huang, Sujian Li, and Bill Yuchen Lin. 2024. Trial and Error: Exploration-Based Trajectory Optimization for LLM Agents. arXiv preprint arXiv:2403.02502 (2024)."},{"key":"e_1_3_2_1_49_1","volume-title":"International conference on computers and games. Springer, 72--83","author":"Coulom R\u00e9mi","year":"2006","unstructured":"R\u00e9mi Coulom. 2006. Efficient selectivity and backup operators in Monte-Carlo tree search. In International conference on computers and games. Springer, 72--83."},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318."},{"key":"e_1_3_2_1_51_1","volume-title":"Generative Judge for Evaluating Alignment. In The Twelfth International Conference on Learning Representations.","author":"Li Junlong","year":"2024","unstructured":"Junlong Li, Shichao Sun, Weizhe Yuan, Run-Ze Fan, Pengfei Liu, et al. 2024. Generative Judge for Evaluating Alignment. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_52_1","volume-title":"Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al.","author":"Abdin Marah","year":"2024","unstructured":"Marah Abdin, Jyoti Aneja, Hany Awadalla, Ahmed Awadallah, Ammar Ahmad Awan, Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al. 2024. Phi-3 technical report: A highly capable language model locally on your phone. arXiv preprint arXiv:2404.14219 (2024)."}],"event":{"name":"SIGIR '25: The 48th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Padua Italy","acronym":"SIGIR '25","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3726302.3729965","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T18:34:28Z","timestamp":1755887668000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3726302.3729965"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,13]]},"references-count":52,"alternative-id":["10.1145\/3726302.3729965","10.1145\/3726302"],"URL":"https:\/\/doi.org\/10.1145\/3726302.3729965","relation":{},"subject":[],"published":{"date-parts":[[2025,7,13]]},"assertion":[{"value":"2025-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}