{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:19:24Z","timestamp":1784179164089,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62502404"],"award-info":[{"award-number":["62502404"]}]},{"name":"Hong Kong Research Grants Council&#x28;Research Impact Fund&#x29;","award":["R1015-23"],"award-info":[{"award-number":["R1015-23"]}]},{"name":"Hong Kong Research Grants Council&#x28;Collaborative Research Fund&#x29;","award":["C1043-24GF"],"award-info":[{"award-number":["C1043-24GF"]}]},{"name":"Hong Kong Research Grants Council&#x28;General Research Fund&#x29;","award":["11218325"],"award-info":[{"award-number":["11218325"]}]},{"name":"Institute of Digital Medicine of City University of Hong Kong","award":["9229503"],"award-info":[{"award-number":["9229503"]}]},{"name":"Huawei Innovation Research Program","award":["N&#x5c;&#x2f;A"],"award-info":[{"award-number":["N&#x5c;&#x2f;A"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792235","type":"proceedings-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T13:28:36Z","timestamp":1777296516000},"page":"2049-2059","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["To Search or Not to Search: Aligning the Decision Boundary of Deep Search Agents via Causal Intervention"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1809-8264","authenticated-orcid":false,"given":"Wenlin","family":"Zhang","sequence":"first","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5564-0641","authenticated-orcid":false,"given":"Kuicai","family":"Dong","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-0480-5593","authenticated-orcid":false,"given":"Junyi","family":"Li","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9062-3428","authenticated-orcid":false,"given":"Yingyi","family":"Zhang","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6162-8500","authenticated-orcid":false,"given":"Xiaopeng","family":"Li","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4712-3676","authenticated-orcid":false,"given":"Pengyue","family":"Jia","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5924-1429","authenticated-orcid":false,"given":"Yi","family":"Wen","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3971-9907","authenticated-orcid":false,"given":"Derong","family":"Xu","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0073-0172","authenticated-orcid":false,"given":"Maolin","family":"Wang","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7053-8269","authenticated-orcid":false,"given":"Yichao","family":"Wang","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9031-9696","authenticated-orcid":false,"given":"Yong","family":"Liu","sequence":"additional","affiliation":[{"name":"Huawei Technologies Ltd., Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2926-4416","authenticated-orcid":false,"given":"Xiangyu","family":"Zhao","sequence":"additional","affiliation":[{"name":"City University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"International Conference on Learning Representations.","author":"Asai Akari","year":"2024","unstructured":"Akari Asai, Zeqiu Wu, Yizhong Wang, Avi Sil, and Hannaneh Hajishirzi. 2024. Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_2_1","unstructured":"Gheorghe Comanici Eric Bieber Mike Schaekermann Ice Pasupat Noveen Sachdeva Inderjit Dhillon Marcel Blistein Ori Ram Dan Zhang Evan Rosen et al. 2025. Gemini 2.5: Pushing the frontier with advanced reasoning multimodality long context and next generation agentic capabilities. arXiv preprint arXiv:2507.06261 (2025)."},{"key":"e_1_3_2_1_3_1","volume-title":"Temporal working memory: Query-guided segment refinement for enhanced multimodal understanding. arXiv preprint arXiv:2502.06020","author":"Diao Xingjian","year":"2025","unstructured":"Xingjian Diao, Chunhui Zhang, Weiyi Wu, Zhongyu Ouyang, Peijun Qing, Ming Cheng, Soroush Vosoughi, and Jiang Gui. 2025. Temporal working memory: Query-guided segment refinement for enhanced multimodal understanding. arXiv preprint arXiv:2502.06020 (2025)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.1576"},{"key":"e_1_3_2_1_5_1","unstructured":"Kuicai Dong Yujing Chang Shijie Huang Yasheng Wang Ruiming Tang and Yong Liu. 2025b. Benchmarking Retrieval-Augmented Multimodal Generation for Document Question Answering."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Kuicai Dong Shurui Huang Fangda Ye Wei Han Zhi Zhang Dexun Li Wenjun Li Qu Yang Gang Wang Yichao Wang Chen Zhang and Yong Liu. 2025c. Doc-Researcher: A Unified System for Multimodal Document Parsing and Deep Research.","DOI":"10.1145\/3774904.3792599"},{"key":"e_1_3_2_1_7_1","volume-title":"A unified framework for multi-domain ctr prediction via large language models. ACM Transactions on Information Systems","author":"Fu Zichuan","year":"2023","unstructured":"Zichuan Fu, Xiangyang Li, Chuhan Wu, Yichao Wang, Kuicai Dong, Xiangyu Zhao, Mengchen Zhao, Huifeng Guo, and Ruiming Tang. 2023. A unified framework for multi-domain ctr prediction via large language models. ACM Transactions on Information Systems (2023)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-025-09422-z"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.580"},{"key":"e_1_3_2_1_10_1","unstructured":"Yuxuan Huang Yihang Chen Haozheng Zhang Kang Li Huichi Zhou Meng Fang Linyi Yang Xiaoguang Li Lifeng Shang Songcen Xu et al. 2025. Deep research agents: A systematic examination and roadmap. arXiv preprint arXiv:2506.18096 (2025)."},{"key":"e_1_3_2_1_11_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.389"},{"key":"e_1_3_2_1_13_1","volume-title":"Deepretrieval: Hacking real search engines and retrievers with large language models via reinforcement learning. arXiv preprint arXiv:2503.00223","author":"Jiang Pengcheng","year":"2025","unstructured":"Pengcheng Jiang, Jiacheng Lin, Lang Cao, Runchu Tian, SeongKu Kang, Zifeng Wang, Jimeng Sun, and Jiawei Han. 2025. Deepretrieval: Hacking real search engines and retrievers with large language models via reinforcement learning. arXiv preprint arXiv:2503.00223 (2025)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.495"},{"key":"e_1_3_2_1_15_1","volume-title":"Search-r1: Training llms to reason and leverage search engines with reinforcement learning. arXiv preprint arXiv:2503.09516","author":"Jin Bowen","year":"2025","unstructured":"Bowen Jin, Hansi Zeng, Zhenrui Yue, Jinsung Yoon, Sercan Arik, Dong Wang, Hamed Zamani, and Jiawei Han. 2025a. Search-r1: Training llms to reason and leverage search engines with reinforcement learning. arXiv preprint arXiv:2503.09516 (2025)."},{"key":"e_1_3_2_1_16_1","volume-title":"Your reward function for rl is your best prm for search: Unifying rl and search-based tts. arXiv preprint arXiv:2508.14313","author":"Jin Can","year":"2025","unstructured":"Can Jin, Yang Zhou, Qixin Zhang, Hongwu Peng, Di Zhang, Marco Pavone, Ligong Han, Zhang-Wei Hong, Tong Che, and Dimitris N Metaxas. 2025b. Your reward function for rl is your best prm for search: Unifying rl and search-based tts. arXiv preprint arXiv:2508.14313 (2025)."},{"key":"e_1_3_2_1_17_1","unstructured":"Saurav Kadavath Tom Conerly Amanda Askell Tom Henighan Dawn Drain Ethan Perez Nicholas Schiefer Zac Hatfield-Dodds Nova DasSarma Eli Tran-Johnson et al. 2022. Language models (mostly) know what they know. arXiv preprint arXiv:2207.05221 (2022)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00276"},{"key":"e_1_3_2_1_19_1","unstructured":"Patrick Lewis Ethan Perez Aleksandra Piktus Fabio Petroni Vladimir Karpukhin Naman Goyal Heinrich K\u00fcttler Mike Lewis Wen-tau Yih Tim Rockt\u00e4schel et al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. Advances in neural information processing systems Vol. 33 (2020) 9459-9474."},{"key":"e_1_3_2_1_20_1","unstructured":"Kuan Li Zhongwang Zhang Huifeng Yin Liwen Zhang Litu Ou Jialong Wu Wenbiao Yin Baixuan Li Zhengwei Tao Xinyu Wang et al. 2025d. WebSailor: Navigating Super-human Reasoning for Web Agent. arXiv preprint arXiv:2507.02592 (2025)."},{"key":"e_1_3_2_1_21_1","volume-title":"Knowledge boundary of large language models: A survey. arXiv preprint arXiv:2412.12472","author":"Li Moxin","year":"2024","unstructured":"Moxin Li, Yong Zhao, Wenxuan Zhang, Shuaiyi Li, Wenya Xie, See-Kiong Ng, Tat-Seng Chua, and Yang Deng. 2024. Knowledge boundary of large language models: A survey. arXiv preprint arXiv:2412.12472 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"Search-o1: Agentic search-enhanced large reasoning models. arXiv preprint arXiv:2501.05366","author":"Li Xiaoxi","year":"2025","unstructured":"Xiaoxi Li, Guanting Dong, Jiajie Jin, Yuyao Zhang, Yujia Zhou, Yutao Zhu, Peitian Zhang, and Zhicheng Dou. 2025b. Search-o1: Agentic search-enhanced large reasoning models. arXiv preprint arXiv:2501.05366 (2025)."},{"key":"e_1_3_2_1_23_1","unstructured":"Xiaopeng Li Pengyue Jia Derong Xu Yi Wen Yingyi Zhang Wenlin Zhang Wanyu Wang Yichao Wang Zhaocheng Du Xiangyang Li et al. 2025c. A survey of personalization: From rag to agent. arXiv preprint arXiv:2504.10147 (2025)."},{"key":"e_1_3_2_1_24_1","volume-title":"Agent4ranking: Semantic robust ranking via personalized query rewriting using multi-agent llm. arXiv preprint arXiv:2312.15450","author":"Li Xiaopeng","year":"2023","unstructured":"Xiaopeng Li, Lixin Su, Pengyue Jia, Xiangyu Zhao, Suqi Cheng, Junfeng Wang, and Dawei Yin. 2023. Agent4ranking: Semantic robust ranking via personalized query rewriting using multi-agent llm. arXiv preprint arXiv:2312.15450 (2023)."},{"key":"e_1_3_2_1_25_1","unstructured":"Yuchen Li Hengyi Cai Rui Kong Xinran Chen Jiamin Chen Jun Yang Haojie Zhang Jiayi Li Jiayi Wu Yiqun Chen et al. 2025a. Towards AI Search Paradigm. arXiv preprint arXiv:2506.17188 (2025)."},{"key":"e_1_3_2_1_26_1","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et al. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437 (2024)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583244"},{"key":"e_1_3_2_1_28_1","volume-title":"Deepdive: Advancing deep search agents with knowledge graphs and multi-turn rl. arXiv preprint arXiv:2509.10446","author":"Lu Rui","year":"2025","unstructured":"Rui Lu, Zhenyu Hou, Zihan Wang, Hanchen Zhang, Xiao Liu, Yujiang Li, Shi Feng, Jie Tang, and Yuxiao Dong. 2025. Deepdive: Advancing deep search agents with knowledge graphs and multi-turn rl. arXiv preprint arXiv:2509.10446 (2025)."},{"key":"e_1_3_2_1_29_1","first-page":"11375","article-title":"When Do LLMs Need Retrieval Augmentation","volume":"2024","author":"Ni Shiyu","year":"2024","unstructured":"Shiyu Ni, Keping Bi, Jiafeng Guo, and Xueqi Cheng. 2024. When Do LLMs Need Retrieval Augmentation? Mitigating LLMs' Overconfidence Helps Retrieval Augmentation. In Findings of the Association for Computational Linguistics ACL 2024. 11375-11388.","journal-title":"Mitigating LLMs' Overconfidence Helps Retrieval Augmentation. In Findings of the Association for Computational Linguistics ACL"},{"key":"e_1_3_2_1_30_1","unstructured":"Judea Pearl. 2009. Causality. Cambridge university press."},{"key":"e_1_3_2_1_31_1","volume-title":"Scent of Knowledge: Optimizing Search-Enhanced Reasoning with Information Foraging. arXiv preprint arXiv:2505.09316","author":"Qian Hongjin","year":"2025","unstructured":"Hongjin Qian and Zheng Liu. 2025. Scent of Knowledge: Optimizing Search-Enhanced Reasoning with Information Foraging. arXiv preprint arXiv:2505.09316 (2025)."},{"key":"e_1_3_2_1_32_1","unstructured":"Zile Qiao Guoxin Chen Xuanzhong Chen Donglei Yu Wenbiao Yin Xinyu Wang Zhen Zhang Baixuan Li Huifeng Yin Kuan Li et al. 2025. WebResearcher: Unleashing unbounded reasoning capability in Long-Horizon Agents. arXiv preprint arXiv:2509.13309 (2025)."},{"key":"e_1_3_2_1_33_1","volume-title":"Alphalora: Assigning lora experts based on layer training quality.","author":"Qing Peijun","year":"2024","unstructured":"Peijun Qing, Chongyang Gao, Yefan Zhou, Xingjian Diao, Yaoqing Yang, and Soroush Vosoughi. 2024. Alphalora: Assigning lora experts based on layer training quality. (2024)."},{"key":"e_1_3_2_1_34_1","volume-title":"Semantic density: Uncertainty quantification for large language models through confidence measurement in semantic space. Advances in neural information processing systems","author":"Qiu Xin","year":"2024","unstructured":"Xin Qiu and Risto Miikkulainen. 2024. Semantic density: Uncertainty quantification for large language models through confidence measurement in semantic space. Advances in neural information processing systems, Vol. 37 (2024), 134507-134533."},{"key":"e_1_3_2_1_35_1","volume-title":"Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems","author":"Rafailov Rafael","year":"2023","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2023. Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems, Vol. 36 (2023), 53728-53741."},{"key":"e_1_3_2_1_36_1","volume-title":"Cognitive offloading. Trends in cognitive sciences","author":"Risko Evan F","year":"2016","unstructured":"Evan F Risko and Sam J Gilbert. 2016. Cognitive offloading. Trends in cognitive sciences, Vol. 20, 9 (2016), 676-688."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.119"},{"key":"e_1_3_2_1_38_1","volume-title":"Lei Fang, and Ji-Rong Wen.","author":"Song Huatong","year":"2025","unstructured":"Huatong Song, Jinhao Jiang, Yingqian Min, Jie Chen, Zhipeng Chen, Wayne Xin Zhao, Lei Fang, and Ji-Rong Wen. 2025. R1-searcher: Incentivizing the search capability in llms via reinforcement learning. arXiv preprint arXiv:2503.05592 (2025)."},{"key":"e_1_3_2_1_39_1","volume-title":"Zerosearch: Incentivize the search capability of llms without searching. arXiv preprint arXiv:2505.04588","author":"Sun Hao","year":"2025","unstructured":"Hao Sun, Zile Qiao, Jiayan Guo, Xuanbo Fan, Yingyan Hou, Yong Jiang, Pengjun Xie, Yan Zhang, Fei Huang, and Jingren Zhou. 2025. Zerosearch: Incentivize the search capability of llms without searching. arXiv preprint arXiv:2505.04588 (2025)."},{"key":"e_1_3_2_1_40_1","unstructured":"Qwen Team et al. 2024. Qwen2 technical report. arXiv preprint arXiv:2407.10671 Vol. 2 (2024) 3."},{"key":"e_1_3_2_1_41_1","volume-title":"Acting Less is Reasoning More! Teaching Model to Act Efficiently. arXiv preprint arXiv:2504.14870","author":"Wang Hongru","year":"2025","unstructured":"Hongru Wang, Cheng Qian, Wanjun Zhong, Xiusi Chen, Jiahao Qiu, Shijue Huang, Bowen Jin, Mengdi Wang, Kam-Fai Wong, and Heng Ji. 2025a. Acting Less is Reasoning More! Teaching Model to Act Efficiently. arXiv preprint arXiv:2504.14870 (2025)."},{"key":"e_1_3_2_1_42_1","volume-title":"Otc: Optimal tool calls via reinforcement learning. arXiv e-prints","author":"Wang Hongru","year":"2025","unstructured":"Hongru Wang, Cheng Qian, Wanjun Zhong, Xiusi Chen, Jiahao Qiu, Shijue Huang, Bowen Jin, Mengdi Wang, Kam-Fai Wong, and Heng Ji. 2025b. Otc: Optimal tool calls via reinforcement learning. arXiv e-prints (2025), arXiv-2504."},{"key":"e_1_3_2_1_43_1","volume-title":"The Eleventh International Conference on Learning Representations (ICLR). https:\/\/openreview.net\/forum?id=d_h-qG_t-a","author":"Wang Liang","year":"2023","unstructured":"Liang Wang, Nan Yang, Xiaolong Huang, Binxing Jiao, Linjun Yang, Daxin Jiang, Rangan Majumder, and Furu Wei. 2023. Text Embeddings by Weakly-Supervised Contrastive Pre-training. In The Eleventh International Conference on Learning Representations (ICLR). https:\/\/openreview.net\/forum?id=d_h-qG_t-a"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714949"},{"key":"e_1_3_2_1_45_1","volume-title":"StepSearch: Igniting LLMs Search Ability via Step-Wise Proximal Policy Optimization. arXiv preprint arXiv:2505.15107","author":"Wang Ziliang","year":"2025","unstructured":"Ziliang Wang, Xuhui Zheng, Kang An, Cijun Ouyang, Jialu Cai, Yuhang Wang, and Yichao Wu. 2025d. StepSearch: Igniting LLMs Search Ability via Step-Wise Proximal Policy Optimization. arXiv preprint arXiv:2505.15107 (2025)."},{"key":"e_1_3_2_1_46_1","unstructured":"Wikimedia Foundation. 2025. English Wikipedia Dump (2025-09-01). https:\/\/dumps.wikimedia.org\/enwiki\/20250901\/. Accessed: 2025-10-08."},{"key":"e_1_3_2_1_47_1","volume-title":"A survey of llm-based deep search agents: Paradigm, optimization, evaluation, and challenges. arXiv preprint arXiv:2508.05668","author":"Xi Yunjia","year":"2025","unstructured":"Yunjia Xi, Jianghao Lin, Yongzhao Xiao, Zheli Zhou, Rong Shan, Te Gao, Jiachen Zhu, Weiwen Liu, Yong Yu, and Weinan Zhang. 2025. A survey of llm-based deep search agents: Paradigm, optimization, evaluation, and challenges. arXiv preprint arXiv:2508.05668 (2025)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i24.34747"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.63317\/3vxue5y8m4xy"},{"key":"e_1_3_2_1_50_1","volume-title":"Enhancing tool retrieval with iterative feedback from large language models. arXiv preprint arXiv:2406.17465","author":"Xu Qiancheng","year":"2024","unstructured":"Qiancheng Xu, Yongqi Li, Heming Xia, and Wenjie Li. 2024a. Enhancing tool retrieval with iterative feedback from large language models. arXiv preprint arXiv:2406.17465 (2024)."},{"key":"e_1_3_2_1_51_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et al. 2025. Qwen3 technical report. arXiv preprint arXiv:2505.09388 (2025)."},{"key":"e_1_3_2_1_52_1","unstructured":"Dong Yang Peijun Qing Yang Li Haonan Lu and Xiaodong Lin. [n.d.]. Gammae: Gamma embeddings for logical queries on knowledge graphs. ([n.d.])."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1259"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.124"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.551"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3690624.3709440"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"crossref","unstructured":"Dingchu Zhang Yida Zhao Jialong Wu Baixuan Li Wenbiao Yin Liwen Zhang Yong Jiang Yufeng Li Kewei Tu Pengjun Xie et al. 2025c. EvolveSearch: An Iterative Self-Evolving Search Agent. arXiv preprint arXiv:2505.22501 (2025).","DOI":"10.18653\/v1\/2025.emnlp-main.663"},{"key":"e_1_3_2_1_58_1","volume-title":"Outcome Reward: Which is Better for Agentic RAG Reinforcement Learning. arXiv preprint arXiv:2505.14069","author":"Zhang Wenlin","year":"2025","unstructured":"Wenlin Zhang, Xiangyang Li, Kuicai Dong, Yichao Wang, Pengyue Jia, Xiaopeng Li, Yingyi Zhang, Derong Xu, Zhaocheng Du, Huifeng Guo, et al., 2025a. Process vs. Outcome Reward: Which is Better for Agentic RAG Reinforcement Learning. arXiv preprint arXiv:2505.14069 (2025)."},{"key":"e_1_3_2_1_59_1","unstructured":"Wenlin Zhang Xiangyang Li Qiyuan Ge Kuicai Dong Pengyue Jia Xiaopeng Li Zijian Zhang Maolin Wang Yichao Wang Huifeng Guo et al. 2026. Exploring Recommender System Evaluation: A Multi-Modal User Agent Framework for A\/B Testing. arXiv preprint arXiv:2601.04554 (2026)."},{"key":"e_1_3_2_1_60_1","volume-title":"Agentic information retrieval. arXiv preprint arXiv:2410.09713","author":"Zhang Weinan","year":"2024","unstructured":"Weinan Zhang, Junwei Liao, Ning Li, Kounianhua Du, and Jianghao Lin. 2024. Agentic information retrieval. arXiv preprint arXiv:2410.09713 (2024)."},{"key":"e_1_3_2_1_61_1","volume-title":"Deep Reinforcement Learning for Search, Recommendation, and Online Advertising: A Survey. ACM sigweb newsletter","author":"Zhao Xiangyu","year":"2019","unstructured":"Xiangyu Zhao, Long Xia, Jiliang Tang, and Dawei Yin. 2019. Deep Reinforcement Learning for Search, Recommendation, and Online Advertising: A Survey. ACM sigweb newsletter, Vol. 2019, Spring (2019), 1-15."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240323.3240374"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219886"}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792235","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:49:28Z","timestamp":1783151368000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792235"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":63,"alternative-id":["10.1145\/3774904.3792235","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792235","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}