{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:07:14Z","timestamp":1784138834548,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808511","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"4896-4901","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Beyond Fluency: Toward Reliable Trajectories in Agentic IR"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-3189-6707","authenticated-orcid":false,"given":"Anushree","family":"Sinha","sequence":"first","affiliation":[{"name":"Google, Mountain View, California, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1358-2974","authenticated-orcid":false,"given":"Srivaths","family":"Ranganathan","sequence":"additional","affiliation":[{"name":"Google, Mountain View, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0233-4623","authenticated-orcid":false,"given":"Debanshu","family":"Das","sequence":"additional","affiliation":[{"name":"Google, Mountain View, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-7335-8233","authenticated-orcid":false,"given":"Abhishek","family":"Dharmaratnakar","sequence":"additional","affiliation":[{"name":"Google, San Bruno, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"V13, 5","author":"Bates Marcia J.","year":"1989","unstructured":"Marcia J. Bates. 1989. The Design of Browsing and Berrypicking Techniques for the Online Search Interface. Online Review, V13, 5 (1989), 407-424. https:\/\/eric.ed.gov\/?id=EJ404172"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442188.3445922"},{"key":"e_1_3_2_1_3_1","unstructured":"Samy Bengio Oriol Vinyals Navdeep Jaitly and Noam Shazeer. 2015. Scheduled Sampling for Sequence Prediction with Recurrent Neural Networks. arXiv:1506.03099 [cs.NE] https:\/\/arxiv.org\/abs\/1506.03099"},{"key":"e_1_3_2_1_4_1","unstructured":"T. Brown et al. 2020. Language Models are Few-Shot Learners. In NeurIPS 2020. https:\/\/arxiv.org\/abs\/2005.14165"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657984"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC3.2016.7880213"},{"key":"e_1_3_2_1_7_1","volume-title":"CCL 2024-23rd Chinese Natl Conf Comput Linguist","volume":"2","author":"Chen X","year":"2024","unstructured":"X Chen, A Zeng, et al., 2024. A survey on large language model based autonomous agents. In CCL 2024-23rd Chinese Natl Conf Comput Linguist, Vol. V2. 141-150. https:\/\/arxiv.org\/abs\/2308.11432"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Jeffrey Dalton Chenyan Xiong and Jamie Callan. 2020. TREC CAsT 2019: The Conversational Assistance Track Overview. arXiv:2003.13624 [cs.IR] https:\/\/arxiv.org\/abs\/2003.13624","DOI":"10.6028\/NIST.SP.1266.cast-overview"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Xiang Deng Yu Gu Boyuan Zheng Shijie Chen Samuel Stevens Boshi Wang Huan Sun and Yu Su. 2023. Mind2Web: Towards a Generalist Agent for the Web. arXiv:2306.06070 [cs.CL] https:\/\/arxiv.org\/abs\/2306.06070","DOI":"10.52202\/075280-1220"},{"key":"e_1_3_2_1_10_1","volume-title":"Generative AI for Video Trailer Synthesis: From Extractive Heuristics to Autoregressive Creativity. Authorea Preprints","author":"Dharmaratnakar Abhishek","year":"2026","unstructured":"Abhishek Dharmaratnakar, Srivaths Ranganathan, Debanshu Das, and Anushree Sinha. 2026. Generative AI for Video Trailer Synthesis: From Extractive Heuristics to Autoregressive Creativity. Authorea Preprints (2026). https:\/\/arxiv.org\/abs\/2604.04953"},{"key":"e_1_3_2_1_11_1","volume-title":"Beyond ten turns: Unlocking long-horizon agentic search with large-scale asynchronous rl. arXiv preprint arXiv:2508.07976","author":"Gao Jiaxuan","year":"2025","unstructured":"Jiaxuan Gao, Wei Fu, Minyang Xie, Shusheng Xu, Chuyi He, Zhiyu Mei, Banghua Zhu, and Yi Wu. 2025. Beyond ten turns: Unlocking long-horizon agentic search with large-scale asynchronous rl. arXiv preprint arXiv:2508.07976 (2025). https:\/\/arxiv.org\/abs\/2508.07976"},{"key":"e_1_3_2_1_12_1","unstructured":"Yonatan Geifman and Ran El-Yaniv. 2017. Selective Classification for Deep Neural Networks. In Advances in Neural Information Processing Systems. arXiv:1705.08500 https:\/\/arxiv.org\/abs\/1705.08500"},{"key":"e_1_3_2_1_13_1","volume-title":"Weinberger","author":"Guo Chuan","year":"2017","unstructured":"Chuan Guo, Geoff Pleiss, Yu Sun, and Kilian Q. Weinberger. 2017. On Calibration of Modern Neural Networks. arXiv:1706.04599 [cs.LG] https:\/\/arxiv.org\/abs\/1706.04599"},{"key":"e_1_3_2_1_14_1","volume-title":"Understanding the planning of llm agents: A survey. arXiv preprint arXiv:2402.02716","author":"Huang Xu","year":"2024","unstructured":"Xu Huang, Weiwen Liu, Xiaolong Chen, Xingmei Wang, Hao Wang, Defu Lian, Yasheng Wang, Ruiming Tang, and Enhong Chen. 2024. Understanding the planning of llm agents: A survey. arXiv preprint arXiv:2402.02716 (2024). https:\/\/arxiv.org\/abs\/2402.02716"},{"key":"e_1_3_2_1_15_1","volume-title":"MM-THEBench: Do Reasoning MLLMs Think Reasonably? arXiv preprint arXiv:2601.22735","author":"Huang Zhidian","year":"2026","unstructured":"Zhidian Huang, Zijun Yao, Ji Qi, Shangqing Tu, Junxian Ma, Jinxin Liu, Weichuan Liu, Xiaoyin Che, Lei Hou, and Juanzi Li. 2026. MM-THEBench: Do Reasoning MLLMs Think Reasonably? arXiv preprint arXiv:2601.22735 (2026). https:\/\/arxiv.org\/abs\/2601.22735"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-78646-7_4"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3571730"},{"key":"e_1_3_2_1_18_1","unstructured":"Carlos E. Jimenez John Yang Alexander Wettig Shunyu Yao Kexin Pei Ofir Press and Karthik Narasimhan. 2023. SWE-bench: Can Language Models Resolve Real-World GitHub Issues? arXiv:2310.06770 [cs.SE] https:\/\/arxiv.org\/abs\/2310.06770"},{"key":"e_1_3_2_1_19_1","volume-title":"Forty-first International Conference on Machine Learning. https:\/\/arxiv.org\/abs\/2405","author":"Kambhampati Subbarao","year":"2024","unstructured":"Subbarao Kambhampati, Karthik Valmeekam, Lin Guan, Mudit Verma, Kaya Stechly, Siddhant Bhambri, Lucas Paul Saldyt, and Anil B Murthy. 2024. Position: LLMs can't plan, but can help planning in LLM-modulo frameworks. In Forty-first International Conference on Machine Learning. https:\/\/arxiv.org\/abs\/2405.17415"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Diane Kelly. 2009. Methods for Evaluating Interactive Information Retrieval Systems with Users. Foundations and Trends in Information Retrieval. https:\/\/dl.acm.org\/doi\/10.1561\/1500000012","DOI":"10.1561\/1500000012"},{"key":"e_1_3_2_1_21_1","unstructured":"Balaji Lakshminarayanan Alexander Pritzel and Charles Blundell. 2017. Simple and Scalable Predictive Uncertainty Estimation using Deep Ensembles. arXiv:1612.01474 [stat.ML] https:\/\/arxiv.org\/abs\/1612.01474"},{"key":"e_1_3_2_1_22_1","unstructured":"Patrick Lewis Ethan Perez Aleksandra Piktus Fabio Petroni Vladimir Karpukhin Naman Goyal Heinrich K\u00fcttler Mike Lewis Wen-tau Yih Tim Rockt\u00e4schel et al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. Advances in neural information processing systems V33 (2020) 9459-9474. https:\/\/arxiv.org\/abs\/2005.11401"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"e_1_3_2_1_24_1","volume-title":"AgentHallu: Benchmarking Automated Hallucination Attribution of LLM-based Agents. arXiv preprint arXiv:2601.06818","author":"Liu Xuannan","year":"2026","unstructured":"Xuannan Liu, Xiao Yang, Zekun Li, Peipei Li, and Ran He. 2026. AgentHallu: Benchmarking Automated Hallucination Attribution of LLM-based Agents. arXiv preprint arXiv:2601.06818 (2026). https:\/\/arxiv.org\/abs\/2601.06818"},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (EMNLP). https:\/\/aclanthology.org\/2023","author":"Manakul Potsawee","unstructured":"Potsawee Manakul, Adian Liusie, and Mark J. F. Gales. 2023. SelfCheckGPT: Zero-Resource Black-Box Hallucination Detection for Generative Large Language Models. In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (EMNLP). https:\/\/aclanthology.org\/2023.emnlp-main.557\/"},{"key":"e_1_3_2_1_26_1","unstructured":"L. Ouyang et al. 2022. Training language models to follow instructions. In NeurIPS 2022. https:\/\/arxiv.org\/abs\/2203.02155"},{"key":"e_1_3_2_1_27_1","volume-title":"Gorilla: Large language model connected with massive apis. Advances in Neural Information Processing Systems, V37","author":"Patil Shishir G","year":"2024","unstructured":"Shishir G Patil, Tianjun Zhang, Xin Wang, and Joseph E Gonzalez. 2024. Gorilla: Large language model connected with massive apis. Advances in Neural Information Processing Systems, V37 (2024), 126544-126565. https:\/\/arxiv.org\/abs\/2305.15334"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2307.16789"},{"key":"e_1_3_2_1_29_1","volume-title":"Multi-Agent Video Recommenders: Evolution, Patterns, and Open Challenges. arXiv preprint arXiv:2604.02211","author":"Ranganathan Srivaths","year":"2026","unstructured":"Srivaths Ranganathan, Abhishek Dharmaratnakar, Anushree Sinha, and Debanshu Das. 2026. Multi-Agent Video Recommenders: Evolution, Patterns, and Open Challenges. arXiv preprint arXiv:2604.02211 (2026). https:\/\/arxiv.org\/abs\/2604.02211"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3705328.3748138"},{"key":"e_1_3_2_1_31_1","volume-title":"Trism for agentic ai: A review of trust, risk, and security management in llm-based agentic multi-agent systems. arXiv preprint arXiv:2506.04133","author":"Raza Shaina","year":"2025","unstructured":"Shaina Raza, Ranjan Sapkota, Manoj Karkee, and Christos Emmanouilidis. 2025. Trism for agentic ai: A review of trust, risk, and security management in llm-based agentic multi-agent systems. arXiv preprint arXiv:2506.04133 (2025). https:\/\/arxiv.org\/abs\/2506.04133"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics (AISTATS). https:\/\/proceedings.mlr.press\/v15\/ross11a.html","author":"Ross St'ephane","unstructured":"St'ephane Ross, Geoffrey J. Gordon, and J. Andrew Bagnell. 2011. A Reduction of Imitation Learning and Structured Prediction to No-Regret Online Learning. In Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics (AISTATS). https:\/\/proceedings.mlr.press\/v15\/ross11a.html"},{"key":"e_1_3_2_1_33_1","volume-title":"Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom.","author":"Schick Timo","year":"2023","unstructured":"Timo Schick, Jane Dwivedi-Yu, Roberto Dess`i, Roberta Raileanu, Maria Lomeli, Eric Hambro, Luke Zettlemoyer, Nicola Cancedda, and Thomas Scialom. 2023. Toolformer: Language models can teach themselves to use tools. Advances in neural information processing systems, V36 (2023), 68539-68551. https:\/\/arxiv.org\/abs\/2302.04761"},{"key":"e_1_3_2_1_34_1","volume-title":"Reflexion: Language Agents with Verbal Reinforcement Learning. In Advances in Neural Information Processing Systems. https:\/\/arxiv.org\/abs\/2303.11366","author":"Shinn Noah","year":"2023","unstructured":"Noah Shinn, Jeffrey Labash, Ashwin Gopinath, Manya Wadhwa, Pratyusha Kumar, Yiming Yang, Ameya Joshi, Shunyu Yao, et al., 2023. Reflexion: Language Agents with Verbal Reinforcement Learning. In Advances in Neural Information Processing Systems. https:\/\/arxiv.org\/abs\/2303.11366"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 35th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR). https:\/\/dl.acm.org\/doi\/10","author":"Mark","unstructured":"Mark D. Smucker and Charles L. A. Clarke. 2012. Time-based Calibration of Effectiveness Measures. In Proceedings of the 35th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR). https:\/\/dl.acm.org\/doi\/10.1145\/2348283.2348300"},{"key":"e_1_3_2_1_36_1","volume-title":"Llama: Open and Efficient Foundation Models. arXiv:2302.13971","author":"H. Touvron","year":"2023","unstructured":"H. Touvron et al., 2023. Llama: Open and Efficient Foundation Models. arXiv:2302.13971 (2023). https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-81-322-2755-7_21"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.813"},{"key":"e_1_3_2_1_39_1","volume-title":"Jiashuo Wang, Jian Wang, and Wenjie Li.","author":"Wang Hanlin","year":"2025","unstructured":"Hanlin Wang, Chak Tou Leong, Jiashuo Wang, Jian Wang, and Wenjie Li. 2025b. Spa-rl: Reinforcing llm agents via stepwise progress attribution. arXiv preprint arXiv:2505.20732 (2025). https:\/\/arxiv.org\/abs\/2505.20732"},{"key":"e_1_3_2_1_40_1","unstructured":"J. Wang et al. 2025a. UltraHorizon: Evaluating Agents in Long-Horizon Tasks. arXiv:2509.21766 (2025). https:\/\/arxiv.org\/abs\/2509.21766"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Jerry Wei Chengrun Yang Xinying Song Yifeng Lu Nathan Hu Jie Huang Dustin Tran Daiyi Peng Ruibo Liu Da Huang et al. 2024. Long-form factuality in large language models. Advances in Neural Information Processing Systems V37 (2024) 80756-80827. https:\/\/arxiv.org\/abs\/2403.18802","DOI":"10.52202\/079017-2567"},{"key":"e_1_3_2_1_42_1","unstructured":"Fangzhi Xu Hang Yan Qiushi Sun Jinyang Wu Zixian Huang Muye Huang Jingyang Gong Zichen Ding Kanzhi Cheng Yian Wang et al. [n.d.]. OdysseyArena: Benchmarking Large Language Models For Long-Horizon Active and Inductive Interactions. arXiv preprint arXiv:2602.05843 ([n.d.]). https:\/\/arxiv.org\/abs\/2602.05843"},{"key":"e_1_3_2_1_43_1","volume-title":"Hallucination is inevitable: An innate limitation of large language models. arXiv preprint arXiv:2401.11817","author":"Xu Ziwei","year":"2024","unstructured":"Ziwei Xu, Sanjay Jain, and Mohan Kankanhalli. 2024. Hallucination is inevitable: An innate limitation of large language models. arXiv preprint arXiv:2401.11817 (2024). https:\/\/arxiv.org\/abs\/2401.11817"},{"key":"e_1_3_2_1_44_1","unstructured":"Shunyu Yao Howard Chen John Yang and Karthik Narasimhan. 2022a. WebShop: Towards Scalable Real-World Web Interaction with Grounded Language Agents. In Advances in Neural Information Processing Systems. arXiv:2207.01206 https:\/\/arxiv.org\/abs\/2207.01206"},{"key":"e_1_3_2_1_45_1","volume-title":"React: Synergizing reasoning and acting in language models. In The eleventh international conference on learning representations. https:\/\/arxiv.org\/abs\/2210.03629","author":"Yao Shunyu","year":"2022","unstructured":"Shunyu Yao, Jeffrey Zhao, Dian Yu, Nan Du, Izhak Shafran, Karthik R Narasimhan, and Yuan Cao. 2022b. React: Synergizing reasoning and acting in language models. In The eleventh international conference on learning representations. https:\/\/arxiv.org\/abs\/2210.03629"},{"key":"e_1_3_2_1_46_1","volume-title":"The reasoning trap: How enhancing LLM reasoning amplifies tool hallucination. arXiv preprint arXiv:2510.22977","author":"Yin Chenlong","year":"2025","unstructured":"Chenlong Yin, Zeyang Sha, Shiwen Cui, and Changhua Meng. 2025. The reasoning trap: How enhancing LLM reasoning amplifies tool hallucination. arXiv preprint arXiv:2510.22977 (2025). https:\/\/arxiv.org\/abs\/2510.22977"},{"key":"e_1_3_2_1_47_1","volume-title":"InfiAgent: An Infinite-Horizon Framework for General-Purpose Autonomous Agents. arXiv preprint arXiv:2601.03204","author":"Yu Chenglin","year":"2026","unstructured":"Chenglin Yu, Yuchen Wang, Songmiao Wang, Hongxia Yang, and Ming Li. 2026. InfiAgent: An Infinite-Horizon Framework for General-Purpose Autonomous Agents. arXiv preprint arXiv:2601.03204 (2026). https:\/\/arxiv.org\/abs\/2601.03204"},{"key":"e_1_3_2_1_48_1","volume-title":"Agentic information retrieval. arXiv preprint arXiv:2410.09713","author":"Zhang Weinan","year":"2024","unstructured":"Weinan Zhang, Junwei Liao, Ning Li, Kounianhua Du, and Jianghao Lin. 2024b. Agentic information retrieval. arXiv preprint arXiv:2410.09713 (2024). https:\/\/arxiv.org\/abs\/2410.09713"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.637"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3121050.3121070"},{"key":"e_1_3_2_1_51_1","volume-title":"Memory as action: Autonomous context curation for long-horizon agentic tasks. arXiv preprint arXiv:2510.12635","author":"Zhang Yuxiang","year":"2025","unstructured":"Yuxiang Zhang, Jiangming Shu, Ye Ma, Xueyuan Lin, Shangxi Wu, and Jitao Sang. 2025. Memory as action: Autonomous context curation for long-horizon agentic tasks. arXiv preprint arXiv:2510.12635 (2025). https:\/\/arxiv.org\/abs\/2510.12635"},{"key":"e_1_3_2_1_52_1","volume-title":"Guardian: Safeguarding llm multi-agent collaborations with temporal graph modeling. arXiv preprint arXiv:2505.19234","author":"Zhou Jialong","year":"2025","unstructured":"Jialong Zhou, Lichao Wang, and Xiao Yang. 2025. Guardian: Safeguarding llm multi-agent collaborations with temporal graph modeling. arXiv preprint arXiv:2505.19234 (2025). https:\/\/arxiv.org\/abs\/2505.19234"},{"key":"e_1_3_2_1_53_1","unstructured":"Shuyan Zhou Frank F. Xu Hao Zhu Xuhui Zhou Robert Lo Abishek Sridhar Xianyi Cheng Tianyue Ou Yonatan Bisk Daniel Fried Uri Alon and Graham Neubig. 2023. WebArena: A Realistic Web Environment for Building Autonomous Agents. arXiv:2307.13854 [cs.CL] https:\/\/arxiv.org\/abs\/2307.13854"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:23:23Z","timestamp":1784136203000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808511"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":53,"alternative-id":["10.1145\/3805712.3808511","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808511","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}