{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T04:06:25Z","timestamp":1779422785726,"version":"3.53.1"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,5,26]],"date-time":"2026-05-26T00:00:00Z","timestamp":1779753600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,5,26]]},"DOI":"10.1145\/3786335.3813131","type":"proceedings-article","created":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T03:16:22Z","timestamp":1779419782000},"page":"1084-1099","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SEAR: Schema-Based Evaluation and Routing for LLM Gateways"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9240-1829","authenticated-orcid":false,"given":"Zecheng","family":"Zhang","sequence":"first","affiliation":[{"name":"Strukto.AI, San Mateo, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6668-8145","authenticated-orcid":false,"given":"Han","family":"Zheng","sequence":"additional","affiliation":[{"name":"Infron.AI, San Mateo, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0924-3567","authenticated-orcid":false,"given":"Yue","family":"Xu","sequence":"additional","affiliation":[{"name":"Infron.AI, San Mateo, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,5,26]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"CIDR 2025","year":"2025","unstructured":"2025. AOP: Automated LLM Pipeline Orchestration. In CIDR 2025. https:\/\/vldb.org\/cidrdb\/papers\/2025\/p32-wang.pdf"},{"key":"e_1_3_3_1_3_2","unstructured":"2025. KARMA: Multi-Agent LLMs for Automated Knowledge Graph Enrichment. OpenReview. https:\/\/openreview.net\/pdf?id=k0wyi4cOGy"},{"key":"e_1_3_3_1_4_2","unstructured":"Anthropic. 2025. Structured Outputs \u2014 Claude API Documentation. Anthropic Developer Docs. https:\/\/docs.anthropic.com\/en\/docs\/build-with-claude\/structured-outputs"},{"key":"e_1_3_3_1_5_2","unstructured":"Malika Aubakirova Alex Atallah Chris Clark Justin Summerville and Anjney Midha. 2026. State of AI: An Empirical 100 Trillion Token Study with OpenRouter. arxiv:https:\/\/arXiv.org\/abs\/2601.10088\u00a0[cs.AI]"},{"key":"e_1_3_3_1_6_2","unstructured":"Sourav Banerjee Ayushi Agarwal and Eishkaran Singh. 2024. The Vulnerability of Language Model Benchmarks: Do They Accurately Reflect True LLM Performance? arxiv:https:\/\/arXiv.org\/abs\/2412.03597\u00a0[cs.CL]"},{"key":"e_1_3_3_1_7_2","unstructured":"Lingjiao Chen Matei Zaharia and James Zou. 2023. FrugalGPT: How to Use Large Language Models While Reducing Cost and Improving Performance. arxiv:https:\/\/arXiv.org\/abs\/2305.05176\u00a0[cs.AI]"},{"key":"e_1_3_3_1_8_2","unstructured":"Wei-Lin Chiang Lianmin Zheng Ying Sheng Zhanghao Wu Yonghao Zhuang Zi Lin Dacheng Li Eric\u00a0P. Xing Hao Zhang Joseph\u00a0E. Gonzalez and Ion Stoica. 2024. Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference. arxiv:https:\/\/arXiv.org\/abs\/2403.04132\u00a0[cs.AI]"},{"key":"e_1_3_3_1_9_2","unstructured":"Zhongjie Dai Tao Feng and Jiaxuan You. [n. d.]. PersonalizedRouter: Personalized LLM Routing via Graph-based User Preference Modeling. Transactions on Machine Learning Research ([n. d.])."},{"key":"e_1_3_3_1_10_2","volume-title":"The Twelfth International Conference on Learning Representations","author":"Ding Dujian","unstructured":"Dujian Ding, Ankur Mallick, Chi Wang, Robert Sim, Subhabrata Mukherjee, Victor R\u00fchle, Laks\u00a0VS Lakshmanan, and Ahmed\u00a0Hassan Awadallah. [n. d.]. Hybrid LLM: Cost-Efficient and Quality-Aware Query Routing. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_3_1_11_2","unstructured":"Dujian Ding Ankur Mallick Chi Wang Robert Sim Subhabrata Mukherjee Victor Ruhle Laks V.\u00a0S. Lakshmanan and Ahmed\u00a0Hassan Awadallah. 2024. Hybrid LLM: Cost-Efficient and Quality-Aware Query Routing. arxiv:https:\/\/arXiv.org\/abs\/2404.14618\u00a0[cs.CL]"},{"key":"e_1_3_3_1_12_2","unstructured":"Liming Dong Qinghua Lu and Liming Zhu. 2024. AgentOps: Enabling Observability of LLM Agents. arxiv:https:\/\/arXiv.org\/abs\/2411.05285\u00a0[cs.SE]"},{"key":"e_1_3_3_1_13_2","unstructured":"Yi Dong Ronghui Mu Gaojie Jin Yi Qi Jinwei Hu Xingyu Zhao Jie Meng Wenjie Ruan and Xiaowei Huang. 2024. Building Guardrails for Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2402.01822\u00a0[cs.CL]"},{"key":"e_1_3_3_1_14_2","unstructured":"Yixin Dong Charlie\u00a0F. Ruan Yaxing Cai Ruihang Lai Ziyi Xu Yilong Zhao and Tianqi Chen. 2024. XGrammar: Flexible and Efficient Structured Generation Engine for Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2411.15100\u00a0[cs.CL]"},{"key":"e_1_3_3_1_15_2","unstructured":"Yufeng Du Minyang Tian Srikanth Ronanki Subendhu Rongali Sravan Bodapati Aram Galstyan Azton Wells Roy Schwartz Eliu\u00a0A. Huerta and Hao Peng. 2025. Context Length Alone Hurts LLM Performance Despite Perfect Retrieval. arxiv:https:\/\/arXiv.org\/abs\/2510.05381\u00a0[cs.CL]"},{"key":"e_1_3_3_1_16_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Feng Tao","unstructured":"Tao Feng, Yanzhen Shen, and Jiaxuan You. [n. d.]. GraphRouter: A Graph-based Router for LLM Selections. In The Thirteenth International Conference on Learning Representations."},{"key":"e_1_3_3_1_17_2","unstructured":"James Fodor. 2025. Line Goes Up? Inherent Limitations of Benchmarks for Evaluating Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2502.14318\u00a0[cs.CL]"},{"key":"e_1_3_3_1_18_2","unstructured":"Jinlan Fu See-Kiong Ng Zhengbao Jiang and Pengfei Liu. 2023. GPTScore: Evaluate as You Desire. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2302.04166 (2023)."},{"key":"e_1_3_3_1_19_2","unstructured":"Saibo Geng Hudson Cooper Micha\u0142 Moskal Samuel Jenkins Julian Berman Nathan Ranchin Robert West Eric Horvitz and Harsha Nori. 2025. JSONSchemaBench: A Rigorous Benchmark of Structured Outputs for Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.10868 (2025)."},{"key":"e_1_3_3_1_20_2","unstructured":"Google. 2025. Structured Outputs \u2014 Gemini API Documentation. Google AI for Developers. https:\/\/ai.google.dev\/gemini-api\/docs\/structured-output"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Rajarshi Haldar and Julia Hockenmaier. 2025. Rating Roulette: Self-Inconsistency in LLM-As-A-Judge Frameworks. arxiv:https:\/\/arXiv.org\/abs\/2510.27106\u00a0[cs.CL]","DOI":"10.18653\/v1\/2025.findings-emnlp.1361"},{"key":"e_1_3_3_1_22_2","unstructured":"Shibo Hao Sainbayar Sukhbaatar DiJia Su Xian Li Zhiting Hu Jason Weston and Yuandong Tian. 2024. Training Large Language Models to Reason in a Continuous Latent Space. arxiv:https:\/\/arXiv.org\/abs\/2412.06769\u00a0[cs.CL]"},{"key":"e_1_3_3_1_23_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Harada Keno","year":"2025","unstructured":"Keno Harada, Yudai Yamazaki, Masachika Taniguchi, Takeshi Kojima, Yusuke Iwasawa, and Yutaka Matsuo. 2025. Curse of Instructions: Large Language Models Cannot Follow Multiple Instructions at Once. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=R6q67CDBCH"},{"key":"e_1_3_3_1_24_2","unstructured":"Md.\u00a0Najib Hasan Md\u00a0Mahadi\u00a0Hassan Sibat Mohammad\u00a0Fakhruddin Babar Souvika Sarkar Monowar Hasan and Santu Karmaker. 2025. Pitfalls of Evaluating Language Models with Open Benchmarks. arxiv:https:\/\/arXiv.org\/abs\/2507.00460\u00a0[cs.CL]"},{"key":"e_1_3_3_1_25_2","volume-title":"Agentic Markets Workshop at ICML 2024","author":"Hu Qitian\u00a0Jason","unstructured":"Qitian\u00a0Jason Hu, Jacob Bieker, Xiuyu Li, Nan Jiang, Benjamin Keigwin, Gaurav Ranganath, Kurt Keutzer, and Shriyash\u00a0Kaustubh Upadhyay. [n. d.]. RouterBench: A Benchmark for Multi-LLM Routing System. In Agentic Markets Workshop at ICML 2024."},{"key":"e_1_3_3_1_26_2","volume-title":"The Twelfth International Conference on Learning Representations","author":"Kim Seungone","year":"2024","unstructured":"Seungone Kim, Jamin Shin, Yejin Cho, Joel Jang, Shayne Longpre, Hwaran Lee, Sangdoo Yun, Seongjin Shin, Sungdong Kim, James Thorne, and Minjoon Seo. 2024. Prometheus: Inducing Fine-Grained Evaluation Capability in Language Models. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=8euJaTveKw"},{"key":"e_1_3_3_1_27_2","unstructured":"Langfuse. 2024. Langfuse: Open Source LLM Engineering Platform. https:\/\/langfuse.com. Accessed: 2026-02-27."},{"key":"e_1_3_3_1_28_2","unstructured":"Qingquan Li Shaoyu Dou Kailai Shao Chao Chen and Haixiang Hu. 2025. Evaluating Scoring Bias in LLM-as-a-Judge. arxiv:https:\/\/arXiv.org\/abs\/2506.22316\u00a0[cs.CL]"},{"key":"e_1_3_3_1_29_2","unstructured":"Xunzhuo Liu Huamin Chen Samzong Lu Yossi Ovadia Guohong Wen Hao Wu Zhengda Tan Jintao Zhang Senan Zedan Yehudit Kerido Liav Weiss Haichen Zhang Bishen Yu Asaad Balum Noa Limoy Abdallah Samara Baofa Fan Brent Salisbury Ryan Cook Zhijie Wang Qiping Pan Rehan Khan Avishek Goswami Houston\u00a0H. Zhang Shuyi Wang Ziang Tang Fang Han Zohaib Hassan Jianqiao Zheng and Avinash Changrani. 2025. vLLM Semantic Router: Signal Driven Decision Routing for Mixture-of-Modality Models. arxiv:https:\/\/arXiv.org\/abs\/2603.04444\u00a0[cs.AI]"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.153"},{"key":"e_1_3_3_1_31_2","unstructured":"Isaac Ong Amjad Almahairi Vincent Wu Wei-Lin Chiang Tianhao Wu Joseph\u00a0E. Gonzalez M\u00a0Waleed Kadous and Ion Stoica. 2024. RouteLLM: Learning to Route LLMs with Preference Data. arxiv:https:\/\/arXiv.org\/abs\/2406.18665\u00a0[cs.LG]"},{"key":"e_1_3_3_1_32_2","unstructured":"OpenAI. 2024. Introducing Structured Outputs in the API. OpenAI Blog. https:\/\/openai.com\/index\/introducing-structured-outputs-in-the-api\/ Accessed: 2026-02-27."},{"key":"e_1_3_3_1_33_2","volume-title":"Advances in Neural Information Processing Systems","author":"Park Kanghee","year":"2024","unstructured":"Kanghee Park, Jiayu Wang, Taylor Berg-Kirkpatrick, Nadia Polikarpova, and Loris D\u2019Antoni. 2024. Grammar-Aligned Decoding. In Advances in Neural Information Processing Systems, Vol.\u00a037. arxiv:https:\/\/arXiv.org\/abs\/2405.21047\u00a0[cs.CL]"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.40"},{"key":"e_1_3_3_1_35_2","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing: Industry Track","author":"Tam Zhi\u00a0Rui","year":"2024","unstructured":"Zhi\u00a0Rui Tam, Cheng-Kuang Wu, Yi-Lin Tsai, Chieh-Yen Lin, Hung yi Lee, and Yun-Nung Chen. 2024. Let Me Speak Freely? A Study on the Impact of Format Restrictions on Performance of Large Language Models. In Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing: Industry Track. arxiv:https:\/\/arXiv.org\/abs\/2408.02442\u00a0[cs.CL]"},{"key":"e_1_3_3_1_36_2","unstructured":"TensorZero. 2025. TensorZero: Open-Source LLM Gateway with Observability and Optimization. https:\/\/www.tensorzero.com. Accessed: 2026-02-27."},{"key":"e_1_3_3_1_37_2","unstructured":"Clovis Varangot-Reille Christophe Bouvard Antoine Gourru Mathieu Ciancone Marion Schaeffer and Fran\u00e7ois Jacquenet. 2025. Doing More with Less: A Survey on Routing Strategies for Resource Optimisation in Large Language Model-Based Systems. arxiv:https:\/\/arXiv.org\/abs\/2502.00409\u00a0[cs.CL]"},{"key":"e_1_3_3_1_38_2","unstructured":"Jiaan Wang Yunlong Liang Fandong Meng Haoxiang Shi Zhixu Li Jinan Xu Jianfeng Qu and Jie Zhou. 2023. Is ChatGPT a Good NLG Evaluator? A Preliminary Study. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.04048 (2023)."},{"key":"e_1_3_3_1_39_2","first-page":"24824","volume-title":"Advances in Neural Information Processing Systems","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed Chi, Quoc\u00a0V. Le, and Denny Zhou. 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Advances in Neural Information Processing Systems, Vol.\u00a035. 24824\u201324837."},{"key":"e_1_3_3_1_40_2","unstructured":"Brandon\u00a0T. Willard and R\u00e9mi Louf. 2023. Efficient Guided Generation for Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2307.09702\u00a0[cs.CL]"},{"key":"e_1_3_3_1_41_2","unstructured":"Boming Xia Qinghua Lu Liming Zhu Zhenchang Xing Dehai Zhao and Hao Zhang. 2024. Evaluation-Driven Development and Operations of LLM Agents: A Process Model and Reference Architecture. arxiv:https:\/\/arXiv.org\/abs\/2411.13768\u00a0[cs.SE]"},{"key":"e_1_3_3_1_42_2","unstructured":"Shunyu Yao. 2025. The Second Half. https:\/\/ysymyth.github.io\/The-Second-Half\/. Accessed: 2026-02-26."},{"key":"e_1_3_3_1_43_2","unstructured":"Seonghyeon Ye Doyoung Kim Sungdong Kim Hyeonbin Hwang Seungone Kim Yongrae Jo James Thorne Juho Kim and Minjoon Seo. 2023. FLASK: Fine-grained Language Model Evaluation based on Alignment Skill Sets. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.10928 (2023)."},{"key":"e_1_3_3_1_44_2","volume-title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems","author":"Zhang Haozhen","unstructured":"Haozhen Zhang, Tao Feng, and Jiaxuan You. [n. d.]. Router-r1: Teaching llms multi-round routing and aggregation via reinforcement learning. In The Thirty-ninth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_3_1_45_2","doi-asserted-by":"crossref","unstructured":"Cen\u00a0Mia Zhao Tiantian Zhang Hanchen Su Yufeng\u00a0Wayne Zhang Shaowei Su Mingzhi Xu Yu\u00a0Elaine Liu Wei Han Jeremy Werner Claire\u00a0Na Cheng and Yashar Mehdad. 2025. Agent-in-the-Loop: A Data Flywheel for Continuous Improvement in LLM-based Customer Support. arxiv:https:\/\/arXiv.org\/abs\/2510.06674\u00a0[cs.CL]","DOI":"10.18653\/v1\/2025.emnlp-industry.135"},{"key":"e_1_3_3_1_46_2","volume-title":"Thirty-seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track","author":"Zheng Lianmin","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric Xing, et\u00a0al. [n. d.]. Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. In Thirty-seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_3_1_47_2","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric\u00a0P. Xing Hao Zhang Joseph\u00a0E. Gonzalez and Ion Stoica. 2023. Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena. arxiv:https:\/\/arXiv.org\/abs\/2306.05685\u00a0[cs.CL]"},{"key":"e_1_3_3_1_48_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.131"},{"key":"e_1_3_3_1_49_2","unstructured":"Lianghui Zhu Xinggang Wang and Xinlong Wang. 2023. JudgeLM: Fine-tuned Large Language Models are Scalable Judges. arxiv:https:\/\/arXiv.org\/abs\/2310.17631\u00a0[cs.CL]"}],"event":{"name":"CAIS '26: ACM Conference on AI and Agentic Systems","location":"San Jose CA USA","acronym":"CAIS '26"},"container-title":["Proceedings of the ACM Conference on AI and Agentic Systems"],"original-title":[],"deposited":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T03:18:48Z","timestamp":1779419928000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3786335.3813131"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,26]]},"references-count":48,"alternative-id":["10.1145\/3786335.3813131","10.1145\/3786335"],"URL":"https:\/\/doi.org\/10.1145\/3786335.3813131","relation":{},"subject":[],"published":{"date-parts":[[2026,5,26]]},"assertion":[{"value":"2026-05-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}