{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T08:54:42Z","timestamp":1766566482343,"version":"3.48.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","funder":[{"name":"Fonds National de la Recherche - FNR","award":["NCER22\/IS\/16570468\/NCERFT"],"award-info":[{"award-number":["NCER22\/IS\/16570468\/NCERFT"]}]},{"name":"Fonds National de la Recherche - FNR","award":["BRIDGES2021\/IS\/16229163\/LuxemBERT"],"award-info":[{"award-number":["BRIDGES2021\/IS\/16229163\/LuxemBERT"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,17]]},"DOI":"10.1145\/3756681.3756975","type":"proceedings-article","created":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T08:30:04Z","timestamp":1766565004000},"page":"114-125","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["CallNavi, A challenge and empirical study on LLM function calling and routing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6314-7515","authenticated-orcid":false,"given":"Yewei","family":"Song","sequence":"first","affiliation":[{"name":"University of Luxembourg, Luxembourg, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2831-6472","authenticated-orcid":false,"given":"Xunzhu","family":"Tang","sequence":"additional","affiliation":[{"name":"University of Luxembourg, Luxembourg, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5372-7970","authenticated-orcid":false,"given":"Cedric","family":"Lothritz","sequence":"additional","affiliation":[{"name":"Luxembourg Institute of Science and Technology, Esch-sur-Alzette, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7657-4738","authenticated-orcid":false,"given":"Saad","family":"Ezzini","sequence":"additional","affiliation":[{"name":"King Fahad University of Petroleum and Minerals, Dharan, Saudi Arabia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4052-475X","authenticated-orcid":false,"given":"Jacques","family":"Klein","sequence":"additional","affiliation":[{"name":"University of Luxembourg, Luxembourg, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7270-9869","authenticated-orcid":false,"given":"Tegawend\u00e9","family":"Bissyande","sequence":"additional","affiliation":[{"name":"University of Luxembourg, Luxembourg, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1166-5908","authenticated-orcid":false,"given":"Andrey","family":"Boytsov","sequence":"additional","affiliation":[{"name":"BGL BNP PARIBAS, Luxembourg, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8805-8654","authenticated-orcid":false,"given":"Ulrick","family":"Ble","sequence":"additional","affiliation":[{"name":"BGL BNP PARIBAS, Luxembourg, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2734-7422","authenticated-orcid":false,"given":"Anne","family":"Goujon","sequence":"additional","affiliation":[{"name":"BGL BNP PARIBAS, Luxembourg, Luxembourg"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,24]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"Marah Abdin Sam\u00a0Ade Jacobs Ammar\u00a0Ahmad Awan Jyoti Aneja Ahmed Awadallah Hany Awadalla Nguyen Bach Amit Bahree Arash Bakhtiari Harkirat Behl et\u00a0al. 2024. Phi-3 technical report: A highly capable language model locally on your phone. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.14219 (2024)."},{"key":"e_1_3_3_3_3_2","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia\u00a0Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et\u00a0al. 2023. Gpt-4 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.08774 (2023)."},{"key":"e_1_3_3_3_4_2","unstructured":"Mistral Ai. 2024. AI in abundance. https:\/\/mistral.ai\/news\/september-24-release\/"},{"key":"e_1_3_3_3_5_2","unstructured":"Mistral Ai. 2024. Mistral NeMo. https:\/\/mistral.ai\/news\/mistral-nemo\/"},{"key":"e_1_3_3_3_6_2","unstructured":"Luca Beurer-Kellner Marc Fischer and Martin Vechev. 2024. Guiding LLMs The Right Way: Fast Non-Invasive Constrained Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.06988 (2024)."},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"crossref","unstructured":"Pawe\u0142 Budzianowski Tsung-Hsien Wen Bo-Hsiang Tseng Inigo Casanueva Stefan Ultes Osman Ramadan and Milica Ga\u0161i\u0107. 2018. Multiwoz\u2013a large-scale multi-domain wizard-of-oz dataset for task-oriented dialogue modelling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1810.00278 (2018).","DOI":"10.18653\/v1\/D18-1547"},{"key":"e_1_3_3_3_8_2","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique Ponde De\u00a0Oliveira Pinto Jared Kaplan Harri Edwards Yuri Burda Nicholas Joseph Greg Brockman et\u00a0al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.03374 (2021)."},{"key":"e_1_3_3_3_9_2","unstructured":"Cohere. [n. d.]. The Command R model \u2014 Cohere. https:\/\/docs.cohere.com\/v2\/docs\/command-r"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3597926.3598048"},{"key":"e_1_3_3_3_11_2","unstructured":"Mengnan Du Fengxiang He Na Zou Dacheng Tao and Xia Hu. 2023. Shortcut Learning of Large Language Models in Natural Language Understanding. arxiv:https:\/\/arXiv.org\/abs\/2208.11857\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2208.11857"},{"key":"e_1_3_3_3_12_2","unstructured":"Pawel Garbacki and Benny Chen. 2024. Firefunction-v2: Function calling capability on par with GPT4o at 2.5x the speed and 10% of the cost. https:\/\/fireworks.ai\/blog\/firefunction-v2-launch-post"},{"key":"e_1_3_3_3_13_2","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et\u00a0al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12948 (2025)."},{"key":"e_1_3_3_3_14_2","unstructured":"Zhicheng Guo Sijie Cheng Hao Wang Shihao Liang Yujia Qin Peng Li Zhiyuan Liu Maosong Sun and Yang Liu. 2024. StableToolBench: Towards Stable Large-Scale Benchmarking on Tool Learning of Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.07714 (2024)."},{"key":"e_1_3_3_3_15_2","unstructured":"Jinwei He and Feng Lu. 2024. CauseJudger: Identifying the Cause with LLMs for Abductive Logical Reasoning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.05559 (2024)."},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"Jonathan Herzig Pawe\u0142\u00a0Krzysztof Nowak Thomas M\u00fcller Francesco Piccinno and Julian\u00a0Martin Eisenschlos. 2020. TaPas: Weakly supervised table parsing via pre-training. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2004.02349 (2020).","DOI":"10.18653\/v1\/2020.acl-main.398"},{"key":"e_1_3_3_3_17_2","unstructured":"Sirui Hong Xiawu Zheng Jonathan Chen Yuheng Cheng Jinlin Wang Ceyao Zhang Zili Wang Steven Ka\u00a0Shing Yau Zijuan Lin Liyang Zhou et\u00a0al. 2023. Metagpt: Meta programming for multi-agent collaborative framework. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.00352 (2023)."},{"key":"e_1_3_3_3_18_2","unstructured":"Charlie Cheng-Jie Ji Huanzhi Mao Shishir G.\u00a0Patil Fanjia\u00a0Yan Tianjun Zhang Ion Stoica and Joseph\u00a0E. Gonzalez. 2024. Gorilla OpenFunctions v2. https:\/\/gorilla.cs.berkeley.edu\/\/blogs\/7_open_functions_v2.html."},{"key":"e_1_3_3_3_19_2","unstructured":"Minghao Li Yingxiu Zhao Bowen Yu Feifan Song Hangyu Li Haiyang Yu Zhoujun Li Fei Huang and Yongbin Li. 2023. Api-bank: A comprehensive benchmark for tool-augmented llms. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.08244 (2023)."},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"crossref","unstructured":"Yujia Li David Choi Junyoung Chung Nate Kushman Julian Schrittwieser R\u00e9mi Leblond Tom Eccles James Keeling Felix Gimeno Agustin Dal\u00a0Lago et\u00a0al. 2022. Competition-level code generation with alphacode. Science 378 6624 (2022) 1092\u20131097.","DOI":"10.1126\/science.abq1158"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3650756"},{"key":"e_1_3_3_3_22_2","unstructured":"Weiwen Liu Xu Huang Xingshan Zeng Xinlong Hao Shuai Yu Dexun Li Shuai Wang Weinan Gan Zhengying Liu Yuanqing Yu et\u00a0al. 2024. ToolACE: Winning the Points of LLM Function Calling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.00920 (2024)."},{"key":"e_1_3_3_3_23_2","unstructured":"Meta-AI. 2024. Introducing Llama 3.1: our most capable models to date. https:\/\/ai.meta.com\/blog\/meta-llama-3-1\/"},{"key":"e_1_3_3_3_24_2","unstructured":"Meta-AI. 2024. Llama 3.2: Revolutionizing edge AI and vision with open customizable models. https:\/\/ai.meta.com\/blog\/llama-3-2-connect-2024-vision-edge-mobile-devices\/"},{"key":"e_1_3_3_3_25_2","unstructured":"Nexusflow.ai. 2023. NexusRaven-V2: Surpassing GPT-4 for Zero-shot Function Calling. https:\/\/nexusflow.ai\/blogs\/ravenv2"},{"key":"e_1_3_3_3_26_2","unstructured":"Shishir\u00a0G Patil Tianjun Zhang Xin Wang and Joseph\u00a0E Gonzalez. 2023. Gorilla: Large language model connected with massive apis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.15334 (2023)."},{"key":"e_1_3_3_3_27_2","unstructured":"Jason Paul. 2024. NVIDIA announces first digital Human Technologies On-Device Small Language Model improving conversation for game characters | NVIDIA blog. https:\/\/blogs.nvidia.com\/blog\/digital-human-technology-mecha-break\/"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"crossref","unstructured":"Yun Peng Shuqing Li Wenwei Gu Yichen Li Wenxuan Wang Cuiyun Gao and Michael\u00a0R Lyu. 2022. Revisiting benchmarking and exploring API recommendation: How far are we? IEEE Transactions on Software Engineering 49 4 (2022) 1876\u20131897.","DOI":"10.1109\/TSE.2022.3197063"},{"key":"e_1_3_3_3_29_2","unstructured":"Yujia Qin Shihao Liang Yining Ye Kunlun Zhu Lan Yan Yaxi Lu Yankai Lin Xin Cong Xiangru Tang Bill Qian et\u00a0al. 2023. Toolllm: Facilitating large language models to master 16000+ real-world apis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.16789 (2023)."},{"key":"e_1_3_3_3_30_2","doi-asserted-by":"crossref","unstructured":"Maxim Rabinovich Mitchell Stern and Dan Klein. 2017. Abstract syntax networks for code generation and semantic parsing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1704.07535 (2017).","DOI":"10.18653\/v1\/P17-1105"},{"key":"e_1_3_3_3_31_2","unstructured":"Machel Reid Nikolay Savinov Denis Teplyashin Dmitry Lepikhin Timothy Lillicrap Jean-baptiste Alayrac Radu Soricut Angeliki Lazaridou Orhan Firat Julian Schrittwieser et\u00a0al. 2024. Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.05530 (2024)."},{"key":"e_1_3_3_3_32_2","unstructured":"Qiaoyu Tang Ziliang Deng Hongyu Lin Xianpei Han Qiao Liang Boxi Cao and Le Sun. 2023. Toolalpaca: Generalized tool learning for language models with 3000 simulated cases. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.05301 (2023)."},{"key":"e_1_3_3_3_33_2","unstructured":"Xunzhu Tang Kisub Kim Yewei Song Cedric Lothritz Bei Li Saad Ezzini Haoye Tian Jacques Klein and Tegawende\u00a0F. Bissyande. 2024. CodeAgent: Autonomous Communicative Agents for Code Review. arxiv:https:\/\/arXiv.org\/abs\/2402.02172\u00a0[cs.SE] https:\/\/arxiv.org\/abs\/2402.02172"},{"key":"e_1_3_3_3_34_2","unstructured":"Gemma Team Morgane Riviere Shreya Pathak Pier\u00a0Giuseppe Sessa Cassidy Hardin Surya Bhupatiraju L\u00e9onard Hussenot Thomas Mesnard Bobak Shahriari Alexandre Ram\u00e9 et\u00a0al. 2024. Gemma 2: Improving open language models at a practical size. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.00118 (2024)."},{"key":"e_1_3_3_3_35_2","unstructured":"Haoye Tian Weiqi Lu Tsz\u00a0On Li Xunzhu Tang Shing-Chi Cheung Jacques Klein and Tegawend\u00e9\u00a0F Bissyand\u00e9. 2023. Is ChatGPT the ultimate programming assistant\u2013how far is it? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.11938 (2023)."},{"key":"e_1_3_3_3_36_2","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et\u00a0al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09288 (2023)."},{"key":"e_1_3_3_3_37_2","unstructured":"Adina Trufinescu. 2024. Discover the new Multi-Lingual High-Quality PHI-3.5 SLMS. https:\/\/techcommunity.microsoft.com\/t5\/ai-azure-ai-services-blog\/discover-the-new-multi-lingual-high-quality-phi-3-5-slms\/ba-p\/4225280"},{"key":"e_1_3_3_3_38_2","unstructured":"Jason Wei Xuezhi Wang Dale Schuurmans Maarten Bosma Fei Xia Ed Chi Quoc\u00a0V Le Denny Zhou et\u00a0al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022) 24824\u201324837."},{"key":"e_1_3_3_3_39_2","unstructured":"Fanjia Yan Huanzhi Mao Charlie Cheng-Jie Ji Tianjun Zhang Shishir\u00a0G. Patil Ion Stoica and Joseph\u00a0E. Gonzalez. 2024. Berkeley Function Calling Leaderboard. https:\/\/gorilla.cs.berkeley.edu\/blogs\/8_berkeley_function_calling_leaderboard.html."},{"key":"e_1_3_3_3_40_2","unstructured":"Shunyu Yao Noah Shinn Pedram Razavi and Karthik Narasimhan. 2024. \u03c4 -bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.12045 (2024)."},{"key":"e_1_3_3_3_41_2","unstructured":"Junjie Ye Guanyu Li Songyang Gao Caishuang Huang Yilong Wu Sixian Li Xiaoran Fan Shihan Dou Qi Zhang Tao Gui et\u00a0al. 2024. Tooleyes: Fine-grained evaluation for tool learning capabilities of large language models in real-world scenarios. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.00741 (2024)."},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASE.2017.8115694"},{"key":"e_1_3_3_3_43_2","unstructured":"Jianguo Zhang Tian Lan Ming Zhu Zuxin Liu Thai Hoang Shirley Kokane Weiran Yao Juntao Tan Akshara Prabhakar Haolin Chen et\u00a0al. 2024. xLAM: A Family of Large Action Models to Empower AI Agent Systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.03215 (2024)."},{"key":"e_1_3_3_3_44_2","doi-asserted-by":"publisher","DOI":"10.1109\/IAEAC.2017.8054419"},{"key":"e_1_3_3_3_45_2","unstructured":"Yinger Zhang Hui Cai Xeirui Song Yicheng Chen Rui Sun and Jing Zheng. 2023. Reverse chain: A generic-rule for llms to master multi-api planning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.04474 (2023)."},{"key":"e_1_3_3_3_46_2","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zi Lin Zhuohan Li Dacheng Li Eric Xing et\u00a0al. 2023. Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems 36 (2023) 46595\u201346623."},{"key":"e_1_3_3_3_47_2","unstructured":"Terry\u00a0Yue Zhuo Minh\u00a0Chien Vu Jenny Chim Han Hu Wenhao Yu Ratnadira Widyasari Imam Nur\u00a0Bani Yusuf Haolan Zhan Junda He Indraneil Paul et\u00a0al. 2024. Bigcodebench: Benchmarking code generation with diverse function calls and complex instructions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.15877 (2024)."}],"event":{"name":"EASE '25: Evaluation and Assessment in Software Engineering","location":"Istanbul Turkiye","acronym":"EASE '25"},"container-title":["Proceedings of the 29th International Conference on Evaluation and Assessment in Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3756681.3756975","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,24]],"date-time":"2025-12-24T08:41:31Z","timestamp":1766565691000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3756681.3756975"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,17]]},"references-count":46,"alternative-id":["10.1145\/3756681.3756975","10.1145\/3756681"],"URL":"https:\/\/doi.org\/10.1145\/3756681.3756975","relation":{},"subject":[],"published":{"date-parts":[[2025,6,17]]},"assertion":[{"value":"2025-12-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}