{"benchmark_id":"bfcl-v2","benchmark_name":"BFCL v2","benchmark_description":"Berkeley Function Calling Leaderboard (BFCL) v2 is a comprehensive benchmark for evaluating large language models' function calling capabilities. It features 2,251 question-function-answer pairs with enterprise and OSS-contributed functions, addressing data contamination and bias through live, user-contributed scenarios. The benchmark evaluates AST accuracy, executable accuracy, irrelevance detection, and relevance detection across multiple programming languages (Python, Java, JavaScript) and includes complex real-world function calling scenarios with multi-lingual prompts.","max_score":1.0,"categories":["reasoning","general","tool_calling"],"modality":"text","total_models":5,"entries":[{"rank":1,"model_id":"llama-3.3-70b-instruct","model_name":"Llama 3.3 70B Instruct","organization_name":"Meta","organization_id":"meta","benchmark_score":0.773,"normalized_score":0.773,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-12-06","announcement_date":"2024-12-06","multimodal":false,"param_count":70000000000,"is_new":false},{"rank":2,"model_id":"llama-3.1-nemotron-ultra-253b-v1","model_name":"Llama 3.1 Nemotron Ultra 253B v1","organization_name":"NVIDIA","organization_id":"nvidia","benchmark_score":0.741,"normalized_score":0.741,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-04-07","announcement_date":"2025-04-07","multimodal":false,"param_count":253000000000,"is_new":false},{"rank":3,"model_id":"llama-3.3-nemotron-super-49b-v1","model_name":"Llama-3.3 Nemotron Super 49B v1","organization_name":"NVIDIA","organization_id":"nvidia","benchmark_score":0.737,"normalized_score":0.737,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-03-18","announcement_date":"2025-03-18","multimodal":false,"param_count":49900000000,"is_new":false},{"rank":4,"model_id":"llama-3.2-3b-instruct","model_name":"Llama 3.2 3B Instruct","organization_name":"Meta","organization_id":"meta","benchmark_score":0.67,"normalized_score":0.67,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-09-25","announcement_date":"2024-09-25","multimodal":false,"param_count":3210000000,"is_new":false},{"rank":5,"model_id":"llama-3.1-nemotron-nano-8b-v1","model_name":"Llama 3.1 Nemotron Nano 8B V1","organization_name":"NVIDIA","organization_id":"nvidia","benchmark_score":0.636,"normalized_score":0.636,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-03-18","announcement_date":"2025-03-18","multimodal":false,"param_count":8000000000,"is_new":false}]}