{"benchmark_id":"bigcodebench-full","benchmark_name":"BigCodeBench-Full","benchmark_description":"A comprehensive benchmark that evaluates large language models' ability to solve complex, practical programming tasks via code generation. Contains 1,140 fine-grained tasks across 7 domains using function calls from 139 libraries. Challenges LLMs to invoke multiple function calls as tools and handle complex instructions for realistic software engineering and general-purpose reasoning tasks.","max_score":1.0,"categories":["reasoning","general"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"qwen-2.5-coder-32b-instruct","model_name":"Qwen2.5-Coder 32B Instruct","organization_name":"Alibaba Cloud / Qwen Team","organization_id":"qwen","benchmark_score":0.496,"normalized_score":0.496,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-09-19","announcement_date":"2024-09-19","multimodal":false,"param_count":32000000000,"is_new":false}]}