{"benchmark_id":"big-bench","benchmark_name":"BIG-Bench","benchmark_description":"Beyond the Imitation Game Benchmark (BIG-bench) is a collaborative benchmark consisting of 204+ tasks designed to probe large language models and extrapolate their future capabilities. It covers diverse domains including linguistics, mathematics, common-sense reasoning, biology, physics, social bias, software development, and more. The benchmark focuses on tasks believed to be beyond current language model capabilities and includes both English and non-English tasks across multiple languages.","max_score":1.0,"categories":["language","math","reasoning"],"modality":"text","total_models":3,"entries":[{"rank":1,"model_id":"gemini-1.0-pro","model_name":"Gemini 1.0 Pro","organization_name":"Google","organization_id":"google","benchmark_score":0.75,"normalized_score":0.75,"verified":false,"self_reported":false,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-02-15","announcement_date":"2024-02-15","multimodal":false,"param_count":null,"is_new":false},{"rank":2,"model_id":"gemma-2-27b-it","model_name":"Gemma 2 27B","organization_name":"Google","organization_id":"google","benchmark_score":0.749,"normalized_score":0.749,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-06-27","announcement_date":"2024-06-27","multimodal":false,"param_count":27200000000,"is_new":false},{"rank":3,"model_id":"gemma-2-9b-it","model_name":"Gemma 2 9B","organization_name":"Google","organization_id":"google","benchmark_score":0.682,"normalized_score":0.682,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-06-27","announcement_date":"2024-06-27","multimodal":false,"param_count":9240000000,"is_new":false}]}