{"benchmark_id":"functionalmath","benchmark_name":"FunctionalMATH","benchmark_description":"A functional variant of the MATH benchmark that tests language models' ability to generalize reasoning patterns across different problem instances, revealing the reasoning gap between static and functional performance.","max_score":1.0,"categories":["math","reasoning"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"gemini-1.5-pro","model_name":"Gemini 1.5 Pro","organization_name":"Google","organization_id":"google","benchmark_score":0.646,"normalized_score":0.646,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-05-01","announcement_date":"2024-05-01","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"gemini-1.5-flash","model_name":"Gemini 1.5 Flash","organization_name":"Google","organization_id":"google","benchmark_score":0.536,"normalized_score":0.536,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-05-01","announcement_date":"2024-05-01","multimodal":true,"param_count":null,"is_new":false}]}