{"benchmark_id":"if","benchmark_name":"IF","benchmark_description":"Instruction-Following Evaluation (IFEval) benchmark for large language models, focusing on verifiable instructions with 25 types of instructions and around 500 prompts containing one or more verifiable constraints","max_score":1.0,"categories":["structured_output","general"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"mistral-small-3.2-24b-instruct-2506","model_name":"Mistral Small 3.2 24B Instruct","organization_name":"Mistral AI","organization_id":"mistral","benchmark_score":0.8478,"normalized_score":0.8478,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-06-20","announcement_date":"2025-06-20","multimodal":true,"param_count":23600000000,"is_new":false},{"rank":2,"model_id":"minimax-m2","model_name":"MiniMax M2","organization_name":"MiniMax","organization_id":"minimax","benchmark_score":0.72,"normalized_score":0.72,"verified":false,"self_reported":true,"provider_id":"novita","input_cost_per_million":0.3,"output_cost_per_million":1.2,"speed_rps":null,"context_window":204800,"release_date":"2025-10-27","announcement_date":"2025-10-27","multimodal":false,"param_count":230000000000,"is_new":false}]}