{"benchmark_id":"aime","benchmark_name":"AIME","benchmark_description":"American Invitational Mathematics Examination (AIME) benchmark for evaluating mathematical reasoning capabilities of large language models. Contains 30 challenging mathematical problems from AIME 2024 competition that require multi-step reasoning and advanced mathematical insight. Each problem has an integer answer between 000-999.","max_score":1.0,"categories":["math","reasoning"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"phi-4-mini-reasoning","model_name":"Phi 4 Mini Reasoning","organization_name":"Microsoft","organization_id":"microsoft","benchmark_score":0.575,"normalized_score":0.575,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-04-30","announcement_date":"2025-04-30","multimodal":false,"param_count":3800000000,"is_new":false},{"rank":2,"model_id":"mimo-v2.5-pro","model_name":"MiMo-V2.5-Pro","organization_name":"Xiaomi","organization_id":"xiaomi","benchmark_score":0.373,"normalized_score":0.373,"verified":false,"self_reported":true,"provider_id":"xiaomi","input_cost_per_million":0.435,"output_cost_per_million":0.87,"speed_rps":null,"context_window":1048576,"release_date":"2026-04-27","announcement_date":"2026-04-27","multimodal":false,"param_count":1023244718976,"is_new":false}]}