{"benchmark_id":"mrcr-1m-(pointwise)","benchmark_name":"MRCR 1M (pointwise)","benchmark_description":"MRCR 1M (pointwise) is a variant of the Multi-Round Coreference Resolution benchmark that uses pointwise evaluation for ultra-long contexts (~1M tokens). This version evaluates each response independently rather than comparatively, testing models' absolute performance on long-context reasoning tasks.","max_score":1.0,"categories":["long_context","reasoning","general"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"gemini-2.5-pro","model_name":"Gemini 2.5 Pro","organization_name":"Google","organization_id":"google","benchmark_score":0.829,"normalized_score":0.829,"verified":false,"self_reported":true,"provider_id":"google","input_cost_per_million":1.25,"output_cost_per_million":10.0,"speed_rps":85.0,"context_window":1048576,"release_date":"2025-05-20","announcement_date":"2025-05-20","multimodal":true,"param_count":null,"is_new":false}]}