{"benchmark_id":"corpusqa-1m","benchmark_name":"CorpusQA 1M","benchmark_description":"CorpusQA 1M is a long-context question answering benchmark designed to evaluate models at approximately 1 million token contexts. Models are scored on accuracy when retrieving and reasoning over information distributed across an extremely long input corpus.","max_score":1.0,"categories":["long_context","reasoning","general"],"modality":"text","total_models":3,"entries":[{"rank":1,"model_id":"deepseek-v4-pro-max","model_name":"DeepSeek-V4-Pro-Max","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.62,"normalized_score":0.62,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2026-04-23","announcement_date":"2026-04-23","multimodal":false,"param_count":1600000000000,"is_new":false},{"rank":2,"model_id":"deepseek-v4-flash-max","model_name":"DeepSeek-V4-Flash-Max","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.605,"normalized_score":0.605,"verified":false,"self_reported":true,"provider_id":"deepseek","input_cost_per_million":0.14,"output_cost_per_million":0.28,"speed_rps":null,"context_window":1048576,"release_date":"2026-04-23","announcement_date":"2026-04-23","multimodal":false,"param_count":284000000000,"is_new":false},{"rank":3,"model_id":"deepseek-v4-flash-0423","model_name":"DeepSeek-V4-Flash-0423","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.593,"normalized_score":0.593,"verified":false,"self_reported":true,"provider_id":"deepinfra","input_cost_per_million":0.1,"output_cost_per_million":0.2,"speed_rps":null,"context_window":1048576,"release_date":"2026-04-23","announcement_date":"2026-04-23","multimodal":false,"param_count":284000000000,"is_new":false}]}