{"benchmark_id":"genebench","name":"GeneBench","parent_benchmark":null,"categories":["reasoning","science","agents"],"modality":"text","multilingual":false,"max_score":1.0,"language":"en","description":"GeneBench is an evaluation focused on multi-stage scientific data analysis in genetics and quantitative biology. Tasks require reasoning about ambiguous or noisy data with minimal supervisory guidance, addressing realistic obstacles such as hidden confounders or QC failures, and correctly implementing and interpreting modern statistical methods.","paper_link":"https://cdn.openai.com/pdf/6dc7175d-d9e7-4b8d-96b8-48fe5798cd5b/oai_genebench_benchmark.pdf","implementation_link":null,"verified":false,"created_at":"2026-05-07T16:53:23.434748+00:00","updated_at":"2026-09-02T17:08:04.324819+00:00","statistics":{"total_models":2,"average_score":0.29100000000000004,"min_score":0.25,"max_score":0.332,"score_stddev":0.05798275605729687,"verified_count":0,"self_reported_count":2},"child_benchmarks":[{"benchmark_id":"genebench-pro","name":"GeneBench-Pro","categories":["reasoning","science","agents"],"modality":"text","max_score":1.0,"description":"GeneBench-Pro is a research-level benchmark of 129 multi-stage computational-biology problems spanning genomics, quantitative biology, and translational biomedicine. Each problem gives the agent a messy dataset, brief context, and a target estimand, and requires navigating dependent inferential decision points to reach a verifiable answer."}],"linked_dataset":null,"models":[{"rank":1,"model_id":"gpt-5.5-pro","model_name":"GPT-5.5 Pro","organization_id":"openai","organization_name":"OpenAI","organization_country":"US","score":0.332,"normalized_score":0.332,"verified":false,"self_reported":true,"self_reported_source":"https://openai.com/index/introducing-gpt-5-5/","analysis_method":"GPT-5.5 Pro - GeneBench.","verification_date":null,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"context_window":null,"announcement_date":"2026-04-23","param_count":null,"is_open_source":false,"is_new":false,"best_latency":null,"latency_provider":null,"best_throughput":null,"throughput_provider":null,"context_provider":null},{"rank":2,"model_id":"gpt-5.5","model_name":"GPT-5.5","organization_id":"openai","organization_name":"OpenAI","organization_country":"US","score":0.25,"normalized_score":0.25,"verified":false,"self_reported":true,"self_reported_source":"https://openai.com/index/introducing-gpt-5-5/","analysis_method":"GeneBench. Reasoning effort xhigh.","verification_date":null,"provider_id":"openai","input_cost_per_million":5.0,"output_cost_per_million":30.0,"context_window":1050000,"announcement_date":"2026-04-23","param_count":null,"is_open_source":false,"is_new":false,"best_latency":null,"latency_provider":"OpenAI","best_throughput":null,"throughput_provider":"OpenAI","context_provider":"OpenAI"}]}