{"benchmark_id":"ai2-reasoning-challenge-(arc)","benchmark_name":"AI2 Reasoning Challenge (ARC)","benchmark_description":"A dataset of 7,787 genuine grade-school level, multiple-choice science questions assembled to encourage research in advanced question-answering. The dataset is partitioned into a Challenge Set and Easy Set, where the Challenge Set contains only questions answered incorrectly by both retrieval-based and word co-occurrence algorithms. Covers multiple scientific domains including biology, physics, earth science, and chemistry, requiring scientific reasoning, causal understanding, and conceptual knowledge beyond simple fact retrieval. Includes a supporting corpus of over 14 million science sentences.","max_score":1.0,"categories":["reasoning","general"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"gpt-4-0613","model_name":"GPT-4","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.963,"normalized_score":0.963,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2023-06-13","announcement_date":"2023-06-13","multimodal":true,"param_count":null,"is_new":false}]}