{"benchmark_id":"openbookqa","benchmark_name":"OpenBookQA","benchmark_description":"OpenBookQA is a question-answering dataset modeled after open book exams for assessing human understanding. It contains 5,957 multiple-choice elementary-level science questions that probe understanding of 1,326 core science facts and their application to novel situations, requiring combination of open book facts with broad common knowledge through multi-hop reasoning.","max_score":1.0,"categories":["reasoning","general"],"modality":"text","total_models":5,"entries":[{"rank":1,"model_id":"phi-3.5-moe-instruct","model_name":"Phi-3.5-MoE-instruct","organization_name":"Microsoft","organization_id":"microsoft","benchmark_score":0.896,"normalized_score":0.896,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-08-23","announcement_date":"2024-08-23","multimodal":false,"param_count":60000000000,"is_new":false},{"rank":2,"model_id":"phi-4-mini","model_name":"Phi 4 Mini","organization_name":"Microsoft","organization_id":"microsoft","benchmark_score":0.792,"normalized_score":0.792,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-02-01","announcement_date":"2025-02-01","multimodal":false,"param_count":3840000000,"is_new":false},{"rank":3,"model_id":"phi-3.5-mini-instruct","model_name":"Phi-3.5-mini-instruct","organization_name":"Microsoft","organization_id":"microsoft","benchmark_score":0.792,"normalized_score":0.792,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-08-23","announcement_date":"2024-08-23","multimodal":false,"param_count":3800000000,"is_new":false},{"rank":4,"model_id":"mistral-nemo-instruct-2407","model_name":"Mistral NeMo Instruct","organization_name":"Mistral AI","organization_id":"mistral","benchmark_score":0.606,"normalized_score":0.606,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-07-18","announcement_date":"2024-07-18","multimodal":false,"param_count":12000000000,"is_new":false},{"rank":5,"model_id":"hermes-3-70b","model_name":"Hermes 3 70B","organization_name":"Nous Research","organization_id":"nous-research","benchmark_score":0.494,"normalized_score":0.494,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-08-15","announcement_date":"2024-08-15","multimodal":false,"param_count":70000000000,"is_new":false}]}