{"benchmark_id":"superglue","benchmark_name":"SuperGLUE","benchmark_description":"SuperGLUE is a new benchmark styled after GLUE with a new set of more difficult language understanding tasks, improved resources, and a new public leaderboard. It includes 8 primary tasks: BoolQ (Boolean Questions), CB (CommitmentBank), COPA (Choice of Plausible Alternatives), MultiRC (Multi-Sentence Reading Comprehension), ReCoRD (Reading Comprehension with Commonsense Reasoning), RTE (Recognizing Textual Entailment), WiC (Word-in-Context), and WSC (Winograd Schema Challenge). The benchmark evaluates diverse language understanding capabilities including reading comprehension, commonsense reasoning, causal reasoning, coreference resolution, textual entailment, and word sense disambiguation across multiple domains.","max_score":1.0,"categories":["language","reasoning","general"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"o1-mini","model_name":"o1-mini","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.75,"normalized_score":0.75,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-09-12","announcement_date":"2024-09-12","multimodal":false,"param_count":null,"is_new":false}]}