{"benchmark_id":"musr","benchmark_name":"MuSR","benchmark_description":"MuSR (Multistep Soft Reasoning) is a benchmark for evaluating language models on multistep soft reasoning tasks specified in natural language narratives. Created through a neurosymbolic synthetic-to-natural generation algorithm, it generates complex reasoning scenarios like murder mysteries roughly 1000 words in length that challenge current LLMs including GPT-4. The benchmark tests chain-of-thought reasoning capabilities across domains involving commonsense reasoning about physical and social situations.","max_score":1.0,"categories":["reasoning"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"kimi-k2-instruct","model_name":"Kimi K2 Instruct","organization_name":"Moonshot AI","organization_id":"moonshotai","benchmark_score":0.764,"normalized_score":0.764,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-07-11","announcement_date":"2025-07-11","multimodal":false,"param_count":1000000000000,"is_new":false},{"rank":2,"model_id":"hermes-3-70b","model_name":"Hermes 3 70B","organization_name":"Nous Research","organization_id":"nous-research","benchmark_score":0.5067,"normalized_score":0.5067,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-08-15","announcement_date":"2024-08-15","multimodal":false,"param_count":70000000000,"is_new":false}]}