{"benchmark_id":"kimi-claw-24-7-bench","benchmark_name":"Kimi Claw 24/7 Bench","benchmark_description":"Kimi Claw 24/7 Bench is Moonshot AI's in-house benchmark for evaluating long-horizon agentic performance in persistent, multi-day coworking tasks. It spans 17 professional scenarios across 610 evaluation points, covering software engineering, ML research, recruiting, trading, and marketing tasks executed through the OpenClaw harness.","max_score":1.0,"categories":["agents","code"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"kimi-k2.7-code","model_name":"Kimi K2.7 Code","organization_name":"Moonshot AI","organization_id":"moonshotai","benchmark_score":0.469,"normalized_score":0.469,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2026-06-12","announcement_date":"2026-06-12","multimodal":true,"param_count":1000000000000,"is_new":false}]}