{"benchmark_id":"terminal-bench-hard","benchmark_name":"Terminal-Bench Hard","benchmark_description":"Terminal-Bench Hard is a harder terminal-agent benchmark variant evaluated with the Terminus-2 harness in Cohere's Command A+ and North Mini Code releases.","max_score":1.0,"categories":["reasoning","agents","code","tool_calling"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"north-mini-code-1.0","model_name":"North Mini Code 1.0","organization_name":"Cohere","organization_id":"cohere","benchmark_score":0.311,"normalized_score":0.311,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2026-06-09","announcement_date":"2026-06-09","multimodal":false,"param_count":30000000000,"is_new":false},{"rank":2,"model_id":"command-a-plus-05-2026","model_name":"Command A+","organization_name":"Cohere","organization_id":"cohere","benchmark_score":0.25,"normalized_score":0.25,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2026-05-20","announcement_date":"2026-05-20","multimodal":true,"param_count":218000000000,"is_new":false}]}