{"benchmark_id":"terminal-bench-3.0","benchmark_name":"Terminal-Bench 3.0","benchmark_description":"Terminal-Bench 3.0 is a release of the Terminal-Bench benchmark that tests AI agents' ability to operate a computer via the terminal on real-world, end-to-end tasks.","max_score":1.0,"categories":["reasoning","agents","code","tool_calling"],"modality":"text","total_models":3,"entries":[{"rank":1,"model_id":"glm-5.3","model_name":"GLM-5.3","organization_name":"Zhipu AI","organization_id":"zai-org","benchmark_score":0.283,"normalized_score":0.283,"verified":false,"self_reported":true,"provider_id":"novita","input_cost_per_million":1.4,"output_cost_per_million":4.4,"speed_rps":null,"context_window":1048576,"release_date":"2026-08-14","announcement_date":"2026-08-14","multimodal":false,"param_count":753000000000,"is_new":false},{"rank":2,"model_id":"grok-4.6","model_name":"Grok 4.6","organization_name":"xAI","organization_id":"xai","benchmark_score":0.26,"normalized_score":0.26,"verified":false,"self_reported":true,"provider_id":"xai","input_cost_per_million":2.0,"output_cost_per_million":6.0,"speed_rps":null,"context_window":500000,"release_date":"2026-08-12","announcement_date":"2026-08-12","multimodal":true,"param_count":null,"is_new":false},{"rank":3,"model_id":"gemini-3.7-flash","model_name":"Gemini 3.7 Flash","organization_name":"Google","organization_id":"google","benchmark_score":0.149,"normalized_score":0.149,"verified":false,"self_reported":true,"provider_id":"google","input_cost_per_million":0.75,"output_cost_per_million":3.75,"speed_rps":null,"context_window":1048576,"release_date":"2026-08-13","announcement_date":"2026-08-13","multimodal":true,"param_count":null,"is_new":false}]}