{"benchmark_id":"osworld-2.0","benchmark_name":"OSWorld 2.0","benchmark_description":"OSWorld 2.0 is a benchmark of 108 long-horizon, real-world computer-use workflows spanning everyday and professional tasks. Each task is an end-to-end workflow that takes human users a median of about 1.6 hours, scored with a binary-completion metric, and targets challenges such as dynamic environments, cross-source reasoning, and implicit-state inference.","max_score":1.0,"categories":["multimodal","general","agents","vision"],"modality":"multimodal","total_models":6,"entries":[{"rank":1,"model_id":"claude-opus-5","model_name":"Claude Opus 5","organization_name":"Anthropic","organization_id":"anthropic","benchmark_score":0.706,"normalized_score":0.706,"verified":false,"self_reported":true,"provider_id":"anthropic","input_cost_per_million":5.0,"output_cost_per_million":25.0,"speed_rps":null,"context_window":1000000,"release_date":"2026-07-24","announcement_date":"2026-07-24","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"gpt-5.6-sol","model_name":"GPT-5.6 Sol","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.626,"normalized_score":0.626,"verified":false,"self_reported":true,"provider_id":"openai","input_cost_per_million":5.0,"output_cost_per_million":30.0,"speed_rps":null,"context_window":1050000,"release_date":"2026-07-09","announcement_date":"2026-07-09","multimodal":true,"param_count":null,"is_new":false},{"rank":3,"model_id":"gpt-5.6-terra","model_name":"GPT-5.6 Terra","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.502,"normalized_score":0.502,"verified":false,"self_reported":true,"provider_id":"openai","input_cost_per_million":2.0,"output_cost_per_million":12.0,"speed_rps":null,"context_window":1050000,"release_date":"2026-07-09","announcement_date":"2026-07-09","multimodal":true,"param_count":null,"is_new":false},{"rank":4,"model_id":"gemini-3.7-flash","model_name":"Gemini 3.7 Flash","organization_name":"Google","organization_id":"google","benchmark_score":0.479,"normalized_score":0.479,"verified":false,"self_reported":true,"provider_id":"google","input_cost_per_million":0.75,"output_cost_per_million":3.75,"speed_rps":null,"context_window":1048576,"release_date":"2026-08-13","announcement_date":"2026-08-13","multimodal":true,"param_count":null,"is_new":false},{"rank":5,"model_id":"gpt-5.6-luna","model_name":"GPT-5.6 Luna","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.456,"normalized_score":0.456,"verified":false,"self_reported":true,"provider_id":"openai","input_cost_per_million":0.2,"output_cost_per_million":1.2,"speed_rps":null,"context_window":1050000,"release_date":"2026-07-09","announcement_date":"2026-07-09","multimodal":true,"param_count":null,"is_new":false},{"rank":6,"model_id":"qwen3.8-flash-next","model_name":"Qwen3.8-Flash-Next","organization_name":"Alibaba Cloud / Qwen Team","organization_id":"qwen","benchmark_score":0.194,"normalized_score":0.194,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2026-08-26","announcement_date":"2026-08-26","multimodal":true,"param_count":125000000000,"is_new":true}]}