{"benchmark_id":"osworld-screenshot-only","benchmark_name":"OSWorld Screenshot-only","benchmark_description":"OSWorld Screenshot-only: A variant of the OSWorld benchmark that evaluates multimodal AI agents using only screenshot observations to complete open-ended computer tasks across real operating systems (Ubuntu, Windows, macOS). Tests agents' ability to perform complex workflows involving web apps, desktop applications, file I/O, and multi-application tasks through visual interface understanding and GUI grounding.","max_score":1.0,"categories":["multimodal","general","grounding","agents","vision"],"modality":"multimodal","total_models":1,"entries":[{"rank":1,"model_id":"claude-3-5-sonnet-20241022","model_name":"Claude 3.5 Sonnet","organization_name":"Anthropic","organization_id":"anthropic","benchmark_score":0.149,"normalized_score":0.149,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-10-22","announcement_date":"2024-10-22","multimodal":true,"param_count":null,"is_new":false}]}