{"benchmark_id":"automationbench-aa","benchmark_name":"AutomationBench-AA","benchmark_description":"AutomationBench-AA is Artificial Analysis's independently run version of AutomationBench, covering 657 real-world SaaS workflow tasks across 40 simulated applications (e.g. Gmail, Slack, Salesforce, HubSpot). It scores the share of objectives an agent completes without violating business guardrails, using a private held-out task set.","max_score":1.0,"categories":["reasoning","agents","tool_calling"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"grok-4.5","model_name":"Grok 4.5","organization_name":"xAI","organization_id":"xai","benchmark_score":0.514,"normalized_score":0.514,"verified":false,"self_reported":false,"provider_id":"xai","input_cost_per_million":2.0,"output_cost_per_million":6.0,"speed_rps":80.0,"context_window":500000,"release_date":"2026-07-16","announcement_date":"2026-07-16","multimodal":true,"param_count":null,"is_new":false}]}