{"benchmark_id":"swe-bench-verified-(agentic-coding)","benchmark_name":"SWE-bench Verified (Agentic Coding)","benchmark_description":"SWE-bench Verified is a human-filtered subset of 500 software engineering problems drawn from real GitHub issues across 12 popular Python repositories. Given a codebase and an issue description, language models are tasked with generating patches that resolve the described problems. This benchmark evaluates AI's real-world agentic coding skills by requiring models to navigate complex codebases, understand software engineering problems, and coordinate changes across multiple functions, classes, and files to fix well-defined issues with clear descriptions.","max_score":1.0,"categories":["reasoning","code"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"claude-sonnet-4-5-20250929","model_name":"Claude Sonnet 4.5","organization_name":"Anthropic","organization_id":"anthropic","benchmark_score":0.772,"normalized_score":0.772,"verified":false,"self_reported":true,"provider_id":"anthropic","input_cost_per_million":3.0,"output_cost_per_million":15.0,"speed_rps":42.0,"context_window":200000,"release_date":"2025-09-29","announcement_date":"2025-09-29","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"kimi-k2-instruct","model_name":"Kimi K2 Instruct","organization_name":"Moonshot AI","organization_id":"moonshotai","benchmark_score":0.658,"normalized_score":0.658,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-07-11","announcement_date":"2025-07-11","multimodal":false,"param_count":1000000000000,"is_new":false}]}