{"benchmark_id":"blueprint-bench-2","benchmark_name":"Blueprint-Bench 2","benchmark_description":"Blueprint-Bench 2 is an agentic spatial reasoning benchmark that evaluates a model's ability to understand, plan, and reason over architectural blueprints and other structured spatial documents. Scores are reported as a normalized score.","max_score":1.0,"categories":["multimodal","reasoning","agents"],"modality":"multimodal","total_models":2,"entries":[{"rank":1,"model_id":"claude-fable-5","model_name":"Claude Fable 5","organization_name":"Anthropic","organization_id":"anthropic","benchmark_score":0.386,"normalized_score":0.386,"verified":false,"self_reported":true,"provider_id":"anthropic","input_cost_per_million":10.0,"output_cost_per_million":50.0,"speed_rps":null,"context_window":1000000,"release_date":"2026-06-09","announcement_date":"2026-06-09","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"gemini-3.5-flash","model_name":"Gemini 3.5 Flash","organization_name":"Google","organization_id":"google","benchmark_score":0.336,"normalized_score":0.336,"verified":false,"self_reported":true,"provider_id":"google","input_cost_per_million":1.5,"output_cost_per_million":9.0,"speed_rps":null,"context_window":1048576,"release_date":"2026-05-19","announcement_date":"2026-05-19","multimodal":true,"param_count":null,"is_new":false}]}