{"benchmark_id":"crag","benchmark_name":"CRAG","benchmark_description":"CRAG (Comprehensive RAG Benchmark) is a factual question answering benchmark consisting of 4,409 question-answer pairs across 5 domains (finance, sports, music, movie, open domain) and 8 question categories. The benchmark includes mock APIs to simulate web and Knowledge Graph search, designed to represent the diverse and dynamic nature of real-world QA tasks with temporal dynamism ranging from years to seconds. It evaluates retrieval-augmented generation systems for trustworthy question answering.","max_score":1.0,"categories":["reasoning","search","finance","economics"],"modality":"text","total_models":3,"entries":[{"rank":1,"model_id":"nova-pro","model_name":"Nova Pro","organization_name":"Amazon","organization_id":"amazon","benchmark_score":0.503,"normalized_score":0.503,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-11-20","announcement_date":"2024-11-20","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"nova-lite","model_name":"Nova Lite","organization_name":"Amazon","organization_id":"amazon","benchmark_score":0.438,"normalized_score":0.438,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-11-20","announcement_date":"2024-11-20","multimodal":true,"param_count":null,"is_new":false},{"rank":3,"model_id":"nova-micro","model_name":"Nova Micro","organization_name":"Amazon","organization_id":"amazon","benchmark_score":0.431,"normalized_score":0.431,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-11-20","announcement_date":"2024-11-20","multimodal":false,"param_count":null,"is_new":false}]}