{"benchmark_id":"alpacaeval-2.0","benchmark_name":"AlpacaEval 2.0","benchmark_description":"AlpacaEval 2.0 is a length-controlled automatic evaluator for instruction-following language models that uses GPT-4 Turbo to assess model responses against a baseline. It evaluates models on 805 diverse instruction-following tasks including creative writing, classification, programming, and general knowledge questions. The benchmark achieves 0.98 Spearman correlation with ChatBot Arena while being fast (< 3 minutes) and affordable (< $10 in OpenAI credits). It addresses length bias in automatic evaluation through length-controlled win-rates and uses weighted scoring based on response quality.","max_score":1.0,"categories":["reasoning","general","creativity","writing"],"modality":"text","total_models":4,"entries":[{"rank":1,"model_id":"granite-3.3-8b-base","model_name":"Granite 3.3 8B Base","organization_name":"IBM","organization_id":"ibm","benchmark_score":0.6268,"normalized_score":0.6268,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-04-16","announcement_date":"2025-04-16","multimodal":true,"param_count":8170000000,"is_new":false},{"rank":2,"model_id":"granite-3.3-8b-instruct","model_name":"Granite 3.3 8B Instruct","organization_name":"IBM","organization_id":"ibm","benchmark_score":0.6268,"normalized_score":0.6268,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-04-16","announcement_date":"2025-04-16","multimodal":true,"param_count":8000000000,"is_new":false},{"rank":3,"model_id":"deepseek-v2.5","model_name":"DeepSeek-V2.5","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.505,"normalized_score":0.505,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-05-08","announcement_date":"2024-05-08","multimodal":false,"param_count":236000000000,"is_new":false},{"rank":4,"model_id":"granite-4.0-tiny-preview","model_name":"IBM Granite 4.0 Tiny Preview","organization_name":"IBM","organization_id":"ibm","benchmark_score":0.3516,"normalized_score":0.3516,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-05-02","announcement_date":"2025-05-02","multimodal":false,"param_count":7000000000,"is_new":false}]}