{"benchmark_id":"mega-mlqa","benchmark_name":"MEGA MLQA","benchmark_description":"MLQA as part of the MEGA (Multilingual Evaluation of Generative AI) benchmark suite. A multi-way aligned extractive QA evaluation benchmark for cross-lingual question answering across 7 languages (English, Arabic, German, Spanish, Hindi, Vietnamese, and Simplified Chinese) with over 12K QA instances in English and 5K in each other language.","max_score":1.0,"categories":["language","reasoning"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"phi-3.5-moe-instruct","model_name":"Phi-3.5-MoE-instruct","organization_name":"Microsoft","organization_id":"microsoft","benchmark_score":0.653,"normalized_score":0.653,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-08-23","announcement_date":"2024-08-23","multimodal":false,"param_count":60000000000,"is_new":false},{"rank":2,"model_id":"phi-3.5-mini-instruct","model_name":"Phi-3.5-mini-instruct","organization_name":"Microsoft","organization_id":"microsoft","benchmark_score":0.617,"normalized_score":0.617,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-08-23","announcement_date":"2024-08-23","multimodal":false,"param_count":3800000000,"is_new":false}]}