{"benchmark_id":"vqav2","benchmark_name":"VQAv2","benchmark_description":"VQAv2 is a balanced Visual Question Answering dataset that addresses language bias by providing complementary images for each question, forcing models to rely on visual understanding rather than language priors. It contains approximately twice the number of image-question pairs compared to the original VQA dataset.","max_score":1.0,"categories":["multimodal","reasoning","image_to_text","vision"],"modality":"multimodal","total_models":3,"entries":[{"rank":1,"model_id":"pixtral-large","model_name":"Pixtral Large","organization_name":"Mistral AI","organization_id":"mistral","benchmark_score":0.809,"normalized_score":0.809,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-11-18","announcement_date":"2024-11-18","multimodal":true,"param_count":124000000000,"is_new":false},{"rank":2,"model_id":"pixtral-12b-2409","model_name":"Pixtral-12B","organization_name":"Mistral AI","organization_id":"mistral","benchmark_score":0.786,"normalized_score":0.786,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-09-17","announcement_date":"2024-09-17","multimodal":true,"param_count":12400000000,"is_new":false},{"rank":3,"model_id":"llama-3.2-90b-instruct","model_name":"Llama 3.2 90B Instruct","organization_name":"Meta","organization_id":"meta","benchmark_score":0.781,"normalized_score":0.781,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-09-25","announcement_date":"2024-09-25","multimodal":true,"param_count":90000000000,"is_new":false}]}