{"benchmark_id":"mmt-bench","benchmark_name":"MMT-Bench","benchmark_description":"MMT-Bench is a comprehensive multimodal benchmark for evaluating Large Vision-Language Models towards multitask AGI. It comprises 31,325 meticulously curated multi-choice visual questions from various multimodal scenarios such as vehicle driving and embodied navigation, covering 32 core meta-tasks and 162 subtasks in multimodal understanding.","max_score":1.0,"categories":["multimodal","reasoning","general","vision"],"modality":"multimodal","total_models":4,"entries":[{"rank":1,"model_id":"qwen2.5-vl-7b","model_name":"Qwen2.5 VL 7B Instruct","organization_name":"Alibaba Cloud / Qwen Team","organization_id":"qwen","benchmark_score":0.636,"normalized_score":0.636,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-01-26","announcement_date":"2025-01-26","multimodal":true,"param_count":8290000000,"is_new":false},{"rank":2,"model_id":"deepseek-vl2","model_name":"DeepSeek VL2","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.636,"normalized_score":0.636,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-12-13","announcement_date":"2024-12-13","multimodal":true,"param_count":27000000000,"is_new":false},{"rank":3,"model_id":"deepseek-vl2-small","model_name":"DeepSeek VL2 Small","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.629,"normalized_score":0.629,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-12-13","announcement_date":"2024-12-13","multimodal":true,"param_count":16000000000,"is_new":false},{"rank":4,"model_id":"deepseek-vl2-tiny","model_name":"DeepSeek VL2 Tiny","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.532,"normalized_score":0.532,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-12-13","announcement_date":"2024-12-13","multimodal":true,"param_count":3000000000,"is_new":false}]}