{"benchmark_id":"video-mme-(long,-no-subtitles)","benchmark_name":"Video-MME (long, no subtitles)","benchmark_description":"Video-MME is the first-ever comprehensive evaluation benchmark for Multi-modal Large Language Models (MLLMs) in video analysis. This variant focuses on long-term videos (30min-60min) without subtitle inputs, testing robust contextual dynamics across 6 primary visual domains with 30 subfields including knowledge, film & television, sports competition, life record, and multilingual content.","max_score":1.0,"categories":["multimodal","video","vision"],"modality":"multimodal","total_models":1,"entries":[{"rank":1,"model_id":"gpt-4.1-2025-04-14","model_name":"GPT-4.1","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.72,"normalized_score":0.72,"verified":false,"self_reported":true,"provider_id":"openai","input_cost_per_million":2.0,"output_cost_per_million":8.0,"speed_rps":100.0,"context_window":1047576,"release_date":"2025-04-14","announcement_date":"2025-04-14","multimodal":true,"param_count":null,"is_new":false}]}