{"benchmark_id":"activitynet","benchmark_name":"ActivityNet","benchmark_description":"A large-scale video benchmark for human activity understanding. Provides samples from 203 activity classes with an average of 137 untrimmed videos per class and 1.41 activity instances per video, for a total of 849 video hours. The benchmark covers a wide range of complex human activities that are of interest to people in their daily living and can be used to compare algorithms for three scenarios: untrimmed video classification, trimmed activity classification, and activity detection.","max_score":1.0,"categories":["video","vision"],"modality":"video","total_models":1,"entries":[{"rank":1,"model_id":"gpt-4o-2024-08-06","model_name":"GPT-4o","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.619,"normalized_score":0.619,"verified":false,"self_reported":true,"provider_id":"openai","input_cost_per_million":2.5,"output_cost_per_million":10.0,"speed_rps":132.0,"context_window":128000,"release_date":"2024-08-06","announcement_date":"2024-08-06","multimodal":true,"param_count":null,"is_new":false}]}