{"benchmark_id":"omnibench","benchmark_name":"OmniBench","benchmark_description":"A novel multimodal benchmark designed to evaluate large language models' ability to recognize, interpret, and reason across visual, acoustic, and textual inputs simultaneously. Comprises 1,142 question-answer pairs covering 8 task categories from basic perception to complex inference, with a unique constraint that accurate responses require integrated understanding of all three modalities.","max_score":1.0,"categories":["multimodal","reasoning","vision"],"modality":"multimodal","total_models":1,"entries":[{"rank":1,"model_id":"qwen2.5-omni-7b","model_name":"Qwen2.5-Omni-7B","organization_name":"Alibaba Cloud / Qwen Team","organization_id":"qwen","benchmark_score":0.5613,"normalized_score":0.5613,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-03-27","announcement_date":"2025-03-27","multimodal":true,"param_count":7000000000,"is_new":false}]}