{"benchmark_id":"vibe-v2","benchmark_name":"VIBE-V2","benchmark_description":"VIBE-V2 is an internal benchmark covering pure front-end and full-stack Web, Android, and iOS projects with build-from-scratch tasks. It uses an Agent-as-a-Verifier paradigm to automatically verify program interaction logic and visual output, scoring models through a unified pipeline that includes a requirement set, containerized deployment, and a dynamic interaction environment.","max_score":1.0,"categories":["agents","code"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"minimax-m3","model_name":"MiniMax M3","organization_name":"MiniMax","organization_id":"minimax","benchmark_score":0.5012,"normalized_score":0.5012,"verified":false,"self_reported":true,"provider_id":"novita","input_cost_per_million":0.3,"output_cost_per_million":1.2,"speed_rps":null,"context_window":1000000,"release_date":"2026-06-01","announcement_date":"2026-06-01","multimodal":true,"param_count":428000000000,"is_new":false}]}