{"benchmark_id":"browsecomp-long-256k","benchmark_name":"BrowseComp Long Context 256k","benchmark_description":"BrowseComp is a benchmark for measuring the ability of agents to browse the web, comprising 1,266 questions that require persistently navigating the internet in search of hard-to-find, entangled information. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers. The benchmark focuses on questions where answers are obscure, time-invariant, and well-supported by evidence scattered across the open web.","max_score":1.0,"categories":["reasoning","search"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"gpt-5.2-2025-12-11","model_name":"GPT-5.2","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.898,"normalized_score":0.898,"verified":false,"self_reported":true,"provider_id":"openai","input_cost_per_million":1.75,"output_cost_per_million":14.0,"speed_rps":100.0,"context_window":400000,"release_date":"2025-12-11","announcement_date":"2025-12-11","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"gpt-5-2025-08-07","model_name":"GPT-5","organization_name":"OpenAI","organization_id":"openai","benchmark_score":0.888,"normalized_score":0.888,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-08-07","announcement_date":"2025-08-07","multimodal":true,"param_count":null,"is_new":false}]}