{"benchmark_id":"repobench","benchmark_name":"RepoBench","benchmark_description":"RepoBench is a benchmark for evaluating repository-level code auto-completion systems through three interconnected tasks: RepoBench-R (retrieval of relevant code snippets across files), RepoBench-C (code completion with cross-file and in-file context), and RepoBench-P (pipeline combining retrieval and prediction). Supports Python and Java programming languages and addresses the gap in evaluating real-world, multi-file programming scenarios by providing a more complete comparison of performance in auto-completion systems.","max_score":1.0,"categories":["reasoning","code"],"modality":"text","total_models":1,"entries":[{"rank":1,"model_id":"codestral-22b","model_name":"Codestral-22B","organization_name":"Mistral AI","organization_id":"mistral","benchmark_score":0.34,"normalized_score":0.34,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-05-29","announcement_date":"2024-05-29","multimodal":false,"param_count":22200000000,"is_new":false}]}