{"benchmark_id":"paperbench","benchmark_name":"PaperBench","benchmark_description":"PaperBench is a benchmark for evaluating AI agents on their ability to replicate research papers. It tests models on complex, multi-step workflows involving code implementation, experimentation, and reproducing scientific results from academic publications.","max_score":1.0,"categories":["reasoning","agents","code"],"modality":"text","total_models":3,"entries":[{"rank":1,"model_id":"qwen3.8-max","model_name":"Qwen3.8 Max","organization_name":"Alibaba Cloud / Qwen Team","organization_id":"qwen","benchmark_score":0.93,"normalized_score":0.93,"verified":false,"self_reported":true,"provider_id":"deepinfra","input_cost_per_million":1.65,"output_cost_per_million":4.951,"speed_rps":null,"context_window":256000,"release_date":"2026-08-02","announcement_date":"2026-08-02","multimodal":true,"param_count":2400000000000,"is_new":false},{"rank":2,"model_id":"kimi-k2.5","model_name":"Kimi K2.5","organization_name":"Moonshot AI","organization_id":"moonshotai","benchmark_score":0.635,"normalized_score":0.635,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2026-01-27","announcement_date":"2026-01-27","multimodal":true,"param_count":1000000000000,"is_new":false},{"rank":3,"model_id":"minimax-m3","model_name":"MiniMax M3","organization_name":"MiniMax","organization_id":"minimax","benchmark_score":0.526,"normalized_score":0.526,"verified":false,"self_reported":true,"provider_id":"novita","input_cost_per_million":0.3,"output_cost_per_million":1.2,"speed_rps":null,"context_window":1000000,"release_date":"2026-06-01","announcement_date":"2026-06-01","multimodal":true,"param_count":428000000000,"is_new":false}]}