{"benchmark_id":"arxivmath","benchmark_name":"ArXivMath","benchmark_description":"ArXivMath is a final-answer benchmark of research-level mathematics maintained by MathArena. Problems are extracted monthly from recent arXiv paper abstracts, then filtered through automated and manual checks to ensure they are self-contained, non-trivial, and verifiable. Because problems are drawn from active research, the benchmark is more realistic and more closely connected to mathematical research than contest or olympiad benchmarks.","max_score":1.0,"categories":["math","reasoning"],"modality":"text","total_models":2,"entries":[{"rank":1,"model_id":"claude-sonnet-5","model_name":"Claude Sonnet 5","organization_name":"Anthropic","organization_id":"anthropic","benchmark_score":0.722,"normalized_score":0.722,"verified":false,"self_reported":true,"provider_id":"anthropic","input_cost_per_million":2.0,"output_cost_per_million":10.0,"speed_rps":42.0,"context_window":1000000,"release_date":"2026-06-30","announcement_date":"2026-06-30","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"hy3","model_name":"Hy3","organization_name":"Tencent","organization_id":"tencent","benchmark_score":0.522,"normalized_score":0.522,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2026-07-06","announcement_date":"2026-07-06","multimodal":false,"param_count":295000000000,"is_new":false}]}