{"benchmark_id":"cluewsc","benchmark_name":"CLUEWSC","benchmark_description":"CLUEWSC2020 is the Chinese version of the Winograd Schema Challenge, part of the CLUE benchmark. It focuses on pronoun disambiguation and coreference resolution, requiring models to determine which noun a pronoun refers to in a sentence. The dataset contains 1,244 training samples and 304 development samples extracted from contemporary Chinese literature.","max_score":1.0,"categories":["language","reasoning"],"modality":"text","total_models":3,"entries":[{"rank":1,"model_id":"kimi-k1.5","model_name":"Kimi-k1.5","organization_name":"Moonshot AI","organization_id":"moonshotai","benchmark_score":0.914,"normalized_score":0.914,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-01-20","announcement_date":"2025-01-20","multimodal":true,"param_count":null,"is_new":false},{"rank":2,"model_id":"deepseek-v3","model_name":"DeepSeek-V3","organization_name":"DeepSeek","organization_id":"deepseek","benchmark_score":0.909,"normalized_score":0.909,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2024-12-25","announcement_date":"2024-12-25","multimodal":false,"param_count":671000000000,"is_new":false},{"rank":3,"model_id":"ernie-4.5","model_name":"ERNIE 4.5","organization_name":"Baidu","organization_id":"baidu","benchmark_score":0.486,"normalized_score":0.486,"verified":false,"self_reported":true,"provider_id":null,"input_cost_per_million":null,"output_cost_per_million":null,"speed_rps":null,"context_window":null,"release_date":"2025-06-25","announcement_date":"2025-06-25","multimodal":false,"param_count":21000000000,"is_new":false}]}